From 127874796a411e121474018e4f4d3555808bac33 Mon Sep 17 00:00:00 2001 From: Guanghui Yang <3497809730@qq.com> Date: Sat, 8 Aug 2026 14:13:55 +0800 Subject: [PATCH 0001/1012] btrfs: free unlinked replace target on initialization failure btrfs_init_dev_replace_tgtdev() allocates the replacement target before looking up its dev_t and initializing its zoned device information. If either lookup_bdev() or btrfs_get_dev_zone_info() fails, the device has not been linked into fs_devices->devices yet, but the error path only drops the block device file reference. Free the allocated device on this error path to release its name, allocation state, zone info, and the device itself. The issue was found by a failure-path metadata residual analyzer and verified with targeted failure injection on v6.14. Assisted-by: Codex:gpt-5 Reviewed-by: Qu Wenruo Signed-off-by: Guanghui Yang <3497809730@qq.com> Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/dev-replace.c | 10 +++++++--- fs/btrfs/volumes.c | 2 +- fs/btrfs/volumes.h | 1 + 3 files changed, 9 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c index af1b898029e82a..22eea188a32808 100644 --- a/fs/btrfs/dev-replace.c +++ b/fs/btrfs/dev-replace.c @@ -235,7 +235,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, struct btrfs_device **device_out) { struct btrfs_fs_devices *fs_devices = fs_info->fs_devices; - struct btrfs_device *device; + struct btrfs_device *device = NULL; + struct btrfs_device *tmp_device; struct file *bdev_file; struct block_device *bdev; u64 devid = BTRFS_DEV_REPLACE_DEVID; @@ -264,8 +265,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, sync_blockdev(bdev); - list_for_each_entry(device, &fs_devices->devices, dev_list) { - if (device->bdev == bdev) { + list_for_each_entry(tmp_device, &fs_devices->devices, dev_list) { + if (tmp_device->bdev == bdev) { btrfs_err(fs_info, "target device is in the filesystem!"); ret = -EEXIST; @@ -285,6 +286,7 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, device = btrfs_alloc_device(NULL, &devid, NULL, device_path); if (IS_ERR(device)) { ret = PTR_ERR(device); + device = NULL; goto error; } @@ -328,6 +330,8 @@ static int btrfs_init_dev_replace_tgtdev(struct btrfs_fs_info *fs_info, error: /* Undo the open-time freeze deny. */ + if (device) + btrfs_free_device(device); btrfs_release_device_allow_freeze(bdev_file); return ret; } diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 74584669507fd7..7fb0bf742a2969 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -403,7 +403,7 @@ static struct btrfs_fs_devices *alloc_fs_devices(const u8 *fsid) return fs_devs; } -static void btrfs_free_device(struct btrfs_device *device) +void btrfs_free_device(struct btrfs_device *device) { WARN_ON(!list_empty(&device->post_commit_list)); /* diff --git a/fs/btrfs/volumes.h b/fs/btrfs/volumes.h index 0415d74cad9ba9..337d7007d9e225 100644 --- a/fs/btrfs/volumes.h +++ b/fs/btrfs/volumes.h @@ -799,6 +799,7 @@ void btrfs_rm_dev_replace_remove_srcdev(struct btrfs_device *srcdev); void btrfs_rm_dev_replace_free_srcdev(struct btrfs_device *srcdev); void btrfs_destroy_dev_replace_tgtdev(struct btrfs_device *tgtdev, bool allow_freeze); +void btrfs_free_device(struct btrfs_device *device); unsigned long btrfs_full_stripe_len(struct btrfs_fs_info *fs_info, u64 logical); u64 btrfs_calc_stripe_length(const struct btrfs_chunk_map *map); From dabbd4f4040542114ef796f5c2ee920f027f9ff1 Mon Sep 17 00:00:00 2001 From: Guanghui Yang <3497809730@qq.com> Date: Tue, 11 Aug 2026 09:02:27 +0930 Subject: [PATCH 0002/1012] btrfs: roll back sprout setup after device add failure btrfs_init_new_device() calls btrfs_setup_sprout() before creating the first writable chunks for a seed filesystem. That moves the seed devices out of fs_info->fs_devices, clears the seeding state and installs a new fsid for the sprout filesystem. If a later step fails, the error path removes the new device but leaves fs_info->fs_devices in the partially initialized sprout state. The mounted filesystem can then be left with no open devices after the failed device add. Add the inverse of btrfs_setup_sprout() and use it from the error path so the mounted seed filesystem is restored before the temporary seed_devices copy is released. Fixes: 2b82032c34ec ("Btrfs: Seed device support") Assisted-by: Codex:gpt-5 Reviewed-by: Qu Wenruo Signed-off-by: Guanghui Yang <3497809730@qq.com> [ Fix a conflict with per-profile available space, revert sprout before updating per-profile available space estimation. ] Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/volumes.c | 37 +++++++++++++++++++++++++++++++++++++ 1 file changed, 37 insertions(+) diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 7fb0bf742a2969..949e40baff3343 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -2783,6 +2783,41 @@ static void btrfs_setup_sprout(struct btrfs_fs_info *fs_info, btrfs_set_super_flags(disk_super, super_flags); } +static void btrfs_rollback_sprout(struct btrfs_fs_info *fs_info, + struct btrfs_fs_devices *seed_devices) +{ + struct btrfs_fs_devices *fs_devices = fs_info->fs_devices; + struct btrfs_super_block *disk_super = fs_info->super_copy; + struct btrfs_device *device; + u64 super_flags; + + lockdep_assert_held(&uuid_mutex); + lockdep_assert_held(&fs_devices->device_list_mutex); + + list_del_init(&seed_devices->seed_list); + list_splice_init_rcu(&seed_devices->devices, &fs_devices->devices, synchronize_rcu); + list_for_each_entry(device, &fs_devices->devices, dev_list) { + device->fs_devices = fs_devices; + } + + fs_devices->seeding = true; + fs_devices->num_devices = seed_devices->num_devices; + fs_devices->open_devices = seed_devices->open_devices; + fs_devices->missing_devices = seed_devices->missing_devices; + fs_devices->rotating = seed_devices->rotating; + fs_devices->latest_dev = seed_devices->latest_dev; + + memcpy(fs_devices->fsid, seed_devices->fsid, BTRFS_FSID_SIZE); + memcpy(fs_devices->metadata_uuid, seed_devices->metadata_uuid, BTRFS_FSID_SIZE); + memcpy(disk_super->fsid, seed_devices->fsid, BTRFS_FSID_SIZE); + + super_flags = (btrfs_super_flags(disk_super) | BTRFS_SUPER_FLAG_SEEDING); + btrfs_set_super_flags(disk_super, super_flags); + + seed_devices->opened = 0; + free_fs_devices(seed_devices); +} + /* * Store the expected generation for seed devices in device items. */ @@ -3134,6 +3169,8 @@ int btrfs_init_new_device(struct btrfs_fs_info *fs_info, const char *device_path orig_super_total_bytes); btrfs_set_super_num_devices(fs_info->super_copy, orig_super_num_devices); + if (seeding_dev) + btrfs_rollback_sprout(fs_info, seed_devices); btrfs_update_per_profile_avail(fs_info); mutex_unlock(&fs_info->chunk_mutex); mutex_unlock(&fs_info->fs_devices->device_list_mutex); From a750c00db1bedaea3c9cce510c44eec9b9efa6c6 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Mon, 17 Aug 2026 17:00:43 +0930 Subject: [PATCH 0003/1012] btrfs: refactor read_key_bytes() to remove the dest_folio parameter The function read_key_bytes() have 3 call sites: - For BTRFS_VERITY_DESC_ITEM_KEY offset 0 inside btrfs_get_verity_descriptor() - For BTRFS_VERITY_DESC_ITEM_KEY offset 1 inside btrfs_get_verity_descriptor() Those are to read the description items, which are pretty small with fixed item size. Those call sites do not utilize the @dest_folio parameter. - For btrfs_read_merkle_tree_page() This is to read the BTRFS_VERITY_MERKLE_ITEM_KEY, which can be pretty large and split into multiple items. This is the only call site utilizing the @dest_folio parameter. Just for the only btrfs_read_merkle_tree_page() call site, we have a complex scheme for @dest and @dest_folio parameters. Since @dest can be NULL, it means if we pass @dest as NULL, then no matter if @dest_folio is provided, the merkle data will not be loaded into that @dest_folio. This can lead to a bug where a highmem folio is not mapped, then we pass folio_address(folio), which is NULL, into read_key_bytes(), causing no data to be written into @dest_folio. To address the complex scheme between @dest and @dest_folio, remove the @dest_folio parameter completely, and let the only caller to map the folio and pass the mapped kernel address into read_key_bytes() instead. This not only reduces the parameter list, but also make it much clear on the @dest parameter handling. The only downside is a longer duration of locally mapped page, but this should still be fine, as kmap_local_folio() can survive context switch. Reported-by: Hongling Zeng Link: https://lore.kernel.org/linux-btrfs/20260817022012.19658-1-zenghongling@kylinos.cn/ Fixes: 146054090b08 ("btrfs: initial fsverity support") Reviewed-by: David Sterba Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/verity.c | 31 ++++++++++++++----------------- 1 file changed, 14 insertions(+), 17 deletions(-) diff --git a/fs/btrfs/verity.c b/fs/btrfs/verity.c index 4e0ab584227419..d432fec21c15bb 100644 --- a/fs/btrfs/verity.c +++ b/fs/btrfs/verity.c @@ -272,21 +272,17 @@ static int write_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, * @dest: Buffer to read into. This parameter has slightly tricky * semantics. If it is NULL, the function will not do any copying * and will just return the size of all the items up to len bytes. - * If dest_page is passed, then the function will kmap_local the - * page and ignore dest, but it must still be non-NULL to avoid the - * counting-only behavior. * @len: length in bytes to read - * @dest_folio: copy into this folio instead of the dest buffer * * Helper function to read items from the btree. This returns the number of * bytes read or < 0 for errors. We can return short reads if the items don't * exist on disk or aren't big enough to fill the desired length. Supports - * reading into a provided buffer (dest) or into the page cache + * reading into a provided buffer (dest). * * Returns number of bytes read or a negative error code on failure. */ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, - char *dest, u64 len, struct folio *dest_folio) + char *dest, u64 len) { BTRFS_PATH_AUTO_FREE(path); struct btrfs_root *root = inode->root; @@ -306,7 +302,11 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, if (!path) return -ENOMEM; - if (dest_folio) + /* + * Merkle items can be large and split across multiple items, so enable + * readahead for such cases. + */ + if (key_type == BTRFS_VERITY_MERKLE_ITEM_KEY) path->reada = READA_FORWARD; key.objectid = btrfs_ino(inode); @@ -350,7 +350,7 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, break; } - /* desc = NULL to just sum all the item lengths */ + /* dest == NULL to just sum all the item lengths */ if (!dest) copy_end = item_end; else @@ -363,16 +363,10 @@ static int read_key_bytes(struct btrfs_inode *inode, u8 key_type, u64 offset, copy_offset = offset - key.offset; if (dest) { - if (dest_folio) - kaddr = kmap_local_folio(dest_folio, 0); - data = btrfs_item_ptr(leaf, path->slots[0], void); read_extent_buffer(leaf, kaddr + dest_offset, (unsigned long)data + copy_offset, copy_bytes); - - if (dest_folio) - kunmap_local(kaddr); } offset += copy_bytes; @@ -663,7 +657,7 @@ int btrfs_get_verity_descriptor(struct inode *inode, void *buf, size_t buf_size) memset(&item, 0, sizeof(item)); ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_DESC_ITEM_KEY, 0, - (char *)&item, sizeof(item), NULL); + (char *)&item, sizeof(item)); if (ret < 0) return ret; @@ -680,7 +674,7 @@ int btrfs_get_verity_descriptor(struct inode *inode, void *buf, size_t buf_size) return -ERANGE; ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_DESC_ITEM_KEY, 1, - buf, buf_size, NULL); + buf, buf_size); if (ret < 0) return ret; if (ret != true_size) @@ -706,6 +700,7 @@ static struct page *btrfs_read_merkle_tree_page(struct inode *inode, struct folio *folio; u64 off = (u64)index << PAGE_SHIFT; loff_t merkle_pos = merkle_file_pos(inode); + void *kaddr; int ret; if (merkle_pos < 0) @@ -749,6 +744,7 @@ static struct page *btrfs_read_merkle_tree_page(struct inode *inode, } read_folio: + kaddr = kmap_local_folio(folio, 0); /* * Merkle item keys are indexed from byte 0 in the merkle tree. * They have the form: @@ -756,7 +752,8 @@ static struct page *btrfs_read_merkle_tree_page(struct inode *inode, * [ inode objectid, BTRFS_MERKLE_ITEM_KEY, offset in bytes ] */ ret = read_key_bytes(BTRFS_I(inode), BTRFS_VERITY_MERKLE_ITEM_KEY, off, - folio_address(folio), PAGE_SIZE, folio); + kaddr, PAGE_SIZE); + kunmap_local(kaddr); if (ret < 0) { folio_unlock(folio); folio_put(folio); From 4b0acd1d9dcf2ff0d214500d8a2a55e4be5cc662 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Wed, 19 Aug 2026 10:36:16 +0930 Subject: [PATCH 0004/1012] btrfs: replace btrfs_repair_io_failure() to use bio for page iteration Currently btrfs_repair_io_failure() uses a @paddrs[] array to iterate pages. Such a parameter is required for bs > ps cases, as one fs block crosses several pages. However there is a much simpler and existing way to iterate pages: bio and bvec_iter. This changes btrfs_repair_io_failure() by: - Use a const @bvec_iter pointer to locate where the pages are - Extract file offset/logical from the @bbio - Require no @step parameter Above features allow us to shorten the parameter list. - Rename the function to btrfs_repair_bbio_failure() - Change the caller in btrfs_repair_eb_io_failure() to allocate a bbio Unlike the data read path, we do not have a handy bbio in that case. So we need to allocate one just for btrfs_repair_bbio_failure(). - Change the error reporting in btrfs_repair_bbio_failure() to include root id and use inode number directly Now for btree inode we will report a proper inode number (1). Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/bio.c | 61 ++++++++++++++++++++++++++-------------------- fs/btrfs/bio.h | 5 ++-- fs/btrfs/disk-io.c | 25 +++++++++++++------ 3 files changed, 55 insertions(+), 36 deletions(-) diff --git a/fs/btrfs/bio.c b/fs/btrfs/bio.c index cc0bd03048bae6..f8d4c2d550073a 100644 --- a/fs/btrfs/bio.c +++ b/fs/btrfs/bio.c @@ -186,7 +186,6 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, */ struct bvec_iter saved_iter = repair_bbio->saved_iter; const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u64 logical = repair_bbio->saved_iter.bi_sector << SECTOR_SHIFT; const u32 nr_steps = repair_bbio->saved_iter.bi_size / step; int mirror = repair_bbio->mirror_num; phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; @@ -220,9 +219,8 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, do { mirror = prev_repair_mirror(fbio, mirror); - btrfs_repair_io_failure(fs_info, btrfs_ino(inode), - repair_bbio->file_offset, fs_info->sectorsize, - logical, paddrs, step, mirror); + btrfs_repair_bbio_failure(repair_bbio, &repair_bbio->saved_iter, + fs_info->sectorsize, mirror); } while (mirror != fbio->bbio->mirror_num); done: @@ -925,21 +923,23 @@ void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num) * The I/O is issued synchronously to block the repair read completion from * freeing the bio. * - * @ino: Offending inode number - * @fileoff: File offset inside the inode + * @bbio: Original bbio where the repair is needed + * @orig_iter: Points to where the repair start is * @length: Length of the repair write - * @logical: Logical address of the range - * @paddrs: Physical address array of the content - * @step: Length of for each paddrs * @mirror_num: Mirror number to write to. Must not be zero */ -int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, - u32 length, u64 logical, const phys_addr_t paddrs[], - unsigned int step, int mirror_num) +int btrfs_repair_bbio_failure(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, + u32 length, int mirror_num) { - const u32 nr_steps = DIV_ROUND_UP_POW2(length, step); + struct btrfs_inode *inode = bbio->inode; + struct btrfs_fs_info *fs_info = inode->root->fs_info; struct btrfs_io_stripe smap = { 0 }; - struct bio *bio = NULL; + struct bvec_iter iter = *orig_iter; + struct bio *repair_bio = NULL; + const u64 logical = iter.bi_sector << SECTOR_SHIFT; + const u64 fileoff = bbio->file_offset + + ((iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT); + u32 cur = 0; int ret = 0; BUG_ON(!mirror_num); @@ -950,8 +950,9 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, ASSERT(IS_ALIGNED(fileoff, fs_info->sectorsize)); /* Either it's a single data or metadata block. */ ASSERT(length <= BTRFS_MAX_BLOCKSIZE); - ASSERT(step <= length); - ASSERT(is_power_of_2(step)); + + /* Our current iter should not be before the original bbio saved_iter. */ + ASSERT(iter.bi_sector >= bbio->saved_iter.bi_sector); /* * The fs either mounted RO or hit critical errors, no need @@ -979,15 +980,22 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, goto out_counter_dec; } - bio = bio_alloc(smap.dev->bdev, nr_steps, REQ_OP_WRITE | REQ_SYNC, GFP_NOFS); - bio->bi_iter.bi_sector = smap.physical >> SECTOR_SHIFT; - for (int i = 0; i < nr_steps; i++) { - ret = bio_add_page(bio, phys_to_page(paddrs[i]), step, offset_in_page(paddrs[i])); - /* We should have allocated enough slots to contain all the different pages. */ - ASSERT(ret == step); + repair_bio = bio_alloc(smap.dev->bdev, max(1, length >> PAGE_SHIFT), + REQ_OP_WRITE | REQ_SYNC, GFP_NOFS); + repair_bio->bi_iter.bi_sector = smap.physical >> SECTOR_SHIFT; + while (cur < length) { + struct page *page = bio_iter_page(&bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), length - cur); + + ret = bio_add_page(repair_bio, page, cur_len, pg_off); + ASSERT(ret == cur_len); + bio_advance_iter_single(&bbio->bio, &iter, cur_len); + cur += cur_len; } - ret = submit_bio_wait(bio); - bio_put(bio); + + ret = submit_bio_wait(repair_bio); + bio_put(repair_bio); if (ret) { /* try to remap that extent elsewhere? */ btrfs_dev_stat_inc_and_print(smap.dev, BTRFS_DEV_STAT_WRITE_ERRS); @@ -995,8 +1003,9 @@ int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, } btrfs_info_rl(fs_info, - "read error corrected: ino %llu off %llu (dev %s sector %llu)", - ino, fileoff, btrfs_dev_name(smap.dev), + "read error corrected: root %llu ino %llu off %llu (dev %s sector %llu)", + btrfs_root_id(inode->root), btrfs_ino(inode), fileoff, + btrfs_dev_name(smap.dev), smap.physical >> SECTOR_SHIFT); ret = 0; diff --git a/fs/btrfs/bio.h b/fs/btrfs/bio.h index 303ed6c7103d92..b7bd377a016249 100644 --- a/fs/btrfs/bio.h +++ b/fs/btrfs/bio.h @@ -126,8 +126,7 @@ void btrfs_bio_end_io(struct btrfs_bio *bbio, blk_status_t status); void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num); void btrfs_submit_repair_write(struct btrfs_bio *bbio, int mirror_num, bool dev_replace); -int btrfs_repair_io_failure(struct btrfs_fs_info *fs_info, u64 ino, u64 fileoff, - u32 length, u64 logical, const phys_addr_t paddrs[], - unsigned int step, int mirror_num); +int btrfs_repair_bbio_failure(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, + u32 length, int mirror_num); #endif diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 819727460bcf4d..466fadb1815a81 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -176,19 +176,24 @@ static int btrfs_repair_eb_io_failure(const struct extent_buffer *eb, int mirror_num) { struct btrfs_fs_info *fs_info = eb->fs_info; - const u32 step = min(fs_info->nodesize, PAGE_SIZE); - const u32 nr_steps = eb->len / step; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; + struct btrfs_bio *bbio; + int ret; if (sb_rdonly(fs_info->sb)) return -EROFS; + /* + * This bbio is only to queue all pages for btrfs_repair_bbio_failure(). + * Thus it will never get its endio called. + */ + bbio = btrfs_bio_alloc(max(1, fs_info->nodesize >> PAGE_SHIFT), REQ_OP_READ, + BTRFS_I(fs_info->btree_inode), eb->start, NULL, NULL); + bbio->bio.bi_iter.bi_sector = eb->start >> SECTOR_SHIFT; for (int i = 0; i < num_extent_pages(eb); i++) { struct folio *folio = eb->folios[i]; /* No large folio support yet. */ ASSERT(folio_order(folio) == 0); - ASSERT(i < nr_steps); /* * For nodesize < page size, there is just one paddr, with some @@ -197,11 +202,17 @@ static int btrfs_repair_eb_io_failure(const struct extent_buffer *eb, * For nodesize >= page size, it's one or more paddrs, and eb->start * must be aligned to page boundary. */ - paddrs[i] = page_to_phys(&folio->page) + offset_in_page(eb->start); + ret = bio_add_page(&bbio->bio, &folio->page, min(PAGE_SIZE, fs_info->nodesize), + offset_in_page(eb->start)); + ASSERT(ret == min(PAGE_SIZE, fs_info->nodesize)); } + /* Since the bbio is never submitted, we have to save the iter manually. */ + bbio->saved_iter = bbio->bio.bi_iter; - return btrfs_repair_io_failure(fs_info, 0, eb->start, eb->len, - eb->start, paddrs, step, mirror_num); + ret = btrfs_repair_bbio_failure(bbio, &bbio->saved_iter, fs_info->nodesize, + mirror_num); + bio_put(&bbio->bio); + return ret; } /* From f40bd5da143df250a4d3357b781fc0d9aae625f3 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Wed, 19 Aug 2026 10:36:17 +0930 Subject: [PATCH 0005/1012] btrfs: enhance btrfs_data_csum_ok() to use bio for page iteration Currently btrfs_data_csum_ok() requires a @paddr[] array to iterate all possible pages for bs > ps cases. However for all btrfs_data_csum_ok() call sites, we already have a btrfs_bio, and the bio infrastructure has many flexible ways to iterate multiple pages already. Change btrfs_data_csum_ok() to make full use of btrfs_bio by: - Change the parameter list to require a @bvec_iter pointer And remove @bio_offset, which can be calculated through @bvec_iter and bbio->saved_iter. Also remove paddrs[], we will iterate all the pages using bio interfaces. - Make the same parameter changes to repair_one_sector() - Use bio interfaces to iterate pages from a bio - Rename the function to btrfs_bio_data_csum_ok() - Remove on-stack paddrs[] array usage Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/bio.c | 82 +++++++++++++++--------------------------- fs/btrfs/btrfs_inode.h | 4 +-- fs/btrfs/inode.c | 49 +++++++++++++++++++------ 3 files changed, 70 insertions(+), 65 deletions(-) diff --git a/fs/btrfs/bio.c b/fs/btrfs/bio.c index f8d4c2d550073a..19b4855969f536 100644 --- a/fs/btrfs/bio.c +++ b/fs/btrfs/bio.c @@ -180,29 +180,13 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, struct btrfs_failed_bio *fbio = repair_bbio->private; struct btrfs_inode *inode = repair_bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; - /* - * We can not move forward the saved_iter, as it will be later - * utilized by repair_bbio again. - */ - struct bvec_iter saved_iter = repair_bbio->saved_iter; - const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u32 nr_steps = repair_bbio->saved_iter.bi_size / step; int mirror = repair_bbio->mirror_num; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - phys_addr_t paddr; - unsigned int slot = 0; - /* Repair bbio should be eaxctly one block sized. */ + /* Repair bbio should be exactly one block sized. */ ASSERT(repair_bbio->saved_iter.bi_size == fs_info->sectorsize); - btrfs_bio_for_each_block(paddr, &repair_bbio->bio, &saved_iter, step) { - ASSERT(slot < nr_steps); - paddrs[slot] = paddr; - slot++; - } - if (repair_bbio->bio.bi_status || - !btrfs_data_csum_ok(repair_bbio, dev, 0, paddrs)) { + !btrfs_bio_data_csum_ok(repair_bbio, &repair_bbio->saved_iter, dev)) { bio_reset(&repair_bbio->bio, NULL, REQ_OP_READ); repair_bbio->bio.bi_iter = repair_bbio->saved_iter; @@ -236,25 +220,21 @@ static void btrfs_end_repair_bio(struct btrfs_bio *repair_bbio, * read succeeded to restore the redundancy. */ static struct btrfs_failed_bio *repair_one_sector(struct btrfs_bio *failed_bbio, - u32 bio_offset, - phys_addr_t paddrs[], + const struct bvec_iter *orig_iter, struct btrfs_failed_bio *fbio) { struct btrfs_inode *inode = failed_bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; - const u32 sectorsize = fs_info->sectorsize; - const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u32 nr_steps = sectorsize / step; - /* - * For bs > ps cases, the saved_iter can be partially moved forward. - * In that case we should round it down to the block boundary. - */ - const u64 logical = round_down(failed_bbio->saved_iter.bi_sector << SECTOR_SHIFT, - sectorsize); struct btrfs_bio *repair_bbio; struct bio *repair_bio; + struct bvec_iter iter = *orig_iter; + const u32 sectorsize = fs_info->sectorsize; + const u32 bio_offset = ((iter.bi_sector - failed_bbio->saved_iter.bi_sector) << + SECTOR_SHIFT); + const u64 logical = (iter.bi_sector << SECTOR_SHIFT); int num_copies; int mirror; + u32 cur = 0; btrfs_debug(fs_info, "repair read error: read error at %llu", failed_bbio->file_offset + bio_offset); @@ -275,17 +255,21 @@ static struct btrfs_failed_bio *repair_one_sector(struct btrfs_bio *failed_bbio, atomic_inc(&fbio->repair_count); - repair_bio = bio_alloc_bioset(NULL, nr_steps, REQ_OP_READ, GFP_NOFS, - &btrfs_repair_bioset); + repair_bio = bio_alloc_bioset(NULL, max(1, sectorsize >> PAGE_SHIFT), + REQ_OP_READ, GFP_NOFS, &btrfs_repair_bioset); repair_bio->bi_iter.bi_sector = logical >> SECTOR_SHIFT; - for (int i = 0; i < nr_steps; i++) { + while (cur < sectorsize) { + struct page *page = bio_iter_page(&failed_bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&failed_bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&failed_bbio->bio, iter), + sectorsize - cur); int ret; - ASSERT(offset_in_page(paddrs[i]) + step <= PAGE_SIZE); + ret = bio_add_page(repair_bio, page, cur_len, pg_off); + ASSERT(ret == cur_len); - ret = bio_add_page(repair_bio, phys_to_page(paddrs[i]), step, - offset_in_page(paddrs[i])); - ASSERT(ret == step); + bio_advance_iter_single(&failed_bbio->bio, &iter, cur_len); + cur += cur_len; } repair_bbio = btrfs_bio(repair_bio); @@ -303,18 +287,16 @@ static void btrfs_check_read_bio(struct btrfs_bio *bbio, struct btrfs_device *de struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; const u32 sectorsize = fs_info->sectorsize; - const u32 step = min(sectorsize, PAGE_SIZE); - const u32 nr_steps = sectorsize / step; - struct bvec_iter *iter = &bbio->saved_iter; + struct bvec_iter iter; blk_status_t status = bbio->bio.bi_status; struct btrfs_failed_bio *fbio = NULL; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - phys_addr_t paddr; - u32 offset = 0; /* Read-repair requires the inode field to be set by the submitter. */ ASSERT(inode); + /* The original bbio should be sectorsize aligned. */ + ASSERT(IS_ALIGNED(bbio->saved_iter.bi_size, sectorsize)); + /* * Hand off repair bios to the repair code as there is no upper level * submitter for them. @@ -327,16 +309,10 @@ static void btrfs_check_read_bio(struct btrfs_bio *bbio, struct btrfs_device *de /* Clear the I/O error. A failed repair will reset it. */ bbio->bio.bi_status = BLK_STS_OK; - btrfs_bio_for_each_block(paddr, &bbio->bio, iter, step) { - paddrs[(offset / step) % nr_steps] = paddr; - offset += step; - - if (IS_ALIGNED(offset, sectorsize)) { - if (status || - !btrfs_data_csum_ok(bbio, dev, offset - sectorsize, paddrs)) - fbio = repair_one_sector(bbio, offset - sectorsize, - paddrs, fbio); - } + for (iter = bbio->saved_iter; iter.bi_size; + bio_advance_iter(&bbio->bio, &iter, sectorsize)) { + if (status || !btrfs_bio_data_csum_ok(bbio, &iter, dev)) + fbio = repair_one_sector(bbio, &iter, fbio); } if (bbio->csum != bbio->csum_inline) kvfree(bbio->csum); @@ -924,7 +900,7 @@ void btrfs_submit_bbio(struct btrfs_bio *bbio, int mirror_num) * freeing the bio. * * @bbio: Original bbio where the repair is needed - * @orig_iter: Points to where the repair start is + * @orig_iter: Points to where the repair starts * @length: Length of the repair write * @mirror_num: Mirror number to write to. Must not be zero */ diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 1082fa92c1457a..171f96bdb8aa76 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -513,8 +513,8 @@ void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info, const phys_addr_t paddrs[], u8 *dest); int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum, const u8 * const csum_expected); -bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, - u32 bio_offset, const phys_addr_t paddrs[]); +bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, + struct btrfs_device *dev); noinline int can_nocow_extent(struct btrfs_inode *inode, u64 offset, u64 *len, struct btrfs_file_extent *file_extent, bool nowait); diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 93ef3cec191e36..005f8f9da8b139 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3536,27 +3536,31 @@ int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 * different noncontiguous pages. * * @bbio: btrfs_io_bio which contains the csum - * @dev: device the sector is on - * @bio_offset: offset to the beginning of the bio (in bytes) - * @paddrs: physical addresses which back the fs block + * @orig_iter: bvec iter pointing to the start of the block + * @dev: device the sector is on (optional) * * Check if the checksum on a data block is valid. When a checksum mismatch is * detected, report the error and fill the corrupted range with zero. * * Return %true if the sector is ok or had no checksum to start with, else %false. */ -bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, - u32 bio_offset, const phys_addr_t paddrs[]) +bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, + const struct bvec_iter *orig_iter, + struct btrfs_device *dev) { struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; + struct bvec_iter iter = *orig_iter; + struct btrfs_csum_ctx cctx; const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; + const u32 bio_offset = (iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT; u64 file_offset = bbio->file_offset + bio_offset; u64 end = file_offset + blocksize - 1; u8 *csum_expected; u8 csum[BTRFS_CSUM_SIZE]; + u32 cur = 0; + + ASSERT(iter.bi_sector >= bbio->saved_iter.bi_sector); if (!bbio->csum) return true; @@ -3572,7 +3576,22 @@ bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, csum_expected = bbio->csum + (bio_offset >> fs_info->sectorsize_bits) * fs_info->csum_size; - btrfs_calculate_block_csum_pages(fs_info, paddrs, csum); + btrfs_csum_init(&cctx, fs_info->csum_type); + while (cur < blocksize) { + struct page *page = bio_iter_page(&bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), blocksize - cur); + void *kaddr; + + kaddr = kmap_local_page(page) + pg_off; + btrfs_csum_update(&cctx, kaddr, cur_len); + kunmap_local(kaddr); + + bio_advance_iter_single(&bbio->bio, &iter, cur_len); + cur += cur_len; + } + btrfs_csum_final(&cctx, csum); + if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0)) goto zeroit; return true; @@ -3582,8 +3601,18 @@ bool btrfs_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_device *dev, bbio->mirror_num); if (dev) btrfs_dev_stat_inc_and_print(dev, BTRFS_DEV_STAT_CORRUPTION_ERRS); - for (int i = 0; i < nr_steps; i++) - memzero_page(phys_to_page(paddrs[i]), offset_in_page(paddrs[i]), step); + cur = 0; + iter = *orig_iter; + while (cur < blocksize) { + struct page *page = bio_iter_page(&bbio->bio, iter); + const u32 pg_off = bio_iter_offset(&bbio->bio, iter); + const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), blocksize - cur); + + memzero_page(page, pg_off, cur_len); + + bio_advance_iter_single(&bbio->bio, &iter, cur_len); + cur += cur_len; + } return false; } From d7fac2512b729b60fa17c254c15b58fe14d1903c Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Wed, 19 Aug 2026 10:36:18 +0930 Subject: [PATCH 0006/1012] btrfs: use a shared helper to calculate data checksum for a bio Since we are already calculating data checksum using bio interface, extract the generation part into btrfs_csum_one_bio_block(), and use that to replace the paddrs[] array based solution in csum_one_bio(). This will reduce 128 bytes on-stack memory usage for csum_one_bio(). Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/btrfs_inode.h | 2 ++ fs/btrfs/file-item.c | 19 +++++------------ fs/btrfs/inode.c | 46 +++++++++++++++++++++++++----------------- 3 files changed, 34 insertions(+), 33 deletions(-) diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 171f96bdb8aa76..e137a99151bd07 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -515,6 +515,8 @@ int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 const u8 * const csum_expected); bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, struct btrfs_device *dev); +void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, + const struct bvec_iter *orig_iter, u8 *csum); noinline int can_nocow_extent(struct btrfs_inode *inode, u64 offset, u64 *len, struct btrfs_file_extent *file_extent, bool nowait); diff --git a/fs/btrfs/file-item.c b/fs/btrfs/file-item.c index cf50fd623f41a8..581ca5653be93a 100644 --- a/fs/btrfs/file-item.c +++ b/fs/btrfs/file-item.c @@ -801,25 +801,16 @@ static void csum_one_bio(struct btrfs_bio *bbio, struct bvec_iter *src) { struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; - struct bio *bio = &bbio->bio; struct btrfs_ordered_sum *sums = bbio->sums; - struct bvec_iter iter = *src; - phys_addr_t paddr; + struct bvec_iter iter; const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - u32 offset = 0; int index = 0; - btrfs_bio_for_each_block(paddr, bio, &iter, step) { - paddrs[(offset / step) % nr_steps] = paddr; - offset += step; + for (iter = *src; iter.bi_size; bio_advance_iter(&bbio->bio, &iter, blocksize)) { + btrfs_csum_one_bio_block(fs_info, &bbio->bio, &iter, + sums->sums + index); - if (IS_ALIGNED(offset, blocksize)) { - btrfs_calculate_block_csum_pages(fs_info, paddrs, sums->sums + index); - index += fs_info->csum_size; - } + index += fs_info->csum_size; } } diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 005f8f9da8b139..11f8aad8601f50 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3531,6 +3531,32 @@ int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 return 0; } +/* Generate data checksum for a single fs block, pointed to by @orig_iter. */ +void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, + const struct bvec_iter *orig_iter, u8 *csum) +{ + struct btrfs_csum_ctx cctx; + struct bvec_iter iter = *orig_iter; + const u32 blocksize = fs_info->sectorsize; + u32 cur = 0; + + btrfs_csum_init(&cctx, fs_info->csum_type); + while (cur < blocksize) { + struct page *page = bio_iter_page(bio, iter); + const u32 pg_off = bio_iter_offset(bio, iter); + const u32 cur_len = min(bio_iter_len(bio, iter), blocksize - cur); + void *kaddr; + + kaddr = kmap_local_page(page) + pg_off; + btrfs_csum_update(&cctx, kaddr, cur_len); + kunmap_local(kaddr); + + bio_advance_iter_single(bio, &iter, cur_len); + cur += cur_len; + } + btrfs_csum_final(&cctx, csum); +} + /* * Verify the checksum of a single data sector, which can be scattered at * different noncontiguous pages. @@ -3551,7 +3577,6 @@ bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; struct bvec_iter iter = *orig_iter; - struct btrfs_csum_ctx cctx; const u32 blocksize = fs_info->sectorsize; const u32 bio_offset = (iter.bi_sector - bbio->saved_iter.bi_sector) << SECTOR_SHIFT; u64 file_offset = bbio->file_offset + bio_offset; @@ -3576,22 +3601,7 @@ bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, csum_expected = bbio->csum + (bio_offset >> fs_info->sectorsize_bits) * fs_info->csum_size; - btrfs_csum_init(&cctx, fs_info->csum_type); - while (cur < blocksize) { - struct page *page = bio_iter_page(&bbio->bio, iter); - const u32 pg_off = bio_iter_offset(&bbio->bio, iter); - const u32 cur_len = min(bio_iter_len(&bbio->bio, iter), blocksize - cur); - void *kaddr; - - kaddr = kmap_local_page(page) + pg_off; - btrfs_csum_update(&cctx, kaddr, cur_len); - kunmap_local(kaddr); - - bio_advance_iter_single(&bbio->bio, &iter, cur_len); - cur += cur_len; - } - btrfs_csum_final(&cctx, csum); - + btrfs_csum_one_bio_block(fs_info, &bbio->bio, orig_iter, csum); if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0)) goto zeroit; return true; @@ -3601,8 +3611,6 @@ bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, bbio->mirror_num); if (dev) btrfs_dev_stat_inc_and_print(dev, BTRFS_DEV_STAT_CORRUPTION_ERRS); - cur = 0; - iter = *orig_iter; while (cur < blocksize) { struct page *page = bio_iter_page(&bbio->bio, iter); const u32 pg_off = bio_iter_offset(&bbio->bio, iter); From 9eaab1841d5216cb6443b56a32f0dc828d9bab80 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Wed, 19 Aug 2026 10:36:19 +0930 Subject: [PATCH 0007/1012] btrfs: remove on-stack paddrs[] array usage Since the bs > ps support, we have to handle cases where a data block is inside several discontiguous pages. Thus we need a local paddrs[] array to assemble a data block for bs > ps cases. However to handle all possible bs/ps combinations, we have to declare such array using the max block size vs page size, no matter the current block size and page size. This adds 128 bytes on-stack memory usage for several call sites, and also introduced several duplicated helpers to calculate checksum for a data block: - btrfs_calculate_block_csum_folio() - btrfs_calculate_block_csum_pages() - btrfs_check_block_csum() The differences are mostly in how the data is passed. The first one accepts a contiguous paddr range. The second one accepts an array of paddrs[]. The last one is just a simple wrapper of the first one. However the most common interface to iterate a data block is through bio, and we have already converted most callers to use the bio based interface, e.g. btrfs_bio_data_csum_ok() and btrfs_csum_one_bio_block(). Convert the remaining two call sites to address the remaining paddrs[] usage: - btrfs_calculate_block_csum_pages() inside verify_bio_data_sectors() This can be switched to btrfs_csum_one_bio_block(). This removes the 128 bytes on-stack memory usage. - btrfs_calculate_block_csum_pages() inside verify_one_sector() This call site doesn't use on-stack memory for paddrs[], but reuses the existing btrfs_raid_bio::bio_paddrs[] or btrfs_raid_bio::stripe_paddrs[]. So implement a local version called calculate_block_csum_paddrs(). Now there is no fixed on-stack paddrs[] usage anymore. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/btrfs_inode.h | 6 ---- fs/btrfs/inode.c | 75 ------------------------------------------ fs/btrfs/raid56.c | 46 +++++++++++++++----------- 3 files changed, 27 insertions(+), 100 deletions(-) diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index e137a99151bd07..89e5e9c0c904f6 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -507,12 +507,6 @@ static inline void btrfs_set_inode_mapping_order(struct btrfs_inode *inode) inode->root->fs_info->block_max_order); } -void btrfs_calculate_block_csum_folio(struct btrfs_fs_info *fs_info, - const phys_addr_t paddr, u8 *dest); -void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info, - const phys_addr_t paddrs[], u8 *dest); -int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum, - const u8 * const csum_expected); bool btrfs_bio_data_csum_ok(struct btrfs_bio *bbio, const struct bvec_iter *orig_iter, struct btrfs_device *dev); void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 11f8aad8601f50..2967a307d8f066 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3456,81 +3456,6 @@ int btrfs_finish_ordered_io(struct btrfs_ordered_extent *ordered) return btrfs_finish_one_ordered(ordered); } -/* - * Calculate the checksum of an fs block at physical memory address @paddr, - * and save the result to @dest. - * - * The folio containing @paddr must be large enough to contain a full fs block. - */ -void btrfs_calculate_block_csum_folio(struct btrfs_fs_info *fs_info, - const phys_addr_t paddr, u8 *dest) -{ - struct folio *folio = page_folio(phys_to_page(paddr)); - const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - - /* The full block must be inside the folio. */ - ASSERT(offset_in_folio(folio, paddr) + blocksize <= folio_size(folio)); - - for (int i = 0; i < nr_steps; i++) { - u32 pindex = offset_in_folio(folio, paddr + i * step) >> PAGE_SHIFT; - - /* - * For bs <= ps cases, we will only run the loop once, so the offset - * inside the page will only added to paddrs[0]. - * - * For bs > ps cases, the block must be page aligned, thus offset - * inside the page will always be 0. - */ - paddrs[i] = page_to_phys(folio_page(folio, pindex)) + offset_in_page(paddr); - } - return btrfs_calculate_block_csum_pages(fs_info, paddrs, dest); -} - -/* - * Calculate the checksum of a fs block backed by multiple noncontiguous pages - * at @paddrs[] and save the result to @dest. - * - * The folio containing @paddr must be large enough to contain a full fs block. - */ -void btrfs_calculate_block_csum_pages(struct btrfs_fs_info *fs_info, - const phys_addr_t paddrs[], u8 *dest) -{ - const u32 blocksize = fs_info->sectorsize; - const u32 step = min(blocksize, PAGE_SIZE); - const u32 nr_steps = blocksize / step; - struct btrfs_csum_ctx csum; - - btrfs_csum_init(&csum, fs_info->csum_type); - for (int i = 0; i < nr_steps; i++) { - const phys_addr_t paddr = paddrs[i]; - void *kaddr; - - ASSERT(offset_in_page(paddr) + step <= PAGE_SIZE); - kaddr = kmap_local_page(phys_to_page(paddr)) + offset_in_page(paddr); - btrfs_csum_update(&csum, kaddr, step); - kunmap_local(kaddr); - } - btrfs_csum_final(&csum, dest); -} - -/* - * Verify the checksum for a single sector without any extra action that depend - * on the type of I/O. - * - * @kaddr must be a properly kmapped address. - */ -int btrfs_check_block_csum(struct btrfs_fs_info *fs_info, phys_addr_t paddr, u8 *csum, - const u8 * const csum_expected) -{ - btrfs_calculate_block_csum_folio(fs_info, paddr, csum); - if (unlikely(memcmp(csum, csum_expected, fs_info->csum_size) != 0)) - return -EIO; - return 0; -} - /* Generate data checksum for a single fs block, pointed to by @orig_iter. */ void btrfs_csum_one_bio_block(struct btrfs_fs_info *fs_info, struct bio *bio, const struct bvec_iter *orig_iter, u8 *csum) diff --git a/fs/btrfs/raid56.c b/fs/btrfs/raid56.c index 1ee52a9dcee36f..a5d0ef09d92abb 100644 --- a/fs/btrfs/raid56.c +++ b/fs/btrfs/raid56.c @@ -1652,12 +1652,7 @@ static void verify_bio_data_sectors(struct btrfs_raid_bio *rbio, struct bio *bio) { struct btrfs_fs_info *fs_info = rbio->bioc->fs_info; - const u32 step = min(fs_info->sectorsize, PAGE_SIZE); - const u32 nr_steps = rbio->sector_nsteps; int total_sector_nr = get_bio_sector_nr(rbio, bio); - u32 offset = 0; - phys_addr_t paddrs[BTRFS_MAX_BLOCKSIZE / PAGE_SIZE]; - phys_addr_t paddr; /* No data csum for the whole stripe, no need to verify. */ if (!rbio->csum_bitmap || !rbio->csum_buf) @@ -1667,28 +1662,20 @@ static void verify_bio_data_sectors(struct btrfs_raid_bio *rbio, if (total_sector_nr >= rbio->nr_data * rbio->stripe_nsectors) return; - btrfs_bio_for_each_block_all(paddr, bio, step) { + for (struct bvec_iter iter = init_bvec_iter_for_bio(bio); + iter.bi_size; + bio_advance_iter(bio, &iter, fs_info->sectorsize), total_sector_nr++) { u8 csum_buf[BTRFS_CSUM_SIZE]; u8 *expected_csum; - paddrs[(offset / step) % nr_steps] = paddr; - offset += step; - - /* Not yet covering the full fs block, continue to the next step. */ - if (!IS_ALIGNED(offset, fs_info->sectorsize)) - continue; - /* No csum for this sector, skip to the next sector. */ - if (!test_bit(total_sector_nr, rbio->csum_bitmap)) { - total_sector_nr++; + if (!test_bit(total_sector_nr, rbio->csum_bitmap)) continue; - } expected_csum = rbio->csum_buf + total_sector_nr * fs_info->csum_size; - btrfs_calculate_block_csum_pages(fs_info, paddrs, csum_buf); + btrfs_csum_one_bio_block(fs_info, bio, &iter, csum_buf); if (unlikely(memcmp(csum_buf, expected_csum, fs_info->csum_size) != 0)) set_bit(total_sector_nr, rbio->error_bitmap); - total_sector_nr++; } } @@ -1879,6 +1866,27 @@ void raid56_parity_write(struct bio *bio, struct btrfs_io_context *bioc) start_async_work(rbio, rmw_rbio_work); } +static void calculate_block_csum_paddrs(struct btrfs_fs_info *fs_info, + const phys_addr_t paddrs[], u8 *dest) +{ + const u32 blocksize = fs_info->sectorsize; + const u32 step = min(blocksize, PAGE_SIZE); + const u32 nr_steps = blocksize / step; + struct btrfs_csum_ctx csum; + + btrfs_csum_init(&csum, fs_info->csum_type); + for (int i = 0; i < nr_steps; i++) { + const phys_addr_t paddr = paddrs[i]; + void *kaddr; + + ASSERT(offset_in_page(paddr) + step <= PAGE_SIZE); + kaddr = kmap_local_page(phys_to_page(paddr)) + offset_in_page(paddr); + btrfs_csum_update(&csum, kaddr, step); + kunmap_local(kaddr); + } + btrfs_csum_final(&csum, dest); +} + static int verify_one_sector(struct btrfs_raid_bio *rbio, int stripe_nr, int sector_nr) { @@ -1906,7 +1914,7 @@ static int verify_one_sector(struct btrfs_raid_bio *rbio, csum_expected = rbio->csum_buf + (stripe_nr * rbio->stripe_nsectors + sector_nr) * fs_info->csum_size; - btrfs_calculate_block_csum_pages(fs_info, paddrs, csum_buf); + calculate_block_csum_paddrs(fs_info, paddrs, csum_buf); if (unlikely(memcmp(csum_buf, csum_expected, fs_info->csum_size) != 0)) return -EIO; return 0; From 86d5a2d854a8a60b301d7422fd0e4712d44f5c0e Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Mon, 24 Aug 2026 04:54:45 -0700 Subject: [PATCH 0008/1012] btrfs: skip extent tree lock in the shrinker for inodes without extent maps find_first_inode_to_shrink() takes inode->extent_tree.lock in write mode on every inode it walks, only to find out whether that inode has any extent maps. Most inodes have none, so the lock is taken and dropped again without any work being done. Check whether the tree is empty before taking the lock. tree->root is only modified with the tree lock held for write, so the unlocked read is a harmless race: a false empty just defers the inode to a later scan, which already happens whenever the write_trylock() below fails, and a false non-empty falls through to the existing check under the lock. Across the Meta production fleet the extent map shrinker is ~0.35% of non-idle kernel CPU. Attributing callees to their caller, find_first_inode_to_shrink() is ~65% of that, and the write_trylock() it does is ~30% of the whole shrinker. Micro benchmark: a 6 GiB btrfs on a loop device, 100000 empty files kept open, plus 200 1 MiB files created last so they get the highest inode numbers and every scan has to walk all the empty ones first. Each round drops the page cache, re-reads the data files to recreate the extent maps, then triggers the shrinker with "echo 2 > /proc/sys/vm/drop_caches". 15 rounds per run on ARM64 (Neoverse V2), 8 CPUs, no lock debugging. Cost of find_first_inode_to_shrink() from the ftrace function profiler, in ns per inode walked, median of runs: base patched delta idle 46.4 40.1 -13.6% 4 concurrent readers 47.8 38.4 -19.7% A separate build with CONFIG_LOCK_STAT, same test, for the extent map tree rwlock. The shrinker is not the only user of that lock, every extent map insert and lookup takes it too, which is why the acquisition count drops by two thirds rather than to nothing: base patched delta write acquisitions 628016 228000 -63.7% hold time total (us) 47512 22717 -52.2% acq cacheline bounces 1574 1288 -18.2% Signed-off-by: Breno Leitao Reviewed-by: Filipe Manana Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/extent_map.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/fs/btrfs/extent_map.c b/fs/btrfs/extent_map.c index 6ad7b39ae358b2..86d9c6f5ff4bd2 100644 --- a/fs/btrfs/extent_map.c +++ b/fs/btrfs/extent_map.c @@ -1219,6 +1219,14 @@ static struct btrfs_inode *find_first_inode_to_shrink(struct btrfs_root *root, tree = &inode->extent_tree; + /* + * Most inodes have no extent maps, so check without the lock. + * The race is harmless: a false empty just defers the inode to + * a later scan, and a false non-empty is caught under the lock. + */ + if (data_race(RB_EMPTY_ROOT(&tree->root))) + goto next; + /* * We want to be fast so if the lock is busy we don't want to * spend time waiting for it (some task is about to do IO for From 90bb57315c3949bebc0d793f25949b1391970ae4 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 25 Aug 2026 13:42:30 +0930 Subject: [PATCH 0009/1012] btrfs: remove unused variable flags from btrfs_read_qgroup_config() Since commit e562a8bdf652 ("btrfs: introduce BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN"), that @flags variable is no longer utilized. Just remove it. Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/qgroup.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index f68b696b4bf720..f3685cbb8f2e39 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -426,7 +426,6 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) struct extent_buffer *l; int slot; int ret = 0; - u64 flags = 0; u64 rescan_progress = 0; if (!fs_info->quota_root) @@ -609,7 +608,6 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) } out: btrfs_free_path(path); - fs_info->qgroup_flags |= flags; if (ret >= 0) { if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON) set_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags); From f5e60e8d7080d29739a4f056d5dbc28eafd4578e Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 25 Aug 2026 13:42:31 +0930 Subject: [PATCH 0010/1012] btrfs: qgroup: use atomic operations for btrfs_fs_info::qgroup_flags Currently we define btrfs_fs_info::qgroup_flags as u64, to match the on-disk qgroup status item's flag. But for now we have only 4 bits utilized for that flag, and since it's u64 we have no way to properly use the existing atomic bit operations (requires an unsigned long pointer). This results in a lot of non-atomic operations inside qgroup code. Some maybe fine as other locks are involved, but still it's not a good practice. Remove those non-atomic operations by: - Re-define btrfs_fs_info::qgroup_flags as unsigned long - Define BTRFS_QGROUP_STATUS_BIT_* and BTRFS_QGROUP_RUNTIME_BIT_* Instead of the old value define the bit number. - Use set_bit()/clear_bit()/test_bit() to replace open-coded bit operations - Add one extra check at qgroup status item read time To make sure the on-disk flag is still inside ULONG_MAX. Otherwise reject the status item and disable qgroup. - Get rid of unnecessary spinlock when checking a single bit Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/fs.h | 2 +- fs/btrfs/ioctl.c | 2 +- fs/btrfs/qgroup.c | 97 ++++++++++++++++----------------- fs/btrfs/qgroup.h | 4 +- fs/btrfs/sysfs.c | 8 +-- include/uapi/linux/btrfs_tree.h | 21 ++++--- 6 files changed, 67 insertions(+), 67 deletions(-) diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h index 10e15a319b93de..3eba8438593cde 100644 --- a/fs/btrfs/fs.h +++ b/fs/btrfs/fs.h @@ -811,7 +811,7 @@ struct btrfs_fs_info { struct btrfs_discard_ctl discard_ctl; /* Is qgroup tracking in a consistent state? */ - u64 qgroup_flags; + unsigned long qgroup_flags; /* Holds configuration and tracking. Protected by qgroup_lock. */ struct rb_root qgroup_tree; diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index e4b2da31a0d5de..54960351fbd171 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -3881,7 +3881,7 @@ static long btrfs_ioctl_quota_rescan_status(struct btrfs_fs_info *fs_info, if (!capable(CAP_SYS_ADMIN)) return -EPERM; - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { qsa.flags = 1; qsa.progress = fs_info->qgroup_rescan_progress.objectid; } diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index f3685cbb8f2e39..2c2ac0f16b1e5b 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -34,7 +34,7 @@ enum btrfs_qgroup_mode btrfs_qgroup_mode(const struct btrfs_fs_info *fs_info) { if (!test_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags)) return BTRFS_QGROUP_MODE_DISABLED; - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE) + if (test_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags)) return BTRFS_QGROUP_MODE_SIMPLE; return BTRFS_QGROUP_MODE_FULL; } @@ -384,14 +384,14 @@ static bool squota_check_parent_usage(struct btrfs_fs_info *fs_info, struct btrf __printf(2, 3) static void qgroup_mark_inconsistent(struct btrfs_fs_info *fs_info, const char *fmt, ...) { - const u64 old_flags = fs_info->qgroup_flags; + const unsigned long old_flags = fs_info->qgroup_flags; if (btrfs_qgroup_mode(fs_info) == BTRFS_QGROUP_MODE_SIMPLE) return; - fs_info->qgroup_flags |= (BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT | - BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN | - BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING); - if (!(old_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT)) { + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); + set_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags); + set_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags); + if (!test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &old_flags)) { struct va_format vaf; va_list args; @@ -472,8 +472,12 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) "old qgroup version, quota disabled"); goto out; } - fs_info->qgroup_flags = btrfs_qgroup_status_flags(l, ptr); - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE) + if (btrfs_qgroup_status_flags(l, ptr) > ULONG_MAX) { + btrfs_err(fs_info, "invalid qgroup status flags, quota disabled"); + goto out; + } + fs_info->qgroup_flags = (unsigned long)btrfs_qgroup_status_flags(l, ptr); + if (test_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags)) qgroup_read_enable_gen(fs_info, l, slot, ptr); else if (btrfs_qgroup_status_generation(l, ptr) != fs_info->generation) qgroup_mark_inconsistent(fs_info, "qgroup generation mismatch"); @@ -609,12 +613,12 @@ int btrfs_read_qgroup_config(struct btrfs_fs_info *fs_info) out: btrfs_free_path(path); if (ret >= 0) { - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON) + if (test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags)) set_bit(BTRFS_FS_QUOTA_ENABLED, &fs_info->flags); - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) ret = qgroup_rescan_init(fs_info, rescan_progress, 0); } else { - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN; + clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); btrfs_sysfs_del_qgroups(fs_info); } @@ -1099,9 +1103,9 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info, struct btrfs_qgroup_status_item); btrfs_set_qgroup_status_generation(leaf, ptr, trans->transid); btrfs_set_qgroup_status_version(leaf, ptr, BTRFS_QGROUP_STATUS_VERSION); - fs_info->qgroup_flags = BTRFS_QGROUP_STATUS_FLAG_ON; + fs_info->qgroup_flags = (1UL << BTRFS_QGROUP_STATUS_BIT_ON); if (simple) { - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE; + set_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); btrfs_set_fs_incompat(fs_info, SIMPLE_QUOTA); /* * Set the enable generation to the next transaction, as we cannot @@ -1111,7 +1115,7 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info, */ btrfs_set_qgroup_status_enable_gen(leaf, ptr, trans->transid + 1); } else { - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); } btrfs_set_qgroup_status_flags(leaf, ptr, fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAGS_MASK); @@ -1401,8 +1405,8 @@ int btrfs_quota_disable(struct btrfs_fs_info *fs_info) spin_lock(&fs_info->qgroup_lock); quota_root = fs_info->quota_root; fs_info->quota_root = NULL; - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_ON; - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE; + clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); fs_info->qgroup_drop_subtree_thres = BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT; spin_unlock(&fs_info->qgroup_lock); @@ -1552,7 +1556,7 @@ static int quick_update_accounting(struct btrfs_fs_info *fs_info, } out: if (ret) - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); return ret; } @@ -1873,7 +1877,7 @@ int btrfs_remove_qgroup(struct btrfs_trans_handle *trans, u64 qgroupid) * very frequently. */ if (btrfs_qgroup_mode(fs_info) == BTRFS_QGROUP_MODE_FULL && - !(fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT)) { + !test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags)) { if (unlikely(qgroup->rfer || qgroup->excl || qgroup->rfer_cmpr || qgroup->excl_cmpr)) { DEBUG_WARN(); @@ -2118,7 +2122,7 @@ int btrfs_qgroup_trace_extent_post(struct btrfs_trans_handle *trans, */ ASSERT(trans != NULL); - if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING) + if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags)) return 0; ret = btrfs_find_all_roots(&ctx, true); @@ -2959,7 +2963,7 @@ int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr, * we can't just exit here. */ if (!btrfs_qgroup_full_accounting(fs_info) || - fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING) + test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags)) goto out_free; if (new_roots) { @@ -2981,7 +2985,7 @@ int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr, num_bytes, nr_old_roots, nr_new_roots); mutex_lock(&fs_info->qgroup_rescan_lock); - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { if (fs_info->qgroup_rescan_progress.objectid <= bytenr) { mutex_unlock(&fs_info->qgroup_rescan_lock); ret = 0; @@ -3042,8 +3046,8 @@ int btrfs_qgroup_account_extents(struct btrfs_trans_handle *trans) num_dirty_extents++; trace_btrfs_qgroup_account_extents(fs_info, record, bytenr); - if (!ret && !(fs_info->qgroup_flags & - BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING)) { + if (!ret && !test_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, + &fs_info->qgroup_flags)) { struct btrfs_backref_walk_ctx ctx = { 0 }; ctx.bytenr = bytenr; @@ -3150,9 +3154,9 @@ int btrfs_run_qgroups(struct btrfs_trans_handle *trans) spin_lock(&fs_info->qgroup_lock); } if (btrfs_qgroup_enabled(fs_info)) - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_ON; + set_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); else - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_ON; + clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); spin_unlock(&fs_info->qgroup_lock); ret = update_qgroup_status_item(trans); @@ -3842,7 +3846,7 @@ static bool rescan_should_stop(struct btrfs_fs_info *fs_info) return true; if (!btrfs_qgroup_enabled(fs_info)) return true; - if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) + if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags)) return true; return false; } @@ -3892,12 +3896,10 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work) btrfs_free_path(path); mutex_lock(&fs_info->qgroup_rescan_lock); - if (ret > 0 && - fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT) { - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; - } else if (ret < 0 || stopped) { - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT; - } + if (ret > 0) + clear_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); + else if (ret < 0 || stopped) + set_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); mutex_unlock(&fs_info->qgroup_rescan_lock); /* @@ -3921,9 +3923,9 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work) } mutex_lock(&fs_info->qgroup_rescan_lock); - if (!stopped || - fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN; + if (!stopped || test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, + &fs_info->qgroup_flags)) + clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); if (trans) { int ret2 = update_qgroup_status_item(trans); @@ -3933,7 +3935,7 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work) } } fs_info->qgroup_rescan_running = false; - fs_info->qgroup_flags &= ~BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN; + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags); complete_all(&fs_info->qgroup_rescan_completion); mutex_unlock(&fs_info->qgroup_rescan_lock); @@ -3944,7 +3946,7 @@ static void btrfs_qgroup_rescan_worker(struct btrfs_work *work) if (stopped) { btrfs_info(fs_info, "qgroup scan paused"); - } else if (fs_info->qgroup_flags & BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN) { + } else if (test_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags)) { btrfs_info(fs_info, "qgroup scan cancelled"); } else if (ret >= 0) { btrfs_info(fs_info, "qgroup scan completed%s", @@ -3971,13 +3973,11 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, if (!init_flags) { /* we're resuming qgroup rescan at mount time */ - if (!(fs_info->qgroup_flags & - BTRFS_QGROUP_STATUS_FLAG_RESCAN)) { + if (!(test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags))) { btrfs_debug(fs_info, "qgroup rescan init failed, qgroup rescan is not queued"); ret = -EINVAL; - } else if (!(fs_info->qgroup_flags & - BTRFS_QGROUP_STATUS_FLAG_ON)) { + } else if (!(test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags))) { btrfs_debug(fs_info, "qgroup rescan init failed, qgroup is not enabled"); ret = -ENOTCONN; @@ -3990,10 +3990,9 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, mutex_lock(&fs_info->qgroup_rescan_lock); if (init_flags) { - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { ret = -EINPROGRESS; - } else if (!(fs_info->qgroup_flags & - BTRFS_QGROUP_STATUS_FLAG_ON)) { + } else if (!test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags)) { btrfs_debug(fs_info, "qgroup rescan init failed, qgroup is not enabled"); ret = -ENOTCONN; @@ -4006,13 +4005,13 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, mutex_unlock(&fs_info->qgroup_rescan_lock); return ret; } - fs_info->qgroup_flags |= BTRFS_QGROUP_STATUS_FLAG_RESCAN; + set_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); } memset(&fs_info->qgroup_rescan_progress, 0, sizeof(fs_info->qgroup_rescan_progress)); - fs_info->qgroup_flags &= ~(BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN | - BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags); fs_info->qgroup_rescan_progress.objectid = progress_objectid; init_completion(&fs_info->qgroup_rescan_completion); mutex_unlock(&fs_info->qgroup_rescan_lock); @@ -4063,7 +4062,7 @@ btrfs_qgroup_rescan(struct btrfs_fs_info *fs_info) ret = btrfs_commit_current_transaction(fs_info->fs_root); if (ret) { - fs_info->qgroup_flags &= ~BTRFS_QGROUP_STATUS_FLAG_RESCAN; + clear_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags); return ret; } @@ -4116,7 +4115,7 @@ int btrfs_qgroup_wait_for_completion(struct btrfs_fs_info *fs_info, void btrfs_qgroup_rescan_resume(struct btrfs_fs_info *fs_info) { - if (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_RESCAN) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { mutex_lock(&fs_info->qgroup_rescan_lock); fs_info->qgroup_rescan_running = true; btrfs_queue_work(fs_info->qgroup_rescan_workers, diff --git a/fs/btrfs/qgroup.h b/fs/btrfs/qgroup.h index 80dd2dacd56db4..b3aaad5e617d51 100644 --- a/fs/btrfs/qgroup.h +++ b/fs/btrfs/qgroup.h @@ -121,8 +121,8 @@ struct btrfs_qgroup_swapped_blocks; * To minimize the chance of collision with new persisted status flags, these * count backwards from the MSB. */ -#define BTRFS_QGROUP_RUNTIME_FLAG_CANCEL_RESCAN (1ULL << 63) -#define BTRFS_QGROUP_RUNTIME_FLAG_NO_ACCOUNTING (1ULL << 62) +#define BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN (BITS_PER_LONG - 1) +#define BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING (BITS_PER_LONG - 2) #define BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT (3) diff --git a/fs/btrfs/sysfs.c b/fs/btrfs/sysfs.c index 39cb01ee441ab8..1df6340a71234f 100644 --- a/fs/btrfs/sysfs.c +++ b/fs/btrfs/sysfs.c @@ -2359,9 +2359,7 @@ static ssize_t qgroup_enabled_show(struct kobject *qgroups_kobj, struct btrfs_fs_info *fs_info = to_fs_info(qgroups_kobj->parent); bool enabled; - spin_lock(&fs_info->qgroup_lock); - enabled = fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_ON; - spin_unlock(&fs_info->qgroup_lock); + enabled = test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); return sysfs_emit(buf, "%d\n", enabled); } @@ -2401,9 +2399,7 @@ static ssize_t qgroup_inconsistent_show(struct kobject *qgroups_kobj, struct btrfs_fs_info *fs_info = to_fs_info(qgroups_kobj->parent); bool inconsistent; - spin_lock(&fs_info->qgroup_lock); - inconsistent = (fs_info->qgroup_flags & BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT); - spin_unlock(&fs_info->qgroup_lock); + inconsistent = test_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); return sysfs_emit(buf, "%d\n", inconsistent); } diff --git a/include/uapi/linux/btrfs_tree.h b/include/uapi/linux/btrfs_tree.h index cc3b9f7dccafa2..b6ccaf848e4b3e 100644 --- a/include/uapi/linux/btrfs_tree.h +++ b/include/uapi/linux/btrfs_tree.h @@ -1255,13 +1255,16 @@ static inline __u16 btrfs_qgroup_level(__u64 qgroupid) } /* - * is subvolume quota turned on? - */ -#define BTRFS_QGROUP_STATUS_FLAG_ON (1ULL << 0) -/* - * RESCAN is set during the initialization phase + * The following BTRFS_QGROUP_STATUS_BIT_* are for * btrfs_qgroup_status_item::flags. + * + * Is subvolume quota turned on? */ -#define BTRFS_QGROUP_STATUS_FLAG_RESCAN (1ULL << 1) +#define BTRFS_QGROUP_STATUS_BIT_ON (0) +#define BTRFS_QGROUP_STATUS_FLAG_ON (1UL << BTRFS_QGROUP_STATUS_BIT_ON) + +/* RESCAN is set during the initialization phase */ +#define BTRFS_QGROUP_STATUS_BIT_RESCAN (1) +#define BTRFS_QGROUP_STATUS_FLAG_RESCAN (1UL << BTRFS_QGROUP_STATUS_BIT_RESCAN) /* * Some qgroup entries are known to be out of date, * either because the configuration has changed in a way that @@ -1269,14 +1272,16 @@ static inline __u16 btrfs_qgroup_level(__u64 qgroupid) * with a non-qgroup-aware version. * Turning qouta off and on again makes it inconsistent, too. */ -#define BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT (1ULL << 2) +#define BTRFS_QGROUP_STATUS_BIT_INCONSISTENT (2) +#define BTRFS_QGROUP_STATUS_FLAG_INCONSISTENT (1UL << BTRFS_QGROUP_STATUS_BIT_INCONSISTENT) /* * Whether or not this filesystem is using simple quotas. Not exactly the * incompat bit, because we support using simple quotas, disabling it, then * going back to full qgroup quotas. */ -#define BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE (1ULL << 3) +#define BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE (3) +#define BTRFS_QGROUP_STATUS_FLAG_SIMPLE_MODE (1UL << BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE) #define BTRFS_QGROUP_STATUS_FLAGS_MASK (BTRFS_QGROUP_STATUS_FLAG_ON | \ BTRFS_QGROUP_STATUS_FLAG_RESCAN | \ From 1d7404d4daeacbebe5a75f812dd6e45e3340d340 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 27 Aug 2026 16:25:29 +0930 Subject: [PATCH 0011/1012] btrfs: reject new qgroup rescan during subvolume dropping Commit 011b46c30476 ("btrfs: skip subtree scan if it's too high to avoid low stall in btrfs_commit_transaction()") introduced a threshold to skip huge subtree scan during subvolume dropping. But that's not covering all cases, e.g. rescan can still be started immediately after that huge subtree skipping. This will cause rescan to do the same accounting for that subtree anyway, still causing a long stall during transaction commit. Introduce a new runtime qgroup flag, BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, so that during cleanup of a subvolume, no new qgroup rescan can be initiated. The rejection uses the same -EINPROGRESS, as if there is already a running qgroup rescan. And since we have the extra bit, we can no longer allow plain assignment in btrfs_quota_enable(), as the plain assignment will override the REJECT_RESCAN bit. To co-operate this new flag: - Make btrfs_quota_enable() to only set BTRFS_QGROUP_STATUS_BIT_ON So it won't override the existing BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN bit. - Make btrfs_quota_disable() to clear every non-rescan bit This includes: * BTRFS_QGROUP_STATUS_BIT_ON * BTRFS_QGROUP_STATUS_BIT_INCONSISTENT * BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING For rescan related bits, they are either cleared by the rescan thread, or by the caller who rejects rescan. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 2 ++ fs/btrfs/qgroup.c | 13 +++++++++++-- fs/btrfs/qgroup.h | 11 +++++++++++ 3 files changed, 24 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 466fadb1815a81..a1d83ad9a4c00c 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -1496,7 +1496,9 @@ static int cleaner_kthread(void *arg) btrfs_run_delayed_iputs(fs_info); + set_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, &fs_info->qgroup_flags); again = btrfs_clean_one_deleted_snapshot(fs_info); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, &fs_info->qgroup_flags); mutex_unlock(&fs_info->cleaner_mutex); /* diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index 2c2ac0f16b1e5b..e01b31aa0b1bf0 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -1103,7 +1103,7 @@ int btrfs_quota_enable(struct btrfs_fs_info *fs_info, struct btrfs_qgroup_status_item); btrfs_set_qgroup_status_generation(leaf, ptr, trans->transid); btrfs_set_qgroup_status_version(leaf, ptr, BTRFS_QGROUP_STATUS_VERSION); - fs_info->qgroup_flags = (1UL << BTRFS_QGROUP_STATUS_BIT_ON); + set_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); if (simple) { set_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); btrfs_set_fs_incompat(fs_info, SIMPLE_QUOTA); @@ -1405,8 +1405,14 @@ int btrfs_quota_disable(struct btrfs_fs_info *fs_info) spin_lock(&fs_info->qgroup_lock); quota_root = fs_info->quota_root; fs_info->quota_root = NULL; + /* + * Clear all on-disk and runtime bits, except RESCAN related ones, that + * are either handled by rescan thread, or the caller who rejects rescan. + */ clear_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags); clear_bit(BTRFS_QGROUP_STATUS_BIT_SIMPLE_MODE, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_STATUS_BIT_INCONSISTENT, &fs_info->qgroup_flags); + clear_bit(BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING, &fs_info->qgroup_flags); fs_info->qgroup_drop_subtree_thres = BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT; spin_unlock(&fs_info->qgroup_lock); @@ -3990,7 +3996,10 @@ qgroup_rescan_init(struct btrfs_fs_info *fs_info, u64 progress_objectid, mutex_lock(&fs_info->qgroup_rescan_lock); if (init_flags) { - if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, &fs_info->qgroup_flags)) { + if (test_bit(BTRFS_QGROUP_STATUS_BIT_RESCAN, + &fs_info->qgroup_flags) || + test_bit(BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN, + &fs_info->qgroup_flags)) { ret = -EINPROGRESS; } else if (!test_bit(BTRFS_QGROUP_STATUS_BIT_ON, &fs_info->qgroup_flags)) { btrfs_debug(fs_info, diff --git a/fs/btrfs/qgroup.h b/fs/btrfs/qgroup.h index b3aaad5e617d51..090ba536787269 100644 --- a/fs/btrfs/qgroup.h +++ b/fs/btrfs/qgroup.h @@ -124,6 +124,17 @@ struct btrfs_qgroup_swapped_blocks; #define BTRFS_QGROUP_RUNTIME_BIT_CANCEL_RESCAN (BITS_PER_LONG - 1) #define BTRFS_QGROUP_RUNTIME_BIT_NO_ACCOUNTING (BITS_PER_LONG - 2) +/* + * No new rescan allowed when set. + * + * During huge subtree dropping, qgroup will be marked inconsistent, and skip + * all future accounting to avoid long stall. But, an immediate rescan will + * re-enable qgroup and still stall the system. + * + * This bit is to avoid such rescan during the duration of a subvolume dropping. + */ +#define BTRFS_QGROUP_RUNTIME_BIT_REJECT_RESCAN (BITS_PER_LONG - 3) + #define BTRFS_QGROUP_DROP_SUBTREE_THRES_DEFAULT (3) /* From ddfce12cf94208160df06449025a6e847c2ea8f6 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 27 Aug 2026 16:25:30 +0930 Subject: [PATCH 0012/1012] btrfs: avoid long stall when dropping a non-shared large subvolume Commit 011b46c30476 ("btrfs: skip subtree scan if it's too high to avoid low stall in btrfs_commit_transaction()") introduced a mechanism to skip large subtree during snapshot dropping. But even for a subvolume without any shared tree blocks, we can still queue quite a lot of qgroup records into one transaction, and cause a long qgroup related stall. So also add a check against the subvolume root level, to determine if we need to mark qgroup inconsistent. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/extent-tree.c | 10 ++++++++++ fs/btrfs/qgroup.c | 18 ++++++++++++++++++ fs/btrfs/qgroup.h | 1 + 3 files changed, 29 insertions(+) diff --git a/fs/btrfs/extent-tree.c b/fs/btrfs/extent-tree.c index d6a4390ee34ac9..a0d5ab03aae264 100644 --- a/fs/btrfs/extent-tree.c +++ b/fs/btrfs/extent-tree.c @@ -6315,6 +6315,16 @@ int btrfs_drop_snapshot(struct btrfs_root *root, bool update_ref, bool for_reloc set_bit(BTRFS_ROOT_DELETING, &root->state); unfinished_drop = test_bit(BTRFS_ROOT_UNFINISHED_DROP, &root->state); + /* + * For subvolume dropping, check if the subvolume is large enough so + * that we need to mark qgroup inconsistent to avoid long qgroup stall. + * + * Even for a subvolume without any snapshot, there can still be + * a lot of qgroup records queued into one transaction. + */ + if (!for_reloc) + btrfs_qgroup_check_tree_drop(fs_info, rootid, + btrfs_header_level(root->node)); if (btrfs_disk_key_objectid(&root_item->drop_progress) == 0) { level = btrfs_header_level(root->node); path->nodes[level] = btrfs_lock_root_node(root); diff --git a/fs/btrfs/qgroup.c b/fs/btrfs/qgroup.c index e01b31aa0b1bf0..05e35eb126dc5b 100644 --- a/fs/btrfs/qgroup.c +++ b/fs/btrfs/qgroup.c @@ -2748,6 +2748,24 @@ int btrfs_qgroup_trace_subtree(struct btrfs_trans_handle *trans, return 0; } +void btrfs_qgroup_check_tree_drop(struct btrfs_fs_info *fs_info, u64 rootid, u8 level) +{ + u8 drop_subtree_thres; + + if (btrfs_qgroup_mode(fs_info) != BTRFS_QGROUP_MODE_FULL) + return; + + if (!btrfs_is_fstree(rootid)) + return; + + spin_lock(&fs_info->qgroup_lock); + drop_subtree_thres = fs_info->qgroup_drop_subtree_thres; + spin_unlock(&fs_info->qgroup_lock); + + if (level >= drop_subtree_thres) + qgroup_mark_inconsistent(fs_info, "subtree level reached threshold"); +} + static void qgroup_iterator_nested_add(struct list_head *head, struct btrfs_qgroup *qgroup) { if (!list_empty(&qgroup->nested_iterator)) diff --git a/fs/btrfs/qgroup.h b/fs/btrfs/qgroup.h index 090ba536787269..c64b26b09c22f2 100644 --- a/fs/btrfs/qgroup.h +++ b/fs/btrfs/qgroup.h @@ -376,6 +376,7 @@ int btrfs_qgroup_trace_leaf_items(struct btrfs_trans_handle *trans, int btrfs_qgroup_trace_subtree(struct btrfs_trans_handle *trans, struct extent_buffer *root_eb, u64 root_gen, int root_level); +void btrfs_qgroup_check_tree_drop(struct btrfs_fs_info *fs_info, u64 rootid, u8 level); int btrfs_qgroup_account_extent(struct btrfs_trans_handle *trans, u64 bytenr, u64 num_bytes, struct ulist *old_roots, struct ulist *new_roots); From c8febd585c5124ec96d9346584b79c57e2def261 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Fri, 21 Aug 2026 19:42:10 +0930 Subject: [PATCH 0013/1012] btrfs: remove runtime tweakable feature sysfs interface There are 2 features that are marked runtime tweakable inside /sys/fs/btrfs/features/ - acl Which is a mount option, and it will not show up in /sys/fs/btrfs//features/ directory anyway. - extended_iref This feature can only be enabled, but not disabled at runtime. Furthermore it's already the default behavior since 3.12. So it means this feature is always enabled and cannot be disabled for modern btrfs. So there is no need to maintain the ability to modify btrfs' runtime features through sysfs. And furthermore, the existing btrfs_feature_attr_store() is race-prone, it relies on fs_info->transaction_kthread, but our sysfs interfaces are enabled before transaction_kthread. Meaning at mount time a sysfs write can trigger NULL pointer dereference if the transaction_kthread is not yet initialized. The opposite is also possible during unmount. Thankfully that race is not possible in the real world, as the only supported feature is already enabled. But it also means we do not really need to keep the race-prone infrastructure, so just remove it completely, and make the per-module and per-mount features files to be completely read-only. Even with the sysfs tweakable features removed, we can still enable extended_iref feature through ioctl. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/sysfs.c | 121 ++--------------------------------------------- 1 file changed, 4 insertions(+), 117 deletions(-) diff --git a/fs/btrfs/sysfs.c b/fs/btrfs/sysfs.c index 1df6340a71234f..c5bb1c7eac6afe 100644 --- a/fs/btrfs/sysfs.c +++ b/fs/btrfs/sysfs.c @@ -83,8 +83,7 @@ struct raid_kobject { #define BTRFS_FEAT_ATTR(_name, _feature_set, _feature_prefix, _feature_bit) \ static struct btrfs_feature_attr btrfs_attr_features_##_name = { \ .kobj_attr = __INIT_KOBJ_ATTR(_name, S_IRUGO, \ - btrfs_feature_attr_show, \ - btrfs_feature_attr_store), \ + btrfs_feature_attr_show, NULL), \ .feature_set = _feature_set, \ .feature_bit = _feature_prefix ##_## _feature_bit, \ } @@ -130,130 +129,20 @@ static u64 get_features(struct btrfs_fs_info *fs_info, return btrfs_super_incompat_flags(disk_super); } -static void set_features(struct btrfs_fs_info *fs_info, - enum btrfs_feature_set set, u64 features) -{ - struct btrfs_super_block *disk_super = fs_info->super_copy; - if (set == FEAT_COMPAT) - btrfs_set_super_compat_flags(disk_super, features); - else if (set == FEAT_COMPAT_RO) - btrfs_set_super_compat_ro_flags(disk_super, features); - else - btrfs_set_super_incompat_flags(disk_super, features); -} - -static int can_modify_feature(struct btrfs_feature_attr *fa) -{ - int val = 0; - u64 set, clear; - switch (fa->feature_set) { - case FEAT_COMPAT: - set = BTRFS_FEATURE_COMPAT_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_SAFE_CLEAR; - break; - case FEAT_COMPAT_RO: - set = BTRFS_FEATURE_COMPAT_RO_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_RO_SAFE_CLEAR; - break; - case FEAT_INCOMPAT: - set = BTRFS_FEATURE_INCOMPAT_SAFE_SET; - clear = BTRFS_FEATURE_INCOMPAT_SAFE_CLEAR; - break; - default: - btrfs_warn(NULL, "sysfs: unknown feature set %d", fa->feature_set); - return 0; - } - - if (set & fa->feature_bit) - val |= 1; - if (clear & fa->feature_bit) - val |= 2; - - return val; -} - static ssize_t btrfs_feature_attr_show(struct kobject *kobj, struct kobj_attribute *a, char *buf) { int val = 0; struct btrfs_fs_info *fs_info = to_fs_info(kobj); struct btrfs_feature_attr *fa = to_btrfs_feature_attr(a); + if (fs_info) { u64 features = get_features(fs_info, fa->feature_set); if (features & fa->feature_bit) val = 1; - } else - val = can_modify_feature(fa); - - return sysfs_emit(buf, "%d\n", val); -} - -static ssize_t btrfs_feature_attr_store(struct kobject *kobj, - struct kobj_attribute *a, - const char *buf, size_t count) -{ - struct btrfs_fs_info *fs_info; - struct btrfs_feature_attr *fa = to_btrfs_feature_attr(a); - u64 features, set, clear; - unsigned long val; - int ret; - - fs_info = to_fs_info(kobj); - if (!fs_info) - return -EPERM; - - if (sb_rdonly(fs_info->sb)) - return -EROFS; - - ret = kstrtoul(skip_spaces(buf), 0, &val); - if (ret) - return ret; - - if (fa->feature_set == FEAT_COMPAT) { - set = BTRFS_FEATURE_COMPAT_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_SAFE_CLEAR; - } else if (fa->feature_set == FEAT_COMPAT_RO) { - set = BTRFS_FEATURE_COMPAT_RO_SAFE_SET; - clear = BTRFS_FEATURE_COMPAT_RO_SAFE_CLEAR; - } else { - set = BTRFS_FEATURE_INCOMPAT_SAFE_SET; - clear = BTRFS_FEATURE_INCOMPAT_SAFE_CLEAR; } - features = get_features(fs_info, fa->feature_set); - - /* Nothing to do */ - if ((val && (features & fa->feature_bit)) || - (!val && !(features & fa->feature_bit))) - return count; - - if ((val && !(set & fa->feature_bit)) || - (!val && !(clear & fa->feature_bit))) { - btrfs_info(fs_info, - "%sabling feature %s on mounted fs is not supported.", - val ? "En" : "Dis", fa->kobj_attr.attr.name); - return -EPERM; - } - - btrfs_info(fs_info, "%s %s feature flag", - val ? "Setting" : "Clearing", fa->kobj_attr.attr.name); - - spin_lock(&fs_info->super_lock); - features = get_features(fs_info, fa->feature_set); - if (val) - features |= fa->feature_bit; - else - features &= ~fa->feature_bit; - set_features(fs_info, fa->feature_set, features); - spin_unlock(&fs_info->super_lock); - - /* - * We don't want to do full transaction commit from inside sysfs - */ - set_bit(BTRFS_FS_NEED_TRANS_COMMIT, &fs_info->flags); - wake_up_process(fs_info->transaction_kthread); - - return count; + return sysfs_emit(buf, "%d\n", val); } static umode_t btrfs_feature_visible(struct kobject *kobj, @@ -269,9 +158,7 @@ static umode_t btrfs_feature_visible(struct kobject *kobj, fa = attr_to_btrfs_feature_attr(attr); features = get_features(fs_info, fa->feature_set); - if (can_modify_feature(fa)) - mode |= S_IWUSR; - else if (!(features & fa->feature_bit)) + if (!(features & fa->feature_bit)) mode = 0; } From 377cbea9b53e60a393c4c62867cfe03ae8a7e648 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 1 Sep 2026 09:31:01 +0930 Subject: [PATCH 0014/1012] btrfs: use ordered extent to grab the logical address for submission In submit_one_sector() we call btrfs_get_extent() to grab the IO extent map so that we know where the logical location to submit the block. However there is no guarantee that there is an IO extent map for the block, and if there is no IO extent map nor ordered extent, btrfs_get_extent() can grab the file extent from on-disk metadata. That's why we have ASSERT()s to reject holes and compressed file extents. On the other hand, for the write range we should have both an IO extent map and an ordered extent, so there is no reason not to grab the ordered extent instead. There is some minor advantages: - No hole ordered extent So no need to rely on ASSERT()s to reject hole extents. And the ASSERT()s are depending on the kernel config, without CONFIG_BTRFS_ASSERT those ASSERT()s won't even trigger. - No IO errors Unlike btrfs_get_extent() which can return IO error when doing the metadata tree search, btrfs_lookup_ordered_extent() will either return an OE or not found. - Cached OE in bio_ctrl->bbio At bbio allocation we have already did an OE lookup, and we have a high chance that the current block also belongs to that OE. Use that cached OE can reduce the frequency to do an rb-tree search. - Smaller rb-tree Unlike extent-map-tree, which can contain cached extent maps, the life span of ordered extents are much shorter, they get removed from the ordered tree after the file extent item is inserted into the subvolume tree. So doing ordered extent tree search can be a tiny faster. And since we're here, also address some minor points: - Add error message for every EUCLEAN error - Remove a dead comment on btrfs_folio_clear_dirty() We no longer call folio_clear_dirty_for_io() since commit 095be159f3eb ("btrfs: unify folio dirty flag clearing"), so the folio flag is still dirty, and the folio dirty flag will be cleared by the last dirty block. Reviewed-by: Boris Burkov Reviewed-by: Johannes Thumshirn Reviewed-by: Daniel Vacek Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/extent_io.c | 64 +++++++++++++++++++++++++++----------------- 1 file changed, 40 insertions(+), 24 deletions(-) diff --git a/fs/btrfs/extent_io.c b/fs/btrfs/extent_io.c index d7600e5fa3d95d..a221b63bdb205d 100644 --- a/fs/btrfs/extent_io.c +++ b/fs/btrfs/extent_io.c @@ -1808,6 +1808,22 @@ static noinline_for_stack int writepage_delalloc(struct btrfs_inode *inode, return 0; } +static struct btrfs_ordered_extent *get_oe_from_bbio(const struct btrfs_bio *bbio, + u64 filepos) +{ + struct btrfs_ordered_extent *oe; + + if (!bbio || !bbio->ordered) + return NULL; + + oe = bbio->ordered; + if (!in_range(filepos, oe->file_offset, oe->num_bytes)) + return NULL; + + refcount_inc(&oe->refs); + return oe; +} + /* * Return 0 if we have submitted or queued the sector for submission. * Return <0 for critical errors, and the involved sector will be cleaned up. @@ -1820,11 +1836,10 @@ static int submit_one_sector(struct btrfs_inode *inode, loff_t i_size) { struct btrfs_fs_info *fs_info = inode->root->fs_info; - struct extent_map *em; + struct btrfs_ordered_extent *oe; u64 block_start; u64 disk_bytenr; u64 extent_offset; - u64 em_end; const u32 sectorsize = fs_info->sectorsize; unsigned int queued; @@ -1833,8 +1848,11 @@ static int submit_one_sector(struct btrfs_inode *inode, /* @filepos >= i_size case should be handled by the caller. */ ASSERT(filepos < i_size); - em = btrfs_get_extent(inode, NULL, filepos, sectorsize); - if (IS_ERR(em)) { + /* Try to reuse the existing OE from bbio first. */ + oe = get_oe_from_bbio(bio_ctrl->bbio, filepos); + if (!oe) + oe = btrfs_lookup_ordered_extent(inode, filepos); + if (unlikely(!oe)) { /* * bio_ctrl may contain a bio crossing several folios. * Submit it immediately so that the bio has a chance @@ -1857,31 +1875,25 @@ static int submit_one_sector(struct btrfs_inode *inode, */ btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize, false); - return PTR_ERR(em); + btrfs_err_rl(fs_info, + "no ordered extent for root %lld ino %llu filepos %llu", + btrfs_root_id(inode->root), btrfs_ino(inode), + filepos); + return -EUCLEAN; } - extent_offset = filepos - em->start; - em_end = btrfs_extent_map_end(em); - ASSERT(filepos <= em_end); - ASSERT(IS_ALIGNED(em->start, sectorsize)); - ASSERT(IS_ALIGNED(em->len, sectorsize)); - - block_start = btrfs_extent_map_block_start(em); - disk_bytenr = btrfs_extent_map_block_start(em) + extent_offset; + extent_offset = filepos - oe->file_offset; + ASSERT(filepos < oe->file_offset + oe->num_bytes); + ASSERT(IS_ALIGNED(oe->file_offset, sectorsize)); + ASSERT(IS_ALIGNED(oe->num_bytes, sectorsize)); + ASSERT(oe->compress_type == BTRFS_COMPRESS_NONE); + ASSERT(!test_bit(BTRFS_ORDERED_COMPRESSED, &oe->flags)); - ASSERT(!btrfs_extent_map_is_compressed(em)); - ASSERT(block_start != EXTENT_MAP_HOLE); - ASSERT(block_start != EXTENT_MAP_INLINE); + block_start = oe->disk_bytenr + oe->offset; + disk_bytenr = block_start + extent_offset; - btrfs_free_extent_map(em); - em = NULL; + btrfs_put_ordered_extent(oe); - /* - * Although the PageDirty bit is cleared before entering this - * function, subpage dirty bit is not cleared. - * So clear subpage dirty bit here so next time we won't submit - * a folio for a range already written to disk. - */ btrfs_folio_clear_dirty(fs_info, folio, filepos, sectorsize); btrfs_folio_set_writeback(fs_info, folio, filepos, sectorsize); /* @@ -1898,6 +1910,10 @@ static int submit_one_sector(struct btrfs_inode *inode, btrfs_folio_clear_writeback(fs_info, folio, filepos, sectorsize); btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize, false); + btrfs_err_rl(fs_info, + "failed to queue sector for root %lld ino %llu filepos %llu", + btrfs_root_id(inode->root), + btrfs_ino(inode), filepos); return -EUCLEAN; } return 0; From 202c9312cbc25e165c8a4af20dfe83ea2653ce00 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 1 Sep 2026 10:01:33 +0930 Subject: [PATCH 0015/1012] btrfs: tree-checker: reject file extent items for special files File extent items are only utilized by regular files or symlinks, other files like directory/char/block/FIFO/sock files should not have any file extent item. Previously we were unable to reject such cases, as the inode item may not be in the same leaf. But we already have @prev_key in check_leaf_item(), this means we just need a new way to pass the mode of the previously hit inode item, then we can detect such problems. Introduce a new helper structure, saved_inode_info, to record the inode number and its mode hit in the same leaf, and keep it across the whole leaf. Then if we hit a file extent item, and the inode item is in the same leaf, we can refer to that to determine if we need to reject the file extent item. Now with the following corrupted fs tree, the kernel can safely reject the leaf: item 0 key (256 INODE_ITEM 0) itemoff 16123 itemsize 160 generation 3 transid 9 size 12 nbytes 16384 block group 0 mode 40755 links 1 uid 0 gid 0 rdev 0 sequence 1 flags 0x0(none) item 1 key (256 INODE_REF 256) itemoff 16111 itemsize 12 index 0 namelen 2 name: .. item 2 key (256 DIR_ITEM 496027801) itemoff 16075 itemsize 36 location key (257 INODE_ITEM 0) type FILE transid 9 data_len 0 name_len 6 name: foobar item 3 key (256 DIR_INDEX 2) itemoff 16039 itemsize 36 location key (257 INODE_ITEM 0) type FILE transid 9 data_len 0 name_len 6 name: foobar item 4 key (257 INODE_ITEM 0) itemoff 15879 itemsize 160 generation 9 transid 9 size 8192 nbytes 8192 block group 0 mode 60600 links 1 uid 0 gid 0 rdev 0 ^^ This is BLK type, not REG. sequence 2 flags 0x0(none) item 5 key (257 INODE_REF 256) itemoff 15863 itemsize 16 index 2 namelen 6 name: foobar item 6 key (257 EXTENT_DATA 0) itemoff 15810 itemsize 53 generation 9 type 1 (regular) extent data disk byte 13631488 nr 8192 extent data offset 0 nr 8192 ram 8192 extent compression 0 (none) extent encryption 0 With the patch, kernel will reject it with the following tree-checker errors: BTRFS critical (device loop0): corrupt leaf: root=5 block=30408704 slot=6 ino=257 file_offset=0, unexpected file extent item type 1 for inode mode 060600 BTRFS error (device loop0): read time tree block corruption detected on logical 30408704 mirror 1 Reported-by: ZhengYuan Huang Link: https://lore.kernel.org/linux-btrfs/20260817132051.267646-1-gality369@gmail.com/ Assisted-by: LLM (for generating the corrupted image) Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tree-checker.c | 66 ++++++++++++++++++++++++++++++++++------- 1 file changed, 55 insertions(+), 11 deletions(-) diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index ab5abbb475e2ca..401fd40ec7b22e 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -163,6 +163,12 @@ static void dir_item_err(const struct extent_buffer *eb, int slot, va_end(args); } +/* Record info for the last hit inode. */ +struct saved_inode_info { + u64 ino; + u32 mode; +}; + /* * This functions checks prev_key->objectid, to ensure current key and prev_key * share the same objectid as inode number. @@ -204,15 +210,41 @@ static bool check_prev_ino(struct extent_buffer *leaf, prev_key->objectid, key->objectid); return false; } + +static bool can_have_extent_data(struct extent_buffer *leaf, + struct btrfs_key *key, int slot, u8 fi_type, + const struct saved_inode_info *inode_info) +{ + /* No inode item in this leaf. */ + if (inode_info->ino != key->objectid) + return true; + if (S_ISREG(inode_info->mode)) + return true; + if (S_ISLNK(inode_info->mode)) { + /* For a symlink, the file extent item should always be inlined. */ + if (unlikely(fi_type != BTRFS_FILE_EXTENT_INLINE)) + return false; + return true; + } + + /* + * The rest are special files, e.g. block/FIFO files, which cannnot + * have any file extent. + */ + return false; +} + static int check_extent_data_item(struct extent_buffer *leaf, struct btrfs_key *key, int slot, - struct btrfs_key *prev_key) + struct btrfs_key *prev_key, + const struct saved_inode_info *inode_info) { struct btrfs_fs_info *fs_info = leaf->fs_info; struct btrfs_file_extent_item *fi; u32 sectorsize = fs_info->sectorsize; u32 item_size = btrfs_item_size(leaf, slot); u64 extent_end; + u8 fi_type; if (unlikely(!IS_ALIGNED(key->offset, sectorsize))) { file_extent_err(leaf, slot, @@ -243,12 +275,18 @@ static int check_extent_data_item(struct extent_buffer *leaf, SZ_4K); return -EUCLEAN; } - if (unlikely(btrfs_file_extent_type(leaf, fi) >= - BTRFS_NR_FILE_EXTENT_TYPES)) { + fi_type = btrfs_file_extent_type(leaf, fi); + if (unlikely(fi_type >= BTRFS_NR_FILE_EXTENT_TYPES)) { file_extent_err(leaf, slot, "invalid type for file extent, have %u expect range [0, %u]", - btrfs_file_extent_type(leaf, fi), - BTRFS_NR_FILE_EXTENT_TYPES - 1); + fi_type, BTRFS_NR_FILE_EXTENT_TYPES - 1); + return -EUCLEAN; + } + + if (unlikely(!can_have_extent_data(leaf, key, slot, fi_type, inode_info))) { + file_extent_err(leaf, slot, + "unexpected file extent item type %u for inode mode 0%o", + fi_type, inode_info->mode); return -EUCLEAN; } @@ -270,7 +308,8 @@ static int check_extent_data_item(struct extent_buffer *leaf, btrfs_file_extent_encryption(leaf, fi)); return -EUCLEAN; } - if (btrfs_file_extent_type(leaf, fi) == BTRFS_FILE_EXTENT_INLINE) { + + if (fi_type == BTRFS_FILE_EXTENT_INLINE) { /* Inline extent must have 0 as key offset */ if (unlikely(key->offset)) { file_extent_err(leaf, slot, @@ -1206,7 +1245,8 @@ static int check_dev_item(struct extent_buffer *leaf, } static int check_inode_item(struct extent_buffer *leaf, - struct btrfs_key *key, int slot) + struct btrfs_key *key, int slot, + struct saved_inode_info *inode_info) { struct btrfs_fs_info *fs_info = leaf->fs_info; struct btrfs_inode_item *iitem; @@ -1291,6 +1331,8 @@ static int check_inode_item(struct extent_buffer *leaf, ro_flags); return -EUCLEAN; } + inode_info->ino = key->objectid; + inode_info->mode = mode; return 0; } @@ -2348,14 +2390,15 @@ static int check_free_space_bitmap(struct extent_buffer *leaf, static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf, struct btrfs_key *key, int slot, - struct btrfs_key *prev_key) + struct btrfs_key *prev_key, + struct saved_inode_info *inode_info) { int ret = 0; struct btrfs_chunk *chunk; switch (key->type) { case BTRFS_EXTENT_DATA_KEY: - ret = check_extent_data_item(leaf, key, slot, prev_key); + ret = check_extent_data_item(leaf, key, slot, prev_key, inode_info); break; case BTRFS_EXTENT_CSUM_KEY: ret = check_csum_item(leaf, key, slot, prev_key); @@ -2385,7 +2428,7 @@ static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf, ret = check_dev_extent_item(leaf, key, slot, prev_key); break; case BTRFS_INODE_ITEM_KEY: - ret = check_inode_item(leaf, key, slot); + ret = check_inode_item(leaf, key, slot, inode_info); break; case BTRFS_ROOT_ITEM_KEY: ret = check_root_item(leaf, key, slot); @@ -2433,6 +2476,7 @@ static enum btrfs_tree_block_status check_leaf_item(struct extent_buffer *leaf, enum btrfs_tree_block_status __btrfs_check_leaf(struct extent_buffer *leaf) { struct btrfs_fs_info *fs_info = leaf->fs_info; + struct saved_inode_info inode_info = { 0 }; /* No valid key type is 0, so all key should be larger than this key */ struct btrfs_key prev_key = {0, 0, 0}; struct btrfs_key key; @@ -2568,7 +2612,7 @@ enum btrfs_tree_block_status __btrfs_check_leaf(struct extent_buffer *leaf) } /* Check if the item size and content meet other criteria. */ - ret = check_leaf_item(leaf, &key, slot, &prev_key); + ret = check_leaf_item(leaf, &key, slot, &prev_key, &inode_info); if (unlikely(ret != BTRFS_TREE_BLOCK_CLEAN)) return ret; From 7d596776e6dd98d141029751ca502bada99ee232 Mon Sep 17 00:00:00 2001 From: Johannes Thumshirn Date: Mon, 24 Aug 2026 18:19:10 +0200 Subject: [PATCH 0016/1012] btrfs: zoned: handle RAID profiles in btrfs_can_activate_zone() btrfs_can_activate_zone() only accounts for the single and DUP profiles. For a RAID0, RAID1, RAID1C3, RAID1C4 or RAID10 block group the profile switch matches no case, so 'ret' stays false and the function reports that no zone can be activated, even when the devices have plenty of active zones left. As a side effect BTRFS_FS_NEED_ZONE_FINISH gets set and, since btrfs_can_activate_zone() bails out early once that bit is set, data allocations will fail permanently: writers loop on -EAGAIN and hang in btrfs_new_extent_direct() waiting for the bit to clear. Each of these profiles needs one active zone per device, just like single, so handle them the same way. Reviewed-by: Boris Burkov Signed-off-by: Johannes Thumshirn Signed-off-by: David Sterba --- fs/btrfs/zoned.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c index 9cc2c9c1a606b6..08a15465a0877d 100644 --- a/fs/btrfs/zoned.c +++ b/fs/btrfs/zoned.c @@ -2688,6 +2688,11 @@ bool btrfs_can_activate_zone(struct btrfs_fs_devices *fs_devices, u64 flags) switch (flags & BTRFS_BLOCK_GROUP_PROFILE_MASK) { case 0: /* single */ + case BTRFS_BLOCK_GROUP_RAID0: + case BTRFS_BLOCK_GROUP_RAID1: + case BTRFS_BLOCK_GROUP_RAID1C3: + case BTRFS_BLOCK_GROUP_RAID1C4: + case BTRFS_BLOCK_GROUP_RAID10: ret = (atomic_read(&zinfo->active_zones_left) >= (1 + reserved)); break; case BTRFS_BLOCK_GROUP_DUP: From b077fa306ebe9095130720841588c4adf6b549a0 Mon Sep 17 00:00:00 2001 From: Daniel Vacek Date: Wed, 2 Sep 2026 11:20:24 +0200 Subject: [PATCH 0017/1012] btrfs: consume given iter directly instead of copying in csum_one_bio() Avoid copying the iter twice in async case. We already have a copy csum_one_bio() can consume directly. No need to copy it again the second time. We can use this copy also in the sync case and get rid of the parameter. Reviewed-by: Qu Wenruo Signed-off-by: Daniel Vacek Signed-off-by: David Sterba --- fs/btrfs/file-item.c | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/fs/btrfs/file-item.c b/fs/btrfs/file-item.c index 581ca5653be93a..2c6c14dfe39320 100644 --- a/fs/btrfs/file-item.c +++ b/fs/btrfs/file-item.c @@ -797,18 +797,18 @@ int btrfs_lookup_csums_bitmap(struct btrfs_root *root, struct btrfs_path *path, return ret; } -static void csum_one_bio(struct btrfs_bio *bbio, struct bvec_iter *src) +static void csum_one_bio(struct btrfs_bio *bbio) { struct btrfs_inode *inode = bbio->inode; struct btrfs_fs_info *fs_info = inode->root->fs_info; struct btrfs_ordered_sum *sums = bbio->sums; - struct bvec_iter iter; const u32 blocksize = fs_info->sectorsize; int index = 0; - for (iter = *src; iter.bi_size; bio_advance_iter(&bbio->bio, &iter, blocksize)) { - btrfs_csum_one_bio_block(fs_info, &bbio->bio, &iter, - sums->sums + index); + for (struct bvec_iter *iter = &bbio->csum_saved_iter; + iter->bi_size; + bio_advance_iter(&bbio->bio, iter, blocksize)) { + btrfs_csum_one_bio_block(fs_info, &bbio->bio, iter, sums->sums + index); index += fs_info->csum_size; } @@ -820,7 +820,7 @@ static void csum_one_bio_work(struct work_struct *work) ASSERT(btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE); ASSERT(bbio->async_csum == true); - csum_one_bio(bbio, &bbio->csum_saved_iter); + csum_one_bio(bbio); complete(&bbio->csum_done); } @@ -850,13 +850,13 @@ int btrfs_csum_one_bio(struct btrfs_bio *bbio, bool async) bbio->sums = sums; btrfs_add_ordered_sum(ordered, sums); + bbio->csum_saved_iter = bio->bi_iter; if (!async) { - csum_one_bio(bbio, &bbio->bio.bi_iter); + csum_one_bio(bbio); return 0; } init_completion(&bbio->csum_done); bbio->async_csum = true; - bbio->csum_saved_iter = bbio->bio.bi_iter; INIT_WORK(&bbio->csum_work, csum_one_bio_work); schedule_work(&bbio->csum_work); return 0; From ccc1ddddcc84d99820aa4681f141fa0896b6cb7f Mon Sep 17 00:00:00 2001 From: Daniel Vacek Date: Wed, 2 Sep 2026 15:56:27 +0200 Subject: [PATCH 0018/1012] btrfs: use bio::remaining for async checksumming synchronization We can use bio::remaining counter to sync the offloaded checksumming. As a result we can slim down the btrfs_bio structure by 24 bytes and simplify the code a bit. Difference in pahole output: - /* size: 328, cachelines: 6, members: 15 */ + /* size: 304, cachelines: 5, members: 14 */ Moreover this will allow us enabling async checksumming with encryption where we need to checksum the bounce bio instead of our regular one embedded in btrfs_bio. And so we need to extend it's lifetime. This is the preferred way to do so. This also fixes a bug in experimental build where the async checksumming was using the system workqueue instead of fs_info::endio_workers. Fixes: dd57c78aec39 ("btrfs: introduce btrfs_bio::async_csum") Reviewed-by: Qu Wenruo Signed-off-by: Daniel Vacek Signed-off-by: David Sterba --- fs/btrfs/bio.c | 4 ---- fs/btrfs/bio.h | 4 ---- fs/btrfs/file-item.c | 8 +++----- 3 files changed, 3 insertions(+), 13 deletions(-) diff --git a/fs/btrfs/bio.c b/fs/btrfs/bio.c index 19b4855969f536..771b7d598aeeeb 100644 --- a/fs/btrfs/bio.c +++ b/fs/btrfs/bio.c @@ -103,7 +103,6 @@ static struct btrfs_bio *btrfs_split_bio(struct btrfs_fs_info *fs_info, bbio->can_use_append = orig_bbio->can_use_append; bbio->is_scrub = orig_bbio->is_scrub; bbio->is_remap = orig_bbio->is_remap; - bbio->async_csum = orig_bbio->async_csum; atomic_inc(&orig_bbio->pending_ios); return bbio; @@ -114,9 +113,6 @@ void btrfs_bio_end_io(struct btrfs_bio *bbio, blk_status_t status) /* Make sure we're already in task context. */ ASSERT(in_task()); - if (bbio->async_csum) - wait_for_completion(&bbio->csum_done); - bbio->bio.bi_status = status; if (bbio->bio.bi_pool == &btrfs_clone_bioset) { struct btrfs_bio *orig_bbio = bbio->private; diff --git a/fs/btrfs/bio.h b/fs/btrfs/bio.h index b7bd377a016249..bbf362b8668bbc 100644 --- a/fs/btrfs/bio.h +++ b/fs/btrfs/bio.h @@ -58,7 +58,6 @@ struct btrfs_bio { struct btrfs_ordered_extent *ordered; struct btrfs_ordered_sum *sums; struct work_struct csum_work; - struct completion csum_done; struct bvec_iter csum_saved_iter; u64 orig_physical; u64 orig_logical; @@ -93,9 +92,6 @@ struct btrfs_bio { /* Whether the bio is coming from copy_remapped_data_io(). */ bool is_remap:1; - /* Whether the csum generation for data write is async. */ - bool async_csum:1; - /* Whether the bio is written using zone append. */ bool can_use_append:1; diff --git a/fs/btrfs/file-item.c b/fs/btrfs/file-item.c index 2c6c14dfe39320..ae1fd4da38d31b 100644 --- a/fs/btrfs/file-item.c +++ b/fs/btrfs/file-item.c @@ -819,9 +819,8 @@ static void csum_one_bio_work(struct work_struct *work) struct btrfs_bio *bbio = container_of(work, struct btrfs_bio, csum_work); ASSERT(btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE); - ASSERT(bbio->async_csum == true); csum_one_bio(bbio); - complete(&bbio->csum_done); + bio_endio(&bbio->bio); } /* @@ -855,10 +854,9 @@ int btrfs_csum_one_bio(struct btrfs_bio *bbio, bool async) csum_one_bio(bbio); return 0; } - init_completion(&bbio->csum_done); - bbio->async_csum = true; + bio_inc_remaining(bio); INIT_WORK(&bbio->csum_work, csum_one_bio_work); - schedule_work(&bbio->csum_work); + queue_work(fs_info->endio_workers, &bbio->csum_work); return 0; } From 5cbbbc2b907ea1c0ac1808df543b97dafbb4d890 Mon Sep 17 00:00:00 2001 From: Hemanth Selam Date: Mon, 7 Sep 2026 10:18:01 +0530 Subject: [PATCH 0019/1012] btrfs: fix typos and repeated words in comments Fix misspellings and repeated words in comments, found with scripts/checkpatch.pl and codespell. Only touches comments, no code changes. Assisted-by: Cursor:claude-opus-5 Signed-off-by: Hemanth Selam Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/block-group.h | 2 +- fs/btrfs/extent-io-tree.c | 2 +- fs/btrfs/fs.c | 2 +- fs/btrfs/raid56.c | 11 ++++------- fs/btrfs/send.c | 2 +- fs/btrfs/transaction.h | 2 +- fs/btrfs/tree-checker.c | 2 +- include/uapi/linux/btrfs_tree.h | 4 ++-- 8 files changed, 12 insertions(+), 15 deletions(-) diff --git a/fs/btrfs/block-group.h b/fs/btrfs/block-group.h index 69d56864d4baa0..b349f94cf929ab 100644 --- a/fs/btrfs/block-group.h +++ b/fs/btrfs/block-group.h @@ -135,7 +135,7 @@ struct btrfs_block_group { u64 global_root_id; u64 remap_bytes; u32 identity_remap_count; - /* The last commited identity_remap_count value of this block group. */ + /* The last committed identity_remap_count value of this block group. */ u32 last_identity_remap_count; /* * The last committed used bytes of this block group, if the above @used diff --git a/fs/btrfs/extent-io-tree.c b/fs/btrfs/extent-io-tree.c index d6df11f6088c44..992b8b42bdb4e3 100644 --- a/fs/btrfs/extent-io-tree.c +++ b/fs/btrfs/extent-io-tree.c @@ -751,7 +751,7 @@ int btrfs_clear_extent_bit_changeset(struct extent_io_tree *tree, u64 start, u64 btrfs_split_delalloc_extent(tree->inode, state, start); /* - * Temporarilly ajdust this state's range to match the + * Temporarily ajdust this state's range to match the * range for which we are clearing bits. */ state->start = start; diff --git a/fs/btrfs/fs.c b/fs/btrfs/fs.c index de160d29dde82f..75a1217727a7bb 100644 --- a/fs/btrfs/fs.c +++ b/fs/btrfs/fs.c @@ -79,7 +79,7 @@ void btrfs_csum_init(struct btrfs_csum_ctx *ctx, u16 csum_type) blake2b_init(&ctx->blake2b, 32); break; default: - /* Checksume type is validated at mount time. */ + /* Checksum type is validated at mount time. */ BUG(); } } diff --git a/fs/btrfs/raid56.c b/fs/btrfs/raid56.c index a5d0ef09d92abb..8ec24dbb180f92 100644 --- a/fs/btrfs/raid56.c +++ b/fs/btrfs/raid56.c @@ -953,7 +953,7 @@ static void rbio_orig_end_io(struct btrfs_raid_bio *rbio, blk_status_t status) /* * Clear the data bitmap, as the rbio may be cached for later usage. - * do this before before unlock_stripe() so there will be no new bio + * do this before unlock_stripe() so there will be no new bio * for this bio. */ bitmap_clear(&rbio->dbitmap, 0, rbio->stripe_nsectors); @@ -988,7 +988,7 @@ static void rbio_orig_end_io(struct btrfs_raid_bio *rbio, blk_status_t status) * as possible, and only use stripe_sectors as fallback. * * Return NULL if bio_list_only is set but the specified sector has no - * coresponding bio. + * corresponding bio. */ static phys_addr_t *sector_paddrs_in_rbio(struct btrfs_raid_bio *rbio, int stripe_nr, int sector_nr, @@ -1450,10 +1450,7 @@ static int rmw_assemble_write_bios(struct btrfs_raid_bio *rbio, /* We should have at least one data sector. */ ASSERT(bitmap_weight(&rbio->dbitmap, rbio->stripe_nsectors)); - /* - * Reset errors, as we may have errors inherited from from degraded - * write. - */ + /* Reset errors, as we may have errors inherited from degraded write. */ bitmap_clear(rbio->error_bitmap, 0, rbio->nr_sectors); /* @@ -2632,7 +2629,7 @@ static int alloc_rbio_essential_pages(struct btrfs_raid_bio *rbio) return 0; } -/* Return true if the content of the step matches the caclulated one. */ +/* Return true if the content of the step matches the calculated one. */ static bool verify_one_parity_step(struct btrfs_raid_bio *rbio, void *pointers[], unsigned int sector_nr, unsigned int step_nr) diff --git a/fs/btrfs/send.c b/fs/btrfs/send.c index 5c59b9abedcd71..c523bf950c898f 100644 --- a/fs/btrfs/send.c +++ b/fs/btrfs/send.c @@ -7023,7 +7023,7 @@ static int changed_extent(struct send_ctx *sctx, * get modified or replaced with a new one). Note that deduplication * updates the inode item, but it only changes the iversion (sequence * field in the inode item) of the inode, so if a file is deduplicated - * the same amount of times in both the parent and send snapshots, its + * the same number of times in both the parent and send snapshots, its * iversion becomes the same in both snapshots, whence the inode item is * the same on both snapshots. */ diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h index 3a57f227b5ed28..89153cd2259678 100644 --- a/fs/btrfs/transaction.h +++ b/fs/btrfs/transaction.h @@ -288,7 +288,7 @@ do { \ * Call btrfs_abort_transaction() as early as possible when an error condition * is detected, that way the exact stack trace is reported for some errors. * - * Error number must be negative as it encodes wheather it's the first abort. + * Error number must be negative as it encodes whether it's the first abort. */ #define btrfs_abort_transaction(trans, error) \ do { \ diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index 401fd40ec7b22e..b8ca9980e1ddd2 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -228,7 +228,7 @@ static bool can_have_extent_data(struct extent_buffer *leaf, } /* - * The rest are special files, e.g. block/FIFO files, which cannnot + * The rest are special files, e.g. block/FIFO files, which cannot * have any file extent. */ return false; diff --git a/include/uapi/linux/btrfs_tree.h b/include/uapi/linux/btrfs_tree.h index b6ccaf848e4b3e..47ee52859b4589 100644 --- a/include/uapi/linux/btrfs_tree.h +++ b/include/uapi/linux/btrfs_tree.h @@ -230,7 +230,7 @@ * * Stored as an inline ref rather to avoid wasting space on a separate item on * top of the existing extent item. However, unlike the other inline refs, - * there is one one owner ref per extent rather than one per extent. + * there is one owner ref per extent rather than one per extent. * * Because of this, it goes at the front of the list of inline refs, and thus * must have a lower type value than any other inline ref type (to satisfy the @@ -243,7 +243,7 @@ #define BTRFS_EXTENT_DATA_REF_KEY 178 /* - * Obsolete key. Defintion removed in 6.6, value may be reused in the future. + * Obsolete key. Definition removed in 6.6, value may be reused in the future. * * #define BTRFS_EXTENT_REF_V0_KEY 180 */ From f7eee4dd5734886dbd929abda7cf42c4fa1bce33 Mon Sep 17 00:00:00 2001 From: Jeff Layton Date: Tue, 25 Aug 2026 12:04:17 -0400 Subject: [PATCH 0020/1012] btrfs: split btrfs_insert_delayed_dir_index() into prealloc and commit phases Split btrfs_insert_delayed_dir_index() into three functions using a new btrfs_dir_index_prealloc struct to bundle the pre-allocated resources: - btrfs_prealloc_delayed_dir_index(): allocates the struct and performs the two GFP_NOFS allocations (delayed node + delayed item) that can fail with -ENOMEM. Returns the struct, or ERR_PTR on failure. - btrfs_insert_delayed_dir_index_prealloc(): populates the item data, inserts into the rb-tree, and reserves metadata space. Cannot fail with -ENOMEM since all allocations were done in the prealloc step. - btrfs_free_delayed_dir_index_prealloc(): frees pre-allocated resources when the caller's btree insertion fails. Tolerates NULL. The prealloc is returned as a pointer rather than filled into a caller-provided struct, so that a plain NULL means "no prealloc" and callers do not need a separate flag to track whether one exists. It is consumed (and freed) by either the commit or the free helper, so ownership is unambiguous. The original btrfs_insert_delayed_dir_index() is refactored into a thin wrapper that calls the prealloc and commit functions. This split allows callers to move the fallible memory allocations before the point of no return (the DIR_ITEM btree insertion), so that -ENOMEM can be returned cleanly without aborting the transaction. Assisted-by: LLM Reviewed-by: Qu Wenruo Signed-off-by: Jeff Layton Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/delayed-inode.c | 128 +++++++++++++++++++++++++++++++-------- fs/btrfs/delayed-inode.h | 17 ++++++ 2 files changed, 121 insertions(+), 24 deletions(-) diff --git a/fs/btrfs/delayed-inode.c b/fs/btrfs/delayed-inode.c index db2ffab0941a6a..21a2d123df14f1 100644 --- a/fs/btrfs/delayed-inode.c +++ b/fs/btrfs/delayed-inode.c @@ -6,6 +6,7 @@ #include #include +#include #include "ctree.h" #include "fs.h" #include "messages.h" @@ -1469,35 +1470,93 @@ static void btrfs_release_dir_index_item_space(struct btrfs_trans_handle *trans) trans->bytes_reserved -= bytes; } -/* Will return 0, -ENOMEM or -EEXIST (index number collision, unexpected). */ -int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, - const char *name, int name_len, - struct btrfs_inode *dir, - const struct btrfs_disk_key *disk_key, u8 flags, - u64 index) +/* + * Pre-allocate a delayed node and delayed item for a dir index insertion and + * copy the name into the item. Call this before modifying the btree so that + * ENOMEM can be returned before any on-disk state has changed. + * + * The returned prealloc is consumed by either + * btrfs_insert_delayed_dir_index_prealloc() or + * btrfs_free_delayed_dir_index_prealloc(); it must not be used afterwards. + * + * Returns a prealloc on success, ERR_PTR on allocation failure. + */ +struct btrfs_dir_index_prealloc *btrfs_prealloc_delayed_dir_index(struct btrfs_inode *dir, + const char *name, + int name_len) { + struct btrfs_dir_index_prealloc *prealloc; + struct btrfs_delayed_node *node; + struct btrfs_delayed_item *item; + + prealloc = kzalloc_obj(*prealloc, GFP_NOFS); + if (!prealloc) + return ERR_PTR(-ENOMEM); + + node = btrfs_get_or_create_delayed_node(dir, &prealloc->tracker); + if (IS_ERR(node)) { + kfree(prealloc); + return ERR_CAST(node); + } + + item = btrfs_alloc_delayed_item(sizeof(struct btrfs_dir_item) + name_len, + node, BTRFS_DELAYED_INSERTION_ITEM); + if (!item) { + btrfs_release_delayed_node(node, &prealloc->tracker); + kfree(prealloc); + return ERR_PTR(-ENOMEM); + } + + memcpy(item->data + sizeof(struct btrfs_dir_item), name, name_len); + + prealloc->node = node; + prealloc->item = item; + return prealloc; +} +ALLOW_ERROR_INJECTION(btrfs_prealloc_delayed_dir_index, ERRNO); + +/* + * Free resources from btrfs_prealloc_delayed_dir_index() when the btree + * insertion failed and we will not commit the delayed dir index. Does nothing + * if @prealloc is NULL. + */ +void btrfs_free_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, + struct btrfs_dir_index_prealloc *prealloc) +{ + if (!prealloc) + return; + + btrfs_release_delayed_item(prealloc->item); + btrfs_release_dir_index_item_space(trans); + btrfs_release_delayed_node(prealloc->node, &prealloc->tracker); + kfree(prealloc); +} + +/* + * Commit a pre-allocated delayed dir index item. @prealloc must have been + * returned by btrfs_prealloc_delayed_dir_index(). This populates the item, + * adds it to the delayed node's rb-tree, and reserves metadata space. It + * cannot fail with ENOMEM. @prealloc is freed here in all cases. + * + * Return 0 or -EEXIST (index number collision, unexpected). + */ +int btrfs_insert_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, + struct btrfs_inode *dir, + struct btrfs_dir_index_prealloc *prealloc, + const struct btrfs_disk_key *disk_key, + u8 flags, u64 index) +{ + struct btrfs_delayed_node *delayed_node = prealloc->node; + struct btrfs_ref_tracker *tracker = &prealloc->tracker; + struct btrfs_delayed_item *delayed_item = prealloc->item; struct btrfs_fs_info *fs_info = trans->fs_info; const unsigned int leaf_data_size = BTRFS_LEAF_DATA_SIZE(fs_info); - struct btrfs_delayed_node *delayed_node; - struct btrfs_ref_tracker delayed_node_tracker; - struct btrfs_delayed_item *delayed_item; + const int name_len = delayed_item->data_len - sizeof(struct btrfs_dir_item); struct btrfs_dir_item *dir_item; bool reserve_leaf_space; u32 data_len; int ret; - delayed_node = btrfs_get_or_create_delayed_node(dir, &delayed_node_tracker); - if (IS_ERR(delayed_node)) - return PTR_ERR(delayed_node); - - delayed_item = btrfs_alloc_delayed_item(sizeof(*dir_item) + name_len, - delayed_node, - BTRFS_DELAYED_INSERTION_ITEM); - if (!delayed_item) { - ret = -ENOMEM; - goto release_node; - } - delayed_item->index = index; dir_item = (struct btrfs_dir_item *)delayed_item->data; @@ -1506,7 +1565,7 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, btrfs_set_stack_dir_data_len(dir_item, 0); btrfs_set_stack_dir_name_len(dir_item, name_len); btrfs_set_stack_dir_flags(dir_item, flags); - memcpy((char *)(dir_item + 1), name, name_len); + /* Name was already copied by btrfs_prealloc_delayed_dir_index(). */ data_len = delayed_item->data_len + sizeof(struct btrfs_item); @@ -1524,7 +1583,9 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, if (unlikely(ret)) { btrfs_err(trans->fs_info, "error adding delayed dir index item, name: %.*s, index: %llu, root: %llu, dir: %llu, dir->index_cnt: %llu, delayed_node->index_cnt: %llu, error: %pe", - name_len, name, index, btrfs_root_id(delayed_node->root), + name_len, + (const char *)(dir_item + 1), + index, btrfs_root_id(delayed_node->root), delayed_node->inode_id, dir->index_cnt, delayed_node->index_cnt, ERR_PTR(ret)); btrfs_release_delayed_item(delayed_item); @@ -1562,10 +1623,29 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, mutex_unlock(&delayed_node->mutex); release_node: - btrfs_release_delayed_node(delayed_node, &delayed_node_tracker); + /* Must release the node before freeing @tracker's containing struct. */ + btrfs_release_delayed_node(delayed_node, tracker); + kfree(prealloc); return ret; } +/* Return 0, -ENOMEM or -EEXIST (index number collision, unexpected). */ +int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, + const char *name, int name_len, + struct btrfs_inode *dir, + const struct btrfs_disk_key *disk_key, u8 flags, + u64 index) +{ + struct btrfs_dir_index_prealloc *prealloc; + + prealloc = btrfs_prealloc_delayed_dir_index(dir, name, name_len); + if (IS_ERR(prealloc)) + return PTR_ERR(prealloc); + + return btrfs_insert_delayed_dir_index_prealloc(trans, dir, prealloc, + disk_key, flags, index); +} + static bool btrfs_delete_delayed_insertion_item(struct btrfs_delayed_node *node, u64 index) { diff --git a/fs/btrfs/delayed-inode.h b/fs/btrfs/delayed-inode.h index fc752863f89bcd..6d12a145489f2b 100644 --- a/fs/btrfs/delayed-inode.h +++ b/fs/btrfs/delayed-inode.h @@ -121,6 +121,23 @@ int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, const struct btrfs_disk_key *disk_key, u8 flags, u64 index); +struct btrfs_dir_index_prealloc { + struct btrfs_delayed_node *node; + struct btrfs_ref_tracker tracker; + struct btrfs_delayed_item *item; +}; + +struct btrfs_dir_index_prealloc *btrfs_prealloc_delayed_dir_index(struct btrfs_inode *dir, + const char *name, + int name_len); +void btrfs_free_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, + struct btrfs_dir_index_prealloc *prealloc); +int btrfs_insert_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, + struct btrfs_inode *dir, + struct btrfs_dir_index_prealloc *prealloc, + const struct btrfs_disk_key *disk_key, + u8 flags, u64 index); + int btrfs_delete_delayed_dir_index(struct btrfs_trans_handle *trans, struct btrfs_inode *dir, u64 index); From 954e3439a88a6c6b4c220efaf93d6a993839300e Mon Sep 17 00:00:00 2001 From: Jeff Layton Date: Tue, 25 Aug 2026 12:04:18 -0400 Subject: [PATCH 0021/1012] btrfs: pre-allocate delayed dir index before btree modification Move the delayed dir index allocation in btrfs_insert_dir_item() before the insert_with_overflow() call that modifies the btree. Previously, the allocations happened after the DIR_ITEM was already inserted, meaning an ENOMEM failure left the btree in a partially-modified state that could only be resolved by aborting the transaction. Add an optional caller-provided btrfs_dir_index_prealloc parameter to btrfs_insert_dir_item(). When non-NULL, ownership of the prealloc transfers to btrfs_insert_dir_item(). When NULL, it allocates internally. All existing callers pass NULL to preserve the current behavior. Since ownership transfers, btrfs_insert_dir_item() must free the prealloc on every path that does not commit it. Route all such exits (including the early path allocation failure) through a common out_free_prealloc label, rather than keying cleanup on need_delayed_index. Remove the btrfs_insert_delayed_dir_index() wrapper, as there are no more callers. Assisted-by: LLM Suggested-by: Qu Wenruo Signed-off-by: Jeff Layton Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/delayed-inode.c | 21 ++------------------ fs/btrfs/delayed-inode.h | 5 ----- fs/btrfs/dir-item.c | 41 ++++++++++++++++++++++++++-------------- fs/btrfs/dir-item.h | 5 +++-- fs/btrfs/inode.c | 2 +- fs/btrfs/transaction.c | 3 +-- 6 files changed, 34 insertions(+), 43 deletions(-) diff --git a/fs/btrfs/delayed-inode.c b/fs/btrfs/delayed-inode.c index 21a2d123df14f1..1f20148c1fd44e 100644 --- a/fs/btrfs/delayed-inode.c +++ b/fs/btrfs/delayed-inode.c @@ -687,7 +687,7 @@ static int btrfs_insert_delayed_item(struct btrfs_trans_handle *trans, /* * For delayed items to insert, we track reserved metadata bytes based * on the number of leaves that we will use. - * See btrfs_insert_delayed_dir_index() and + * See btrfs_insert_delayed_dir_index_prealloc() and * btrfs_delayed_item_reserve_metadata()). */ ASSERT(first_item->bytes_reserved == 0); @@ -1629,23 +1629,6 @@ int btrfs_insert_delayed_dir_index_prealloc(struct btrfs_trans_handle *trans, return ret; } -/* Return 0, -ENOMEM or -EEXIST (index number collision, unexpected). */ -int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, - const char *name, int name_len, - struct btrfs_inode *dir, - const struct btrfs_disk_key *disk_key, u8 flags, - u64 index) -{ - struct btrfs_dir_index_prealloc *prealloc; - - prealloc = btrfs_prealloc_delayed_dir_index(dir, name, name_len); - if (IS_ERR(prealloc)) - return PTR_ERR(prealloc); - - return btrfs_insert_delayed_dir_index_prealloc(trans, dir, prealloc, - disk_key, flags, index); -} - static bool btrfs_delete_delayed_insertion_item(struct btrfs_delayed_node *node, u64 index) { @@ -1661,7 +1644,7 @@ static bool btrfs_delete_delayed_insertion_item(struct btrfs_delayed_node *node, /* * For delayed items to insert, we track reserved metadata bytes based * on the number of leaves that we will use. - * See btrfs_insert_delayed_dir_index() and + * See btrfs_insert_delayed_dir_index_prealloc() and * btrfs_delayed_item_reserve_metadata()). */ ASSERT(item->bytes_reserved == 0); diff --git a/fs/btrfs/delayed-inode.h b/fs/btrfs/delayed-inode.h index 6d12a145489f2b..57ba96cfaf9cb3 100644 --- a/fs/btrfs/delayed-inode.h +++ b/fs/btrfs/delayed-inode.h @@ -115,11 +115,6 @@ struct btrfs_delayed_item { }; void btrfs_init_delayed_root(struct btrfs_delayed_root *delayed_root); -int btrfs_insert_delayed_dir_index(struct btrfs_trans_handle *trans, - const char *name, int name_len, - struct btrfs_inode *dir, - const struct btrfs_disk_key *disk_key, u8 flags, - u64 index); struct btrfs_dir_index_prealloc { struct btrfs_delayed_node *node; diff --git a/fs/btrfs/dir-item.c b/fs/btrfs/dir-item.c index 84f1c64423d328..3a90736915af6f 100644 --- a/fs/btrfs/dir-item.c +++ b/fs/btrfs/dir-item.c @@ -106,8 +106,11 @@ int btrfs_insert_xattr_item(struct btrfs_trans_handle *trans, * Will return 0 or -ENOMEM */ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, - const struct fscrypt_str *name, struct btrfs_inode *dir, - const struct btrfs_key *location, u8 type, u64 index) + const struct fscrypt_str *name, + struct btrfs_inode *dir, + const struct btrfs_key *location, u8 type, + u64 index, + struct btrfs_dir_index_prealloc *prealloc) { int ret = 0; int ret2 = 0; @@ -119,17 +122,27 @@ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, struct btrfs_key key; struct btrfs_disk_key disk_key; u32 data_size; + const bool need_delayed_index = (root != root->fs_info->tree_root); key.objectid = btrfs_ino(dir); key.type = BTRFS_DIR_ITEM_KEY; key.offset = btrfs_name_hash(name->name, name->len); path = btrfs_alloc_path(); - if (!path) - return -ENOMEM; + if (!path) { + ret = -ENOMEM; + goto out_free_prealloc; + } btrfs_cpu_key_to_disk(&disk_key, location); + /* Pre-allocate the delayed dir index before modifying the btree. */ + if (need_delayed_index && !prealloc) { + prealloc = btrfs_prealloc_delayed_dir_index(dir, name->name, name->len); + if (IS_ERR(prealloc)) + return PTR_ERR(prealloc); + } + data_size = sizeof(*dir_item) + name->len; dir_item = insert_with_overflow(trans, root, path, &key, data_size, name->name, name->len); @@ -137,7 +150,7 @@ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, ret = PTR_ERR(dir_item); if (ret == -EEXIST) goto second_insert; - goto out_free; + goto out_free_prealloc; } if (IS_ENCRYPTED(&dir->vfs_inode)) @@ -154,21 +167,21 @@ int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, write_extent_buffer(leaf, name->name, name_ptr, name->len); second_insert: - /* FIXME, use some real flag for selecting the extra index */ - if (root == root->fs_info->tree_root) { + if (!need_delayed_index) { ret = 0; - goto out_free; + goto out_free_prealloc; } btrfs_release_path(path); - ret2 = btrfs_insert_delayed_dir_index(trans, name->name, name->len, dir, - &disk_key, type, index); -out_free: + ret2 = btrfs_insert_delayed_dir_index_prealloc(trans, dir, prealloc, + &disk_key, type, index); if (ret) return ret; - if (ret2) - return ret2; - return 0; + return ret2; + +out_free_prealloc: + btrfs_free_delayed_dir_index_prealloc(trans, prealloc); + return ret; } static struct btrfs_dir_item *btrfs_lookup_match_dir( diff --git a/fs/btrfs/dir-item.h b/fs/btrfs/dir-item.h index e52174a8baf92c..8d22976a0a7712 100644 --- a/fs/btrfs/dir-item.h +++ b/fs/btrfs/dir-item.h @@ -13,12 +13,14 @@ struct btrfs_path; struct btrfs_inode; struct btrfs_root; struct btrfs_trans_handle; +struct btrfs_dir_index_prealloc; int btrfs_check_dir_item_collision(struct btrfs_root *root, u64 dir_ino, const struct fscrypt_str *name); int btrfs_insert_dir_item(struct btrfs_trans_handle *trans, const struct fscrypt_str *name, struct btrfs_inode *dir, - const struct btrfs_key *location, u8 type, u64 index); + const struct btrfs_key *location, u8 type, u64 index, + struct btrfs_dir_index_prealloc *prealloc); struct btrfs_dir_item *btrfs_lookup_dir_item(struct btrfs_trans_handle *trans, struct btrfs_root *root, struct btrfs_path *path, u64 dir, @@ -53,5 +55,4 @@ static inline u64 btrfs_name_hash(const char *name, int len) { return crc32c((u32)~1, name, len); } - #endif diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 2967a307d8f066..babc291751e96f 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -6889,7 +6889,7 @@ int btrfs_add_link(struct btrfs_trans_handle *trans, return ret; ret = btrfs_insert_dir_item(trans, name, parent_inode, &key, - btrfs_inode_type(inode), index); + btrfs_inode_type(inode), index, NULL); if (ret == -EEXIST || ret == -EOVERFLOW) goto fail_dir_item; else if (unlikely(ret)) { diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index 6802b94ed76ff4..ca114235bbe1e1 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -1890,8 +1890,7 @@ static noinline int create_pending_snapshot(struct btrfs_trans_handle *trans, goto fail; ret = btrfs_insert_dir_item(trans, &fname.disk_name, - parent_inode, &key, BTRFS_FT_DIR, - index); + parent_inode, &key, BTRFS_FT_DIR, index, NULL); if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto fail; From 22b7d526c54c12ada00f8328544fb305ebd1b05f Mon Sep 17 00:00:00 2001 From: Jeff Layton Date: Tue, 25 Aug 2026 12:04:19 -0400 Subject: [PATCH 0022/1012] btrfs: handle ENOMEM from btrfs_insert_dir_item() without aborting Now that btrfs_insert_dir_item() returns -ENOMEM before modifying the btree (thanks to delayed dir index pre-allocation), callers can handle ENOMEM gracefully instead of aborting the transaction. - btrfs_add_link(): add -ENOMEM to the recoverable errors alongside -EEXIST and -EOVERFLOW. - btrfs_create_new_inode(): on -ENOMEM from btrfs_add_link(), orphan the newly-created inode instead of aborting. The inode item was already written with nlink 1, and discard_new_inode() marks it bad so eviction won't delete it. So clear_nlink() alone is not enough: persist nlink 0 via btrfs_update_inode(), otherwise orphan cleanup would see nlink > 0, drop the orphan item, and leak the inode. Fall back to aborting only if that update also fails. This turns a filesystem-killing abort into a graceful -ENOMEM return for create(), mkdir(), mknod(), symlink(), and link() under memory pressure. Assisted-by: LLM Signed-off-by: Jeff Layton Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/inode.c | 24 ++++++++++++++++++++++-- 1 file changed, 22 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index babc291751e96f..67edf6618bda34 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -6828,7 +6828,27 @@ int btrfs_create_new_inode(struct btrfs_trans_handle *trans, } else { ret = btrfs_add_link(trans, BTRFS_I(dir), BTRFS_I(inode), name, false, BTRFS_I(inode)->dir_index); - if (unlikely(ret)) { + if (ret == -ENOMEM) { + /* + * Orphan the new inode instead of aborting. The inode + * item was already written with nlink 1, and discard's + * eviction won't delete a bad inode, so nlink 0 must be + * persisted here or orphan cleanup would see nlink > 0, + * drop the orphan item, and leak the inode. + */ + clear_nlink(inode); + /* btrfs_orphan_add() aborts the transaction on failure. */ + ret = btrfs_orphan_add(trans, BTRFS_I(inode)); + if (ret) + goto discard; + ret = btrfs_update_inode(trans, BTRFS_I(inode)); + if (ret) { + btrfs_abort_transaction(trans, ret); + goto discard; + } + ret = -ENOMEM; + goto discard; + } else if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto discard; } @@ -6890,7 +6910,7 @@ int btrfs_add_link(struct btrfs_trans_handle *trans, ret = btrfs_insert_dir_item(trans, name, parent_inode, &key, btrfs_inode_type(inode), index, NULL); - if (ret == -EEXIST || ret == -EOVERFLOW) + if (ret == -EEXIST || ret == -EOVERFLOW || ret == -ENOMEM) goto fail_dir_item; else if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); From e642e0704a4fcc714541a349915ba9052c8b275a Mon Sep 17 00:00:00 2001 From: Jeff Layton Date: Tue, 25 Aug 2026 12:04:20 -0400 Subject: [PATCH 0023/1012] btrfs: pre-allocate delayed dir index for non-overwrite rename For rename() without an overwrite target, pre-allocate the delayed dir index before any btree modifications so that ENOMEM can be returned before the source is unlinked from the old directory. Add a prealloc parameter to btrfs_add_link() that allows callers to pass pre-allocated delayed dir index resources. When provided, btrfs_add_link() takes ownership: it either passes the prealloc to btrfs_insert_dir_item() (which commits or frees it), or frees it on early error. All existing callers pass NULL to preserve the current behavior. In btrfs_rename(), when new_inode is NULL (no overwrite), call btrfs_prealloc_delayed_dir_index() before the first btree modification and pass the result through to btrfs_add_link(). If the prealloc fails, -ENOMEM is returned before any btree state has changed. The local prealloc pointer is cleared once ownership passes to btrfs_add_link(), so the out_fail path only frees one we still own. For overwrite rename (new_inode != NULL), the transaction still aborts on ENOMEM since earlier unlink operations have already made irreversible btree modifications. Assisted-by: LLM Signed-off-by: Jeff Layton Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/btrfs_inode.h | 4 +++- fs/btrfs/inode.c | 39 +++++++++++++++++++++++++++++++-------- fs/btrfs/tree-log.c | 4 ++-- 3 files changed, 36 insertions(+), 11 deletions(-) diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 89e5e9c0c904f6..46c62f980c24ac 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -32,6 +32,7 @@ struct btrfs_trans_handle; struct btrfs_bio; struct btrfs_file_extent; struct btrfs_delayed_node; +struct btrfs_dir_index_prealloc; /* * Since we search a directory based on f_pos (struct dir_context::pos) we have @@ -523,7 +524,8 @@ int btrfs_unlink_inode(struct btrfs_trans_handle *trans, const struct fscrypt_str *name); int btrfs_add_link(struct btrfs_trans_handle *trans, struct btrfs_inode *parent_inode, struct btrfs_inode *inode, - const struct fscrypt_str *name, bool add_backref, u64 index); + const struct fscrypt_str *name, bool add_backref, u64 index, + struct btrfs_dir_index_prealloc *prealloc); int btrfs_delete_subvolume(struct btrfs_inode *dir, struct dentry *dentry); int btrfs_truncate_block(struct btrfs_inode *inode, u64 offset, u64 start, u64 end); diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 67edf6618bda34..59e92908c6c593 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -6827,7 +6827,7 @@ int btrfs_create_new_inode(struct btrfs_trans_handle *trans, } } else { ret = btrfs_add_link(trans, BTRFS_I(dir), BTRFS_I(inode), name, - false, BTRFS_I(inode)->dir_index); + false, BTRFS_I(inode)->dir_index, NULL); if (ret == -ENOMEM) { /* * Orphan the new inode instead of aborting. The inode @@ -6879,7 +6879,8 @@ int btrfs_create_new_inode(struct btrfs_trans_handle *trans, */ int btrfs_add_link(struct btrfs_trans_handle *trans, struct btrfs_inode *parent_inode, struct btrfs_inode *inode, - const struct fscrypt_str *name, bool add_backref, u64 index) + const struct fscrypt_str *name, bool add_backref, u64 index, + struct btrfs_dir_index_prealloc *prealloc) { int ret = 0; struct btrfs_key key; @@ -6905,11 +6906,13 @@ int btrfs_add_link(struct btrfs_trans_handle *trans, } /* Nothing to clean up yet */ - if (ret) + if (ret) { + btrfs_free_delayed_dir_index_prealloc(trans, prealloc); return ret; + } ret = btrfs_insert_dir_item(trans, name, parent_inode, &key, - btrfs_inode_type(inode), index, NULL); + btrfs_inode_type(inode), index, prealloc); if (ret == -EEXIST || ret == -EOVERFLOW || ret == -ENOMEM) goto fail_dir_item; else if (unlikely(ret)) { @@ -7063,7 +7066,7 @@ static int btrfs_link(struct dentry *old_dentry, struct inode *dir, inode_set_ctime_current(inode); ret = btrfs_add_link(trans, BTRFS_I(dir), BTRFS_I(inode), - &fname.disk_name, true, index); + &fname.disk_name, true, index, NULL); if (ret) goto fail; @@ -8477,14 +8480,14 @@ static int btrfs_rename_exchange(struct inode *old_dir, } ret = btrfs_add_link(trans, BTRFS_I(new_dir), BTRFS_I(old_inode), - new_name, false, old_idx); + new_name, false, old_idx, NULL); if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto out_fail; } ret = btrfs_add_link(trans, BTRFS_I(old_dir), BTRFS_I(new_inode), - old_name, false, new_idx); + old_name, false, new_idx, NULL); if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto out_fail; @@ -8557,6 +8560,7 @@ static int btrfs_rename(struct mnt_idmap *idmap, struct inode *new_inode = d_inode(new_dentry); struct inode *old_inode = d_inode(old_dentry); struct btrfs_rename_ctx rename_ctx; + struct btrfs_dir_index_prealloc *prealloc = NULL; u64 index = 0; int ret; int ret2; @@ -8680,6 +8684,23 @@ static int btrfs_rename(struct mnt_idmap *idmap, if (ret) goto out_fail; + /* + * When not overwriting an existing entry, pre-allocate the delayed dir + * index now so that ENOMEM is returned before any btree modifications. + * For the overwrite case, too many btree changes have already happened + * by the time btrfs_add_link() is called. + */ + if (!new_inode) { + prealloc = btrfs_prealloc_delayed_dir_index(BTRFS_I(new_dir), + new_fname.disk_name.name, + new_fname.disk_name.len); + if (IS_ERR(prealloc)) { + ret = PTR_ERR(prealloc); + prealloc = NULL; + goto out_fail; + } + } + BTRFS_I(old_inode)->dir_index = 0ULL; if (unlikely(old_ino == BTRFS_FIRST_FREE_OBJECTID)) { /* force full log commit if subvolume involved. */ @@ -8775,7 +8796,8 @@ static int btrfs_rename(struct mnt_idmap *idmap, } ret = btrfs_add_link(trans, BTRFS_I(new_dir), BTRFS_I(old_inode), - &new_fname.disk_name, false, index); + &new_fname.disk_name, false, index, prealloc); + prealloc = NULL; if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); goto out_fail; @@ -8800,6 +8822,7 @@ static int btrfs_rename(struct mnt_idmap *idmap, } } out_fail: + btrfs_free_delayed_dir_index_prealloc(trans, prealloc); if (logs_pinned) { btrfs_end_log_trans(root); btrfs_end_log_trans(dest); diff --git a/fs/btrfs/tree-log.c b/fs/btrfs/tree-log.c index a00094604e544e..cfb0b0e9e1248e 100644 --- a/fs/btrfs/tree-log.c +++ b/fs/btrfs/tree-log.c @@ -1683,7 +1683,7 @@ static noinline int add_inode_ref(struct walk_control *wc) } /* insert our name */ - ret = btrfs_add_link(trans, dir, inode, &name, false, ref_index); + ret = btrfs_add_link(trans, dir, inode, &name, false, ref_index, NULL); if (ret) { btrfs_abort_log_replay(wc, ret, "failed to add link for inode %llu in dir %llu ref_index %llu name %.*s root %llu", @@ -2031,7 +2031,7 @@ static noinline int insert_one_name(struct btrfs_trans_handle *trans, return PTR_ERR(dir); } - ret = btrfs_add_link(trans, dir, inode, name, true, index); + ret = btrfs_add_link(trans, dir, inode, name, true, index, NULL); /* FIXME, put inode into FIXUP list */ From 6c0b211fe098114fe80cabddcab74351ecfa030d Mon Sep 17 00:00:00 2001 From: Usama Arif Date: Fri, 4 Sep 2026 09:41:48 -0700 Subject: [PATCH 0024/1012] btrfs: zstd: avoid a copy in zstd_decompress_bio() zstd_decompress_bio() gives zstd a sectorsize-sized scratch buffer, and btrfs_decompress_buf2page() then copies the part overlapping the read bio into the destination folios. Every delivered byte is written twice. Instead, choose the output buffer per streaming call. zstd_map_dest() kmaps the current page-bounded segment of the read bio, so zstd writes into the page cache directly. The scratch buffer is kept only for output with no destination: the prefix before a read starting inside a compressed extent, which zstd cannot skip, and gaps left by folios already in the page cache. Varying the output buffer across calls is safe: btrfs uses the default ZSTD_bm_buffered mode, where the sliding window lives in the dstream's internal buffer and the caller's dst is a pure sink. The read bio's iterator must still advance by exactly the bytes delivered, since btrfs_decompress_bio() zero-fills from it; that used to happen inside btrfs_decompress_buf2page() and is now an explicit bio_advance(), made only for output that reached a folio. bio_iter_iovec() exposes at most one base page, so direct output is page-bounded. Compared to the old sectorsize-sized chunks, this can increase stream calls when sectorsize exceeds PAGE_SIZE, but eliminates the extra btrfs copy for output delivered to the read bio; the 64 KiB sectorsize row below shows the copy still wins there. Benchmarked the change in 2-vCPU x86-64 KVM guests (4 KiB pages, RAM disk) using a 64 MiB zstd-compressed file. Results are medians of seven cold-cache reads in each of six interleaved A/B boot pairs; mincore confirmed zero resident pages before every run. Normal sequential reads with readahead produced: sectorsize base patched reduction 4 KiB 8.678 ms 8.004 ms 7.80% 16 KiB 8.216 ms 7.934 ms 3.64% 64 KiB 7.875 ms 7.344 ms 6.88% Random 4 KiB preads at 4 KiB sectorsize, means of six interleaved A/B boot pairs, patched better in all six: base patched gain 264.33 MB/s 272.67 MB/s 3.2% Signed-off-by: Usama Arif Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/zstd.c | 69 +++++++++++++++++++++++++++++++++++++++---------- 1 file changed, 56 insertions(+), 13 deletions(-) diff --git a/fs/btrfs/zstd.c b/fs/btrfs/zstd.c index 58d9ff76fe07bb..280ac5273438b7 100644 --- a/fs/btrfs/zstd.c +++ b/fs/btrfs/zstd.c @@ -589,10 +589,48 @@ int zstd_compress_bio(struct list_head *ws, struct compressed_bio *cb) return ret; } +/* + * Map the destination for the next chunk of output. + * + * @decompressed is the offset of the next output byte inside the fully + * decompressed extent. If that offset has reached the current destination + * segment, its page-bounded bio_vec is kmapped so that zstd can write into the + * page cache directly, and the number of bytes writable there is returned. + * Otherwise @kaddr_ret is set to NULL and the number of bytes to skip before + * that segment is returned. This covers both the initial prefix and gaps in + * the destination bio. + */ +static u32 zstd_map_dest(struct compressed_bio *cb, u32 decompressed, + void **kaddr_ret) +{ + struct bio *orig_bio = &cb->orig_bbio->bio; + struct bio_vec bvec; + u32 bvec_offset; + u32 off; + + bvec = bio_iter_iovec(orig_bio, orig_bio->bi_iter); + /* + * cb->start may underflow, but subtracting that value can still give us + * the correct offset inside the full decompressed extent. + */ + bvec_offset = page_offset(bvec.bv_page) + bvec.bv_offset - cb->start; + + if (decompressed < bvec_offset) { + *kaddr_ret = NULL; + return bvec_offset - decompressed; + } + + off = decompressed - bvec_offset; + ASSERT(off < bvec.bv_len); + *kaddr_ret = bvec_kmap_local(&bvec) + off; + return bvec.bv_len - off; +} + int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb) { struct btrfs_fs_info *fs_info = cb_to_fs_info(cb); struct workspace *workspace = list_entry(ws, struct workspace, list); + struct bio *orig_bio = &cb->orig_bbio->bio; struct folio_iter fi; size_t srclen = bio_get_size(&cb->bbio.bio); zstd_dstream *stream; @@ -600,7 +638,6 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb) const unsigned int min_folio_size = btrfs_min_folio_size(fs_info); unsigned long folio_in_index = 0; unsigned long total_folios_in = DIV_ROUND_UP(srclen, min_folio_size); - unsigned long buf_start; unsigned long total_out = 0; bio_first_folio(&fi, &cb->bbio.bio, 0); @@ -624,15 +661,26 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb) workspace->in_buf.pos = 0; workspace->in_buf.size = min_t(size_t, srclen, min_folio_size); - workspace->out_buf.dst = workspace->buf; - workspace->out_buf.pos = 0; - workspace->out_buf.size = fs_info->sectorsize; - - while (1) { + while (orig_bio->bi_iter.bi_size) { size_t ret2; + void *kaddr; + u32 dstlen; + + dstlen = zstd_map_dest(cb, total_out, &kaddr); + if (kaddr) { + workspace->out_buf.dst = kaddr; + workspace->out_buf.size = dstlen; + } else { + workspace->out_buf.dst = workspace->buf; + workspace->out_buf.size = min_t(u32, dstlen, + fs_info->sectorsize); + } + workspace->out_buf.pos = 0; ret2 = zstd_decompress_stream(stream, &workspace->out_buf, &workspace->in_buf); + if (kaddr) + kunmap_local(kaddr); if (unlikely(zstd_is_error(ret2))) { struct btrfs_inode *inode = cb->bbio.inode; @@ -643,14 +691,9 @@ int zstd_decompress_bio(struct list_head *ws, struct compressed_bio *cb) ret = -EIO; goto done; } - buf_start = total_out; total_out += workspace->out_buf.pos; - workspace->out_buf.pos = 0; - - ret = btrfs_decompress_buf2page(workspace->out_buf.dst, - total_out - buf_start, cb, buf_start); - if (ret == 0) - break; + if (kaddr) + bio_advance(orig_bio, workspace->out_buf.pos); if (workspace->in_buf.pos >= srclen) break; From f07031a890eef45f702cde01e5d5d007883ac857 Mon Sep 17 00:00:00 2001 From: Zhen Ni Date: Mon, 22 Dec 2025 11:59:42 +0800 Subject: [PATCH 0025/1012] btrfs: replace is_data_bbio() with is_data_inode() for direct usage After commit 81cea6cd7041 ("btrfs: remove btrfs_bio::fs_info by extracting it from btrfs_bio::inode"), the btrfs_bio::inode field is mandatory for all btrfs_bio allocations. The NULL check is redundant and can be removed. As is_data_bbio() would be a trivial wrapper for is_data_bbio() replace all calls in in bio.c Link: https://lore.kernel.org/linux-btrfs/20251219084316.1164580-1-zhen.ni@easystack.cn Signed-off-by: Zhen Ni Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/bio.c | 18 ++++++------------ 1 file changed, 6 insertions(+), 12 deletions(-) diff --git a/fs/btrfs/bio.c b/fs/btrfs/bio.c index 771b7d598aeeeb..2c6234f6182f89 100644 --- a/fs/btrfs/bio.c +++ b/fs/btrfs/bio.c @@ -27,15 +27,9 @@ struct btrfs_failed_bio { atomic_t repair_count; }; -/* Is this a data path I/O that needs storage layer checksum and repair? */ -static inline bool is_data_bbio(const struct btrfs_bio *bbio) -{ - return bbio->inode && is_data_inode(bbio->inode); -} - static bool bbio_has_ordered_extent(const struct btrfs_bio *bbio) { - return is_data_bbio(bbio) && btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE; + return is_data_inode(bbio->inode) && btrfs_op(&bbio->bio) == BTRFS_MAP_WRITE; } /* @@ -356,7 +350,7 @@ static void simple_end_io_work(struct work_struct *work) if (bio_op(bio) == REQ_OP_READ) { /* Metadata reads are checked and repaired by the submitter. */ - if (is_data_bbio(bbio)) + if (is_data_inode(bbio->inode)) return btrfs_check_read_bio(bbio, bbio->bio.bi_private); return btrfs_bio_end_io(bbio, bbio->bio.bi_status); } @@ -390,7 +384,7 @@ static void btrfs_raid56_end_io(struct bio *bio) btrfs_bio_counter_dec(bioc->fs_info); bbio->mirror_num = bioc->mirror_num; - if (bio_op(bio) == REQ_OP_READ && is_data_bbio(bbio)) + if (bio_op(bio) == REQ_OP_READ && is_data_inode(bbio->inode)) btrfs_check_read_bio(bbio, NULL); else btrfs_bio_end_io(bbio, bbio->bio.bi_status); @@ -753,7 +747,7 @@ static bool btrfs_submit_chunk(struct btrfs_bio *bbio, int mirror_num) * our bio to the physical disk location, so we need to save the * original bytenr so we know what we're checksumming. */ - if (bio_op(bio) == REQ_OP_WRITE && is_data_bbio(bbio)) + if (bio_op(bio) == REQ_OP_WRITE && is_data_inode(bbio->inode)) bbio->orig_logical = logical; bbio->can_use_append = btrfs_use_zone_append(bbio); @@ -779,7 +773,7 @@ static bool btrfs_submit_chunk(struct btrfs_bio *bbio, int mirror_num) * Save the iter for the end_io handler and preload the checksums for * data reads. */ - if (bio_op(bio) == REQ_OP_READ && is_data_bbio(bbio)) { + if (bio_op(bio) == REQ_OP_READ && is_data_inode(bbio->inode)) { bbio->saved_iter = bio->bi_iter; ret = btrfs_lookup_bio_sums(bbio); status = errno_to_blk_status(ret); @@ -788,7 +782,7 @@ static bool btrfs_submit_chunk(struct btrfs_bio *bbio, int mirror_num) } if (btrfs_op(bio) == BTRFS_MAP_WRITE) { - if (is_data_bbio(bbio) && bioc && bioc->use_rst) { + if (is_data_inode(bbio->inode) && bioc && bioc->use_rst) { /* * No locking for the list update, as we only add to * the list in the I/O submission path, and list From 2969ddfd6aacf6628e50001d9fd535a196d28c66 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 8 Sep 2026 07:47:39 +0930 Subject: [PATCH 0026/1012] btrfs: use kvmalloc() for b-tree split_item() [BUG] There is a bug report that the kmalloc() call inside split_item() failed with the following call trace, and triggered a transaction abort: kworker/u69:8: page allocation failure: order:4, mode:0x40c40(GFP_NOFS|__GFP_COMP), nodemask=(null) CPU: 3 UID: 0 PID: 1154528 Comm: kworker/u69:8 Not tainted 7.0.2 #1 PREEMPTLAZY Workqueue: events_unbound btrfs_async_reclaim_metadata_space Call Trace: dump_stack_lvl+0x47/0x60 warn_alloc.cold+0x67/0xec __alloc_pages_slowpath.constprop.0+0x9bf/0xed0 __alloc_frozen_pages_noprof+0x1ac/0x1c0 ___kmalloc_large_node+0x9d/0xc0 __kmalloc_noprof+0x17b/0x1f0 split_item+0x9e/0x2e0 btrfs_del_csums+0x285/0x400 __btrfs_free_extent.isra.0+0x6de/0x12b0 __btrfs_run_delayed_refs+0x522/0x10c0 btrfs_run_delayed_refs+0x4d/0x1d0 flush_space+0x34d/0x4e0 do_async_reclaim_metadata_space+0x89/0x1d0 btrfs_async_reclaim_metadata_space+0x44/0x60 process_one_work+0x145/0x230 worker_thread+0x185/0x2e0 kthread+0xca/0x100 ret_from_fork+0x14e/0x200 ret_from_fork_asm+0x11/0x20 BTRFS error (device dm-3 state A): Transaction aborted (error -12) BTRFS: error (device dm-3 state A) in btrfs_del_csums:1053: errno=-12 Out of memory BTRFS info (device dm-3 state EA): forced readonly BTRFS: error (device dm-3 state EA) in do_free_extent_accounting:3168: errno=-12 Out of memory BTRFS error (device dm-3 state EA): failed to run delayed ref for logical 1202913873920 num_bytes 274432 type 184 action 2 ref_mod 1: -12 BTRFS: error (device dm-3 state EA) in btrfs_run_delayed_refs:2247: errno=-12 Out of memory [CAUSE] The kmalloc() call is to allocate a buffer to store the full item. However as shown in the above call trace, the order can be high (4), and since we're using GFP_NOFS, it's impossible to reclaim memory by writing back dirty pages. When there is no physically contiguous memory left, such high order allocation can easily fail, and if such kmalloc() happens in a critical path we can trigger a transaction abort. [FIX] Instead of kmalloc(), which requires physically contiguous pages, use kvmalloc(). There is no special requirement for physically contiguous pages here, we just want virtually contiguous memory as a buffer. Reported-by: xavierbachmeyer182 Link: https://lore.kernel.org/linux-btrfs/250decb0-d940-4fe6-9b54-d06e1b293a1b@suse.com/ Reviewed-by: Johannes Thumshirn Reviewed-by: Daniel Vacek Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/ctree.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/ctree.c b/fs/btrfs/ctree.c index 8fe330d81b8ff3..fef0e49dd91882 100644 --- a/fs/btrfs/ctree.c +++ b/fs/btrfs/ctree.c @@ -3943,7 +3943,7 @@ static noinline int split_item(struct btrfs_trans_handle *trans, orig_offset = btrfs_item_offset(leaf, path->slots[0]); item_size = btrfs_item_size(leaf, path->slots[0]); - buf = kmalloc(item_size, GFP_NOFS); + buf = kvmalloc(item_size, GFP_NOFS); if (!buf) return -ENOMEM; @@ -3981,7 +3981,7 @@ static noinline int split_item(struct btrfs_trans_handle *trans, btrfs_mark_buffer_dirty(trans, leaf); BUG_ON(btrfs_leaf_free_space(leaf) < 0); - kfree(buf); + kvfree(buf); return 0; } From 881259f538ddc3a61fa0e5173e7e9e2d485ed37b Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 8 Sep 2026 16:45:39 +0930 Subject: [PATCH 0027/1012] btrfs: tree-log: use kvmalloc() for overwrite_item() The @src_copy buffer utilized inside overwrite_item() can be as large as the nodesize. For an existing btrfs with 64KiB nodesize, it means there is a high chance to fail the kmalloc() call if there is not enough physically contiguous pages. Meanwhile there is really no need for such physically contiguous pages, as we only use that buffer to compare the content of the item. Use kvmalloc() to replace the kmalloc() call. For most cases that kvmalloc() call will be easily fulfilled by regular kmalloc(), but for really large items and large nodes, kvmalloc() will have a much higher chance to get memory allocated. Reviewed-by: Daniel Vacek Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tree-log.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/tree-log.c b/fs/btrfs/tree-log.c index cfb0b0e9e1248e..7fb476b19eb457 100644 --- a/fs/btrfs/tree-log.c +++ b/fs/btrfs/tree-log.c @@ -503,7 +503,7 @@ static int overwrite_item(struct walk_control *wc) btrfs_release_path(wc->subvol_path); return 0; } - src_copy = kmalloc(item_size, GFP_NOFS); + src_copy = kvmalloc(item_size, GFP_NOFS); if (!src_copy) { btrfs_abort_log_replay(wc, -ENOMEM, "failed to allocate memory for log leaf item"); @@ -514,7 +514,7 @@ static int overwrite_item(struct walk_control *wc) dst_ptr = btrfs_item_ptr_offset(dst_eb, dst_slot); ret = memcmp_extent_buffer(dst_eb, src_copy, dst_ptr, item_size); - kfree(src_copy); + kvfree(src_copy); /* * they have the same contents, just return, this saves * us from cowing blocks in the destination tree and doing From 52c77a933186510452a15960b7e1b06a2d6b1b90 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 8 Sep 2026 16:45:40 +0930 Subject: [PATCH 0028/1012] btrfs: use kvmalloc() for uncompress_inline() Although btrfs doesn't support inlined extents larger than PAGE_SIZE for bs > ps cases, it's still possible for the experimental bs > ps support to mount a btrfs created on a system with a much larger page size, thus can still hit an inlined extent that is way larger than the current page size. E.g. a compressed inline extent which has 32K compressed size, is created on 64K page sized ARM64 with 64K sectorsize, then mounted on x86_64 with the experimental bs > ps support. In that case, when reading the compressed inline extent, we need to allocate a buffer that is the same size as the compressed inline extent (32K). That kmalloc() call will request physically contiguous memory for that 32K allocation, and if the system has a very fragmented memory space, such allocation can fail. But there is really no reason that we require such buffer to be physically contiguous, so change it to kvmalloc() to reduce the chance of allocation failure for bs > ps cases. And for all bs <= ps cases, the kvmalloc() call will just be fulfilled by kmalloc() so this will not bring any change to the most common cases. Only bs > ps will get the benefit of less memory allocation failure. Reviewed-by: Daniel Vacek Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/inode.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 59e92908c6c593..a85a7c561cf8e9 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -7133,7 +7133,7 @@ static noinline int uncompress_inline(struct btrfs_path *path, compress_type = btrfs_file_extent_compression(leaf, item); max_size = btrfs_file_extent_ram_bytes(leaf, item); inline_size = btrfs_file_extent_inline_item_len(leaf, path->slots[0]); - tmp = kmalloc(inline_size, GFP_NOFS); + tmp = kvmalloc(inline_size, GFP_NOFS); if (!tmp) return -ENOMEM; ptr = btrfs_file_extent_inline_start(item); @@ -7154,7 +7154,7 @@ static noinline int uncompress_inline(struct btrfs_path *path, if (max_size < blocksize) folio_zero_range(folio, max_size, blocksize - max_size); - kfree(tmp); + kvfree(tmp); return ret; } From 353953d3d6094be0d015001c814d57f551756081 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Tue, 8 Sep 2026 16:45:41 +0930 Subject: [PATCH 0029/1012] btrfs: use kvmalloc() to allocate compression workspace buffer for zlib and zstd With the experimental bs > ps support, the workspace buffer for both zlib and zstd can be as large as 64K, and on 4K page sized systems such kmalloc() calls have a much higher chance to fail, as that requires physically contiguous memory to fulfill such allocation. The same also applies to S390's hardware accelerated path, which requires a buffer size of 4 pages. Meanwhile lzo is already using kvmalloc() for its buffer, and there is no special requirement for any physically contiguous memory anyway. So change the zlib and zstd workspace buffer allocation to use kvmalloc() to reduce the chance of memory allocation failure. Reviewed-by: Daniel Vacek Reviewed-by: Johannes Thumshirn Signed-off-by: Qu Wenruo Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/zlib.c | 10 +++++----- fs/btrfs/zstd.c | 4 ++-- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/fs/btrfs/zlib.c b/fs/btrfs/zlib.c index 486b52db583ecb..e995d0b1452b92 100644 --- a/fs/btrfs/zlib.c +++ b/fs/btrfs/zlib.c @@ -49,7 +49,7 @@ void zlib_free_workspace(struct list_head *ws) struct workspace *workspace = list_entry(ws, struct workspace, list); kvfree(workspace->strm.workspace); - kfree(workspace->buf); + kvfree(workspace->buf); kfree(workspace); } @@ -84,13 +84,13 @@ struct list_head *zlib_alloc_workspace(struct btrfs_fs_info *fs_info, unsigned i workspace->level = level; workspace->buf = NULL; if (need_special_buffer(fs_info)) { - workspace->buf = kmalloc(ZLIB_DFLTCC_BUF_SIZE, - __GFP_NOMEMALLOC | __GFP_NORETRY | - __GFP_NOWARN | GFP_NOIO); + workspace->buf = kvmalloc(ZLIB_DFLTCC_BUF_SIZE, + __GFP_NOMEMALLOC | __GFP_NORETRY | + __GFP_NOWARN | GFP_NOIO); workspace->buf_size = ZLIB_DFLTCC_BUF_SIZE; } if (!workspace->buf) { - workspace->buf = kmalloc(fs_info->sectorsize, GFP_KERNEL); + workspace->buf = kvmalloc(fs_info->sectorsize, GFP_KERNEL); workspace->buf_size = fs_info->sectorsize; } if (!workspace->strm.workspace || !workspace->buf) diff --git a/fs/btrfs/zstd.c b/fs/btrfs/zstd.c index 280ac5273438b7..cc92d0b1b948cd 100644 --- a/fs/btrfs/zstd.c +++ b/fs/btrfs/zstd.c @@ -373,7 +373,7 @@ void zstd_free_workspace(struct list_head *ws) struct workspace *workspace = list_entry(ws, struct workspace, list); kvfree(workspace->mem); - kfree(workspace->buf); + kvfree(workspace->buf); kfree(workspace); } @@ -391,7 +391,7 @@ struct list_head *zstd_alloc_workspace(struct btrfs_fs_info *fs_info, int level) workspace->req_level = level; workspace->last_used = jiffies; workspace->mem = kvmalloc(workspace->size, GFP_KERNEL | __GFP_NOWARN); - workspace->buf = kmalloc(fs_info->sectorsize, GFP_KERNEL); + workspace->buf = kvmalloc(fs_info->sectorsize, GFP_KERNEL); if (!workspace->mem || !workspace->buf) goto fail; From 2839600893b79eb3efcf4929c6ee062ed6f08d6e Mon Sep 17 00:00:00 2001 From: Hongling Zeng Date: Mon, 31 Aug 2026 13:38:01 +0800 Subject: [PATCH 0030/1012] btrfs: take commit root semaphore when iterating in mark_block_group_to_copy() mark_block_group_to_copy() iterates over the commit root with skip_locking=true. A concurrent transaction commit can swap and free the commit root during iteration, causing use-after-free when accessing extent buffers. Fix it by using path->need_commit_sem to protect the commit root search. Fixes: 78ce9fc269af ("btrfs: zoned: mark block groups to copy for device-replace") CC: stable@vger.kernel.org Assisted-by: Codex:gpt-5.5 Reviewed-by: Johannes Thumshirn Signed-off-by: Hongling Zeng Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/dev-replace.c | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c index 22eea188a32808..cdfe093e5c5c4f 100644 --- a/fs/btrfs/dev-replace.c +++ b/fs/btrfs/dev-replace.c @@ -498,6 +498,7 @@ static int mark_block_group_to_copy(struct btrfs_fs_info *fs_info, path->reada = READA_FORWARD; path->search_commit_root = true; path->skip_locking = true; + path->need_commit_sem = true; key.objectid = src_dev->devid; key.type = BTRFS_DEV_EXTENT_KEY; From b2e91629a7b61c210089e3133dd77ff751687444 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:19:55 -0400 Subject: [PATCH 0031/1012] btrfs: tests: rename process_page_range() to process_folio_range() It already operates on folios. No functional change. Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tests/extent-io-tests.c | 20 ++++++++++---------- 1 file changed, 10 insertions(+), 10 deletions(-) diff --git a/fs/btrfs/tests/extent-io-tests.c b/fs/btrfs/tests/extent-io-tests.c index 23459cd4e50388..6eb55bfb2bd4e6 100644 --- a/fs/btrfs/tests/extent-io-tests.c +++ b/fs/btrfs/tests/extent-io-tests.c @@ -18,8 +18,8 @@ #define PROCESS_RELEASE (1U << 1) #define PROCESS_TEST_LOCKED (1U << 2) -static noinline int process_page_range(struct inode *inode, u64 start, u64 end, - unsigned long flags) +static noinline int process_folio_range(struct inode *inode, u64 start, u64 end, + unsigned long flags) { int ret; struct folio_batch fbatch; @@ -221,8 +221,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) test_start, max_bytes - 1, start, end); goto out_bits; } - if (process_page_range(inode, start, end, - PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { + if (process_folio_range(inode, start, end, + PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { test_err("there were unlocked pages in the range"); goto out_bits; } @@ -276,8 +276,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) test_start, total_dirty - 1, start, end); goto out_bits; } - if (process_page_range(inode, start, end, - PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { + if (process_folio_range(inode, start, end, + PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { test_err("pages in range were not all locked"); goto out_bits; } @@ -317,8 +317,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) test_start, test_start + PAGE_SIZE - 1, start, end); goto out_bits; } - if (process_page_range(inode, start, end, PROCESS_TEST_LOCKED | - PROCESS_UNLOCK)) { + if (process_folio_range(inode, start, end, PROCESS_TEST_LOCKED | + PROCESS_UNLOCK)) { test_err("pages in range were not all locked"); goto out_bits; } @@ -330,8 +330,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) out: if (locked_page) put_page(locked_page); - process_page_range(inode, 0, total_dirty - 1, - PROCESS_UNLOCK | PROCESS_RELEASE); + process_folio_range(inode, 0, total_dirty - 1, + PROCESS_UNLOCK | PROCESS_RELEASE); iput(inode); out_root_info: btrfs_free_dummy_root(root); From 0f411e6789a02a42f1a41d6cc5f1715dccba2153 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:19:56 -0400 Subject: [PATCH 0032/1012] btrfs: tests: convert test_find_delalloc() to use folios This removes the last btrfs callers of find_or_create_page(), find_lock_page(), SetPageDirty(), ClearPageDirty(), and get_page(), and 15 calls to compound_head(). The folio lookups return an ERR_PTR instead of NULL, so adjust the error handling. Update the comments and test messages accordingly. The test still works in PAGE_SIZE units, which relies on the test inode never getting large folios, so assert that the folios are order-0 where that matters. Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tests/extent-io-tests.c | 97 ++++++++++++++++---------------- 1 file changed, 49 insertions(+), 48 deletions(-) diff --git a/fs/btrfs/tests/extent-io-tests.c b/fs/btrfs/tests/extent-io-tests.c index 6eb55bfb2bd4e6..ee8eabac47f110 100644 --- a/fs/btrfs/tests/extent-io-tests.c +++ b/fs/btrfs/tests/extent-io-tests.c @@ -112,8 +112,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) struct btrfs_root *root = NULL; struct inode *inode = NULL; struct extent_io_tree *tmp; - struct page *page; - struct page *locked_page = NULL; + struct folio *folio; + struct folio *locked_folio = NULL; /* In this test we need at least 2 file extents at its maximum size */ u64 max_bytes = BTRFS_MAX_EXTENT_SIZE; u64 total_dirty = 2 * max_bytes; @@ -152,23 +152,27 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) btrfs_extent_io_tree_init(NULL, tmp, IO_TREE_SELFTEST); /* - * First go through and create and mark all of our pages dirty, we pin - * everything to make sure our pages don't get evicted and screw up our + * First go through and create and mark all of our folios dirty, we pin + * everything to make sure our folios don't get evicted and screw up our * test. */ for (pgoff_t index = 0; index < (total_dirty >> PAGE_SHIFT); index++) { - page = find_or_create_page(inode->i_mapping, index, GFP_KERNEL); - if (!page) { - test_err("failed to allocate test page"); - ret = -ENOMEM; + folio = __filemap_get_folio(inode->i_mapping, index, + FGP_LOCK | FGP_ACCESSED | FGP_CREAT, + GFP_KERNEL); + if (IS_ERR(folio)) { + test_err("failed to allocate test folio"); + ret = PTR_ERR(folio); goto out; } - SetPageDirty(page); + /* The ranges below assume page sized folios. */ + ASSERT(folio_order(folio) == 0); + folio_set_dirty(folio); if (index) { - unlock_page(page); + folio_unlock(folio); } else { - get_page(page); - locked_page = page; + folio_get(folio); + locked_folio = folio; } } @@ -179,8 +183,7 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) btrfs_set_extent_bit(tmp, 0, sectorsize - 1, EXTENT_DELALLOC, NULL); start = 0; end = start + PAGE_SIZE - 1; - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (!found) { test_err("should have found at least one delalloc"); goto out_bits; @@ -191,8 +194,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) goto out_bits; } btrfs_unlock_extent(tmp, start, end, NULL); - unlock_page(locked_page); - put_page(locked_page); + folio_unlock(locked_folio); + folio_put(locked_folio); /* * Test this scenario @@ -201,17 +204,17 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) * |--- search ---| */ test_start = SZ_64M; - locked_page = find_lock_page(inode->i_mapping, - test_start >> PAGE_SHIFT); - if (!locked_page) { - test_err("couldn't find the locked page"); + locked_folio = filemap_lock_folio(inode->i_mapping, test_start >> PAGE_SHIFT); + if (IS_ERR(locked_folio)) { + test_err("couldn't find the locked folio"); + locked_folio = NULL; goto out_bits; } + ASSERT(folio_order(locked_folio) == 0); btrfs_set_extent_bit(tmp, sectorsize, max_bytes - 1, EXTENT_DELALLOC, NULL); start = test_start; end = start + PAGE_SIZE - 1; - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (!found) { test_err("couldn't find delalloc in our range"); goto out_bits; @@ -223,12 +226,12 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) } if (process_folio_range(inode, start, end, PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { - test_err("there were unlocked pages in the range"); + test_err("there were unlocked folios in the range"); goto out_bits; } btrfs_unlock_extent(tmp, start, end, NULL); - /* locked_page was unlocked above */ - put_page(locked_page); + /* locked_folio was unlocked above */ + folio_put(locked_folio); /* * Test this scenario @@ -236,16 +239,16 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) * |--- search ---| */ test_start = max_bytes + sectorsize; - locked_page = find_lock_page(inode->i_mapping, test_start >> - PAGE_SHIFT); - if (!locked_page) { - test_err("couldn't find the locked page"); + locked_folio = filemap_lock_folio(inode->i_mapping, test_start >> PAGE_SHIFT); + if (IS_ERR(locked_folio)) { + test_err("couldn't find the locked folio"); + locked_folio = NULL; goto out_bits; } + ASSERT(folio_order(locked_folio) == 0); start = test_start; end = start + PAGE_SIZE - 1; - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (found) { test_err("found range when we shouldn't have"); goto out_bits; @@ -265,8 +268,7 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) btrfs_set_extent_bit(tmp, max_bytes, total_dirty - 1, EXTENT_DELALLOC, NULL); start = test_start; end = start + PAGE_SIZE - 1; - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (!found) { test_err("didn't find our range"); goto out_bits; @@ -278,36 +280,35 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) } if (process_folio_range(inode, start, end, PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { - test_err("pages in range were not all locked"); + test_err("folios in range were not all locked"); goto out_bits; } btrfs_unlock_extent(tmp, start, end, NULL); /* - * Now to test where we run into a page that is no longer dirty in the + * Now to test where we run into a folio that is no longer dirty in the * range we want to find. */ - page = find_get_page(inode->i_mapping, - (max_bytes + SZ_1M) >> PAGE_SHIFT); - if (!page) { - test_err("couldn't find our page"); + folio = filemap_get_folio(inode->i_mapping, (max_bytes + SZ_1M) >> PAGE_SHIFT); + if (IS_ERR(folio)) { + test_err("couldn't find our folio"); goto out_bits; } - ClearPageDirty(page); - put_page(page); + ASSERT(folio_order(folio) == 0); + folio_clear_dirty(folio); + folio_put(folio); /* We unlocked it in the previous test */ - lock_page(locked_page); + folio_lock(locked_folio); start = test_start; end = start + PAGE_SIZE - 1; /* - * Currently if we fail to find dirty pages in the delalloc range we + * Currently if we fail to find dirty folios in the delalloc range we * will adjust max_bytes down to PAGE_SIZE and then re-search. If * this changes at any point in the future we will need to fix this * tests expected behavior. */ - found = find_lock_delalloc_range(inode, page_folio(locked_page), &start, - &end); + found = find_lock_delalloc_range(inode, locked_folio, &start, &end); if (!found) { test_err("didn't find our range"); goto out_bits; @@ -319,7 +320,7 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) } if (process_folio_range(inode, start, end, PROCESS_TEST_LOCKED | PROCESS_UNLOCK)) { - test_err("pages in range were not all locked"); + test_err("folios in range were not all locked"); goto out_bits; } ret = 0; @@ -328,8 +329,8 @@ static int test_find_delalloc(u32 sectorsize, u32 nodesize) dump_extent_io_tree(tmp); btrfs_clear_extent_bit(tmp, 0, total_dirty - 1, (unsigned)-1, NULL); out: - if (locked_page) - put_page(locked_page); + if (locked_folio) + folio_put(locked_folio); process_folio_range(inode, 0, total_dirty - 1, PROCESS_UNLOCK | PROCESS_RELEASE); iput(inode); From 1975b173fd1c4e94589dd6cb142b45dfe25bd289 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:19:57 -0400 Subject: [PATCH 0033/1012] btrfs: tests: use eb folio helpers in extent buffer memory checks dump_eb_and_memory_contents() and verify_eb_and_memory() hardcode one page per folio instead of using get_eb_folio_index() and get_eb_offset_in_folio() like the rest of the extent buffer code. Use the helpers and folio_address(). This removes the last struct page usage in the file. No functional change. The tests only run with sectorsize == PAGE_SIZE, and the test extent buffers are backed by order-0 folios. Assisted-by: Claude:claude-fable-5-1 Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tests/extent-io-tests.c | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/tests/extent-io-tests.c b/fs/btrfs/tests/extent-io-tests.c index ee8eabac47f110..cd045778400dc0 100644 --- a/fs/btrfs/tests/extent-io-tests.c +++ b/fs/btrfs/tests/extent-io-tests.c @@ -672,8 +672,9 @@ static void dump_eb_and_memory_contents(struct extent_buffer *eb, void *memory, const char *test_name) { for (int i = 0; i < eb->len; i++) { - struct page *page = folio_page(eb->folios[i >> PAGE_SHIFT], 0); - void *addr = page_address(page) + offset_in_page(i); + const unsigned long idx = get_eb_folio_index(eb, i); + void *addr = folio_address(eb->folios[idx]) + + get_eb_offset_in_folio(eb, i); if (memcmp(addr, memory + i, 1) != 0) { test_err("%s failed", test_name); @@ -688,9 +689,12 @@ static int verify_eb_and_memory(struct extent_buffer *eb, void *memory, const char *test_name) { for (int i = 0; i < (eb->len >> PAGE_SHIFT); i++) { - void *eb_addr = folio_address(eb->folios[i]); + const unsigned long offset = i << PAGE_SHIFT; + const unsigned long idx = get_eb_folio_index(eb, offset); + void *eb_addr = folio_address(eb->folios[idx]) + + get_eb_offset_in_folio(eb, offset); - if (memcmp(memory + (i << PAGE_SHIFT), eb_addr, PAGE_SIZE) != 0) { + if (memcmp(memory + offset, eb_addr, PAGE_SIZE) != 0) { dump_eb_and_memory_contents(eb, memory, test_name); return -EUCLEAN; } From 027ffac059fd06eb20900c5a296e96485aa90e2f Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:19:58 -0400 Subject: [PATCH 0034/1012] btrfs: convert btrfs_compr_pool_scan() to use folios The compression pool holds order-0 folios, but btrfs_compr_pool_scan() walks it as struct page through page->lru. Walk it as folios, matching the other compression pool functions. This removes the last use of page->lru in btrfs and saves a call to compound_head() per freed folio. Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/compression.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index c62b5148d5ac19..979b2ffbd8fc7e 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -168,10 +168,10 @@ static unsigned long btrfs_compr_pool_scan(struct shrinker *sh, struct shrink_co spin_unlock(&compr_pool.lock); list_for_each_safe(tmp, next, &remove) { - struct page *page = list_entry(tmp, struct page, lru); + struct folio *folio = list_entry(tmp, struct folio, lru); - ASSERT(page_ref_count(page) == 1); - put_page(page); + ASSERT(folio_ref_count(folio) == 1); + folio_put(folio); } return freed; From 05d82382315192a4bb840166d6d2e6f4198cbcfa Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:19:59 -0400 Subject: [PATCH 0035/1012] btrfs: convert heuristic_collect_sample() to use folios Convert the sampling loop to folios. This removes the last caller of find_get_page() in btrfs and saves a call to compound_head() per sampled page. Document that the lookup is not supposed to fail with an ASSERT(). Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/compression.c | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index 979b2ffbd8fc7e..fa8b92592321ee 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -1488,7 +1488,7 @@ static bool sample_repeated_patterns(struct heuristic_ws *ws) static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, struct heuristic_ws *ws) { - struct page *page; + struct folio *folio; pgoff_t index, index_end; u32 i, curr_sample_pos; u8 *in_data; @@ -1514,8 +1514,10 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, curr_sample_pos = 0; while (index < index_end) { - page = find_get_page(inode->i_mapping, index); - in_data = kmap_local_page(page); + folio = filemap_get_folio(inode->i_mapping, index); + ASSERT(!IS_ERR(folio)); + in_data = kmap_local_folio(folio, + offset_in_folio(folio, (u64)index << PAGE_SHIFT)); /* Handle case where the start is not aligned to PAGE_SIZE */ i = start % PAGE_SIZE; while (i < PAGE_SIZE - SAMPLING_READ_SIZE) { @@ -1529,7 +1531,7 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, curr_sample_pos += SAMPLING_READ_SIZE; } kunmap_local(in_data); - put_page(page); + folio_put(folio); index++; } From f05334e583a040540f0564eab5d36f96385da046 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:20:00 -0400 Subject: [PATCH 0036/1012] btrfs: fix stale function references in compression comments add_ra_bio_pages() was renamed to add_ra_bio_folios(), and btrfs_compress_filemap_get_folio() wraps filemap_get_folio(), not find_get_page(). Update the comments accordingly. Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/compression.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index fa8b92592321ee..20169d02896191 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -431,7 +431,7 @@ static noinline int add_ra_bio_folios(struct inode *inode, u64 compressed_end, } /* - * Since add_ra_bio_pages() is always speculative, suppress + * Since add_ra_bio_folios() is always speculative, suppress * allocation warnings. */ masked_constraint_gfp = mapping_gfp_constraint(mapping, constraint_gfp); @@ -960,7 +960,7 @@ bool btrfs_compress_level_valid(unsigned int type, int level) return levels->min_level <= level && level <= levels->max_level; } -/* Wrapper around find_get_page(), with extra error message. */ +/* Wrapper around filemap_get_folio(), with extra error message. */ int btrfs_compress_filemap_get_folio(struct address_space *mapping, u64 start, struct folio **in_folio_ret) { From 6995a1a22ec71b75b7955777cc47c665702e5ebe Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Mon, 7 Sep 2026 16:20:01 -0400 Subject: [PATCH 0037/1012] btrfs: use folios for reading super blocks from the block device btrfs_read_disk_super() and the zoned super block log comparison go through read_cache_page_gfp() and page_address(), and btrfs_release_disk_super() recovers the page with virt_to_page(). Use mapping_read_folio_gfp(), folio_address(), and virt_to_folio() instead. This removes the last callers of read_cache_page_gfp() and put_page() in btrfs. Compute the super block address with offset_in_folio() as write_dev_supers() does, rather than assuming it is at the start of the page. Assisted-by: Claude:claude-fable-5-1 Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/volumes.c | 16 +++++++--------- fs/btrfs/zoned.c | 12 ++++++------ 2 files changed, 13 insertions(+), 15 deletions(-) diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 949e40baff3343..4ddabadc918852 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -1327,16 +1327,14 @@ int btrfs_open_devices(struct btrfs_fs_devices *fs_devices, void btrfs_release_disk_super(struct btrfs_super_block *super) { - struct page *page = virt_to_page(super); - - put_page(page); + folio_put(virt_to_folio(super)); } struct btrfs_super_block *btrfs_read_disk_super(struct block_device *bdev, int copy_num, bool drop_cache) { struct btrfs_super_block *super; - struct page *page; + struct folio *folio; u64 bytenr, bytenr_orig; struct address_space *mapping = bdev->bd_mapping; int ret; @@ -1357,7 +1355,7 @@ struct btrfs_super_block *btrfs_read_disk_super(struct block_device *bdev, ASSERT(copy_num == 0); /* - * Drop the page of the primary superblock, so later read will + * Drop the folio of the primary superblock, so later read will * always read from the device. */ invalidate_inode_pages2_range(mapping, bytenr >> PAGE_SHIFT, @@ -1365,12 +1363,12 @@ struct btrfs_super_block *btrfs_read_disk_super(struct block_device *bdev, } filemap_invalidate_lock_shared(mapping); - page = read_cache_page_gfp(mapping, bytenr >> PAGE_SHIFT, GFP_NOFS); + folio = mapping_read_folio_gfp(mapping, bytenr >> PAGE_SHIFT, GFP_NOFS); filemap_invalidate_unlock_shared(mapping); - if (IS_ERR(page)) - return ERR_CAST(page); + if (IS_ERR(folio)) + return ERR_CAST(folio); - super = page_address(page); + super = folio_address(folio) + offset_in_folio(folio, bytenr); if (btrfs_super_magic(super) != BTRFS_MAGIC || btrfs_super_bytenr(super) != bytenr_orig) { btrfs_release_disk_super(super); diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c index 08a15465a0877d..a1ef8caaacdab4 100644 --- a/fs/btrfs/zoned.c +++ b/fs/btrfs/zoned.c @@ -123,24 +123,24 @@ static int sb_write_pointer(struct block_device *bdev, struct blk_zone *zones, } else if (full[0] && full[1]) { /* Compare two super blocks */ struct address_space *mapping = bdev->bd_mapping; - struct page *page[BTRFS_NR_SB_LOG_ZONES]; struct btrfs_super_block *super[BTRFS_NR_SB_LOG_ZONES]; for (int i = 0; i < BTRFS_NR_SB_LOG_ZONES; i++) { u64 zone_end = (zones[i].start + zones[i].capacity) << SECTOR_SHIFT; u64 bytenr = ALIGN_DOWN(zone_end, BTRFS_SUPER_INFO_SIZE) - BTRFS_SUPER_INFO_SIZE; + struct folio *folio; filemap_invalidate_lock_shared(mapping); - page[i] = read_cache_page_gfp(mapping, - bytenr >> PAGE_SHIFT, GFP_NOFS); + folio = mapping_read_folio_gfp(mapping, bytenr >> PAGE_SHIFT, + GFP_NOFS); filemap_invalidate_unlock_shared(mapping); - if (IS_ERR(page[i])) { + if (IS_ERR(folio)) { if (i == 1) btrfs_release_disk_super(super[0]); - return PTR_ERR(page[i]); + return PTR_ERR(folio); } - super[i] = page_address(page[i]); + super[i] = folio_address(folio) + offset_in_folio(folio, bytenr); } if (btrfs_super_generation(super[0]) > From 3f7a2f42d6d2f1e4226b394b0c772eff01b90b50 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Wed, 9 Sep 2026 13:10:48 -0400 Subject: [PATCH 0038/1012] btrfs: use u64 for the page indices in heuristic_collect_sample() index and index_end are derived from the u64 start and end offsets, and index is shifted back into a byte offset for offset_in_folio(), which needs a cast to u64 to be safe on 32-bit. Make them u64 instead so the cast goes away. They still fit pgoff_t where they are passed to filemap_get_folio(), as they came from a valid file offset. Signed-off-by: Tal Zussman Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/compression.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index 20169d02896191..228d1cdd7c2950 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -1489,7 +1489,7 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, struct heuristic_ws *ws) { struct folio *folio; - pgoff_t index, index_end; + u64 index, index_end; u32 i, curr_sample_pos; u8 *in_data; @@ -1517,7 +1517,7 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, folio = filemap_get_folio(inode->i_mapping, index); ASSERT(!IS_ERR(folio)); in_data = kmap_local_folio(folio, - offset_in_folio(folio, (u64)index << PAGE_SHIFT)); + offset_in_folio(folio, index << PAGE_SHIFT)); /* Handle case where the start is not aligned to PAGE_SIZE */ i = start % PAGE_SIZE; while (i < PAGE_SIZE - SAMPLING_READ_SIZE) { From 8388590012a11564ddb27286d428ff7bcf55c061 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Wed, 9 Sep 2026 12:26:30 +0100 Subject: [PATCH 0039/1012] btrfs: increment extent count once when logging extents during fast fsync In btrfs_log_changed_extents() we increment the extent count twice, and we then fallback to a transaction commit if the count reaches a threshold of 32K. However we increment the count twice, which is confusing and pointless. So increment the count only once and reduce the threshold to half (16K). Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/tree-log.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/fs/btrfs/tree-log.c b/fs/btrfs/tree-log.c index 7fb476b19eb457..1d9dfb63fc5a4f 100644 --- a/fs/btrfs/tree-log.c +++ b/fs/btrfs/tree-log.c @@ -5350,7 +5350,7 @@ static int btrfs_log_changed_extents(struct btrfs_trans_handle *trans, * have a bunch of extents we just want to commit since it will * be faster. */ - if (++num > 32768) { + if (++num > SZ_16K) { list_del_init(&tree->modified_extents); ret = -EFBIG; goto process; @@ -5368,7 +5368,6 @@ static int btrfs_log_changed_extents(struct btrfs_trans_handle *trans, refcount_inc(&em->refs); em->flags |= EXTENT_FLAG_LOGGING; list_add_tail(&em->list, &extents); - num++; } list_sort(NULL, &extents, extent_cmp); From 2e014081985efa5e317cf38783d54be16b7641fd Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Thu, 10 Sep 2026 17:37:48 +0100 Subject: [PATCH 0040/1012] btrfs: tree-checker: print dev extent offset in error message If a dev extent's offset is not sector size aligned, the error message is printing the dev extent's objectid instead of the offset. This is a copy paste error, as before this check we check the objectid field. Fixes: 008e2512dc56 ("btrfs: tree-checker: add dev extent item checks") Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tree-checker.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index b8ca9980e1ddd2..ea12d097320654 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -2171,7 +2171,7 @@ static int check_dev_extent_item(const struct extent_buffer *leaf, sectorsize))) { generic_err(leaf, slot, "invalid dev extent chunk offset, has %llu not aligned to %u", - btrfs_dev_extent_chunk_objectid(leaf, de), + btrfs_dev_extent_chunk_offset(leaf, de), sectorsize); return -EUCLEAN; } From df90bc736a1511db861de33ef5ef357303bee517 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Thu, 10 Sep 2026 17:48:06 +0100 Subject: [PATCH 0041/1012] btrfs: tree-checker: fix error message regarding free space extent items The error message mentions a free space info item, but we are processing a free space extent item, so fix the message. Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Reviewed-by: David Sterba Signed-off-by: David Sterba --- fs/btrfs/tree-checker.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index ea12d097320654..594266c9c116ff 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -2348,7 +2348,7 @@ static int check_free_space_extent(struct extent_buffer *leaf, struct btrfs_key if (unlikely(btrfs_item_size(leaf, slot) != 0)) { generic_err(leaf, slot, - "invalid item size for free space info, has %u expect 0", + "invalid item size for free space extent, has %u expect 0", btrfs_item_size(leaf, slot)); return -EUCLEAN; } From 6d55b4bf99d73c0bc7180c1b88f3930410ab1591 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Fri, 11 Sep 2026 11:56:29 +0100 Subject: [PATCH 0042/1012] btrfs: tree-checker: cache accessor return value in CHECK_FE_ALIGNED() We are calling an accessor for a file extent item multiple times in the CHECK_FE_ALIGNED() macro, even when we don't find a corruption (once in the if statement's expression and then once again in the expression for the return value). This adds extra runtime overhead (which is critical since the tree checker runs against every extent buffer when it's read or before persisting it) and increases the module's size. So cache the accessor's return value in a variable and use it, reducing runtime, object size and making the source code shorter too. Also avoid repeating twice the IS_ALIGNED() computation. Before: $ size fs/btrfs/btrfs.ko text data bss dec hex filename 2073340 217928 15624 2306892 23334c fs/btrfs/btrfs.ko After: $ size fs/btrfs/btrfs.ko text data bss dec hex filename 2073076 217928 15624 2306628 233244 fs/btrfs/btrfs.ko Also running the following fsstress test and capturing the runtime of check_extent_data_item() (the only caller of CHECK_FE_ALIGNED()) in nanoseconds (using bpftrace), showed the following runtime improvements: Test: mkfs.btrfs -f /dev/nullb0 mount /dev/nullb0 /mnt fsstress -w -p 8 -n 5000 -s 12345 -d /mnt umount /mnt Before: Count: 2033671 Range: 0.000 - 1336060.000; Mean: 679.642; Median: 656.000; Stddev: 2031.323 Percentiles: 90th: 850.000; 95th: 909.000; 99th: 1272.000 0.000 - 6.647: 11 | 6.647 - 28.241: 36 | 28.241 - 110.807: 176 | 110.807 - 426.512: 38012 # 426.512 - 1633.663: 1983372 ##################################################### 1633.663 - 6249.399: 8315 | 6249.399 - 23898.422: 3243 | 23898.422 - 91382.340: 153 | 91382.340 - 349418.114: 36 | 349418.114 - 1336060.000: 11 | After: Count: 2092797 Range: 0.000 - 1677284.000; Mean: 627.650; Median: 617.000; Stddev: 1848.010 Percentiles: 90th: 809.000; 95th: 859.000; 99th: 1209.000 0.000 - 6.823: 18 | 6.823 - 29.602: 82 | 29.602 - 118.702: 416 | 118.702 - 467.232: 515774 ################# 467.232 - 1830.549: 1565599 ##################################################### 1830.549 - 7163.339: 5297 | 7163.339 - 28023.237: 4186 | 28023.237 - 109619.427: 166 | 109619.427 - 428793.471: 25 | 428793.471 - 1677284.000: 4 | Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/tree-checker.c | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index 594266c9c116ff..0f4f1b347c9e8e 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -108,13 +108,14 @@ static void file_extent_err(const struct extent_buffer *eb, int slot, */ #define CHECK_FE_ALIGNED(leaf, slot, fi, name, alignment) \ ({ \ - if (unlikely(!IS_ALIGNED(btrfs_file_extent_##name((leaf), (fi)), \ - (alignment)))) \ + const u64 val = btrfs_file_extent_##name((leaf), (fi)); \ + const bool not_aligned = !IS_ALIGNED(val, (alignment)); \ + \ + if (unlikely(not_aligned)) \ file_extent_err((leaf), (slot), \ "invalid %s for file extent, have %llu, should be aligned to %u", \ - (#name), btrfs_file_extent_##name((leaf), (fi)), \ - (alignment)); \ - (!IS_ALIGNED(btrfs_file_extent_##name((leaf), (fi)), (alignment))); \ + (#name), val, (alignment)); \ + not_aligned; \ }) static u64 file_extent_end(struct extent_buffer *leaf, From 448c6f99ce61d5dc258823e9b834dbfeafddd7c5 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Sun, 13 Sep 2026 20:59:15 +0930 Subject: [PATCH 0043/1012] btrfs: cleanup and rename submit_extent_folio() Cleanup submit_extent_folio() by: - Remove @size parameter Since commit b2e743927fdd ("btrfs: make btrfs_do_readpage() to do block-by-block read"), all callers are passing sectorsize as @size, so there is no need for such parameter. Furthermore since we only write one block at a time, there is no need for a while() loop, nor the advance of various local variables. - Update the comments on the parameter list * @disk_bytenr is shared for both read and write * rename @page to @folio - Update the return value to return 0 or error Since we won't queue multiple blocks anyway, there is no point in returning the queued bytes. It makes more sense to return an error code, although the only error code will be -EUCLEAN for writes. - Rename submit_extent_folio() to submit_one_block() - Rename submit_one_sector() to submit_write_sector() - Update the error message to utilize the new returned error code inside submit_write_sector() Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/extent_io.c | 187 +++++++++++++++++++------------------------ 1 file changed, 81 insertions(+), 106 deletions(-) diff --git a/fs/btrfs/extent_io.c b/fs/btrfs/extent_io.c index a221b63bdb205d..55e9144d47595f 100644 --- a/fs/btrfs/extent_io.c +++ b/fs/btrfs/extent_io.c @@ -110,14 +110,14 @@ struct btrfs_bio_ctrl { * make the decision when submitting the bio. * * The pattern between do_readpage(), submit_one_bio() and - * submit_extent_folio() is quite subtle, so tracking this is tricky. + * submit_one_block() is quite subtle, so tracking this is tricky. * * As we process extent E, we might submit a bio with existing built up * extents before adding E to a new bio, or we might just add E to the * bio. As a result, E's generation could apply to the current bio or * to the next one, so we need to be careful to update the bio_ctrl's * generation with E's only when we are sure E is added to bio_ctrl->bbio - * in submit_extent_folio(). + * in submit_one_block(). * * See the comment in btrfs_lookup_bio_sums() for more detail on the * need for this optimization. @@ -797,118 +797,95 @@ static int alloc_new_bio(struct btrfs_inode *inode, } /* - * @disk_bytenr: logical bytenr where the write will be - * @page: page to add to the bio - * @size: portion of page that we want to write to - * @pg_offset: offset of the new bio or to check whether we are adding - * a contiguous page to the previous one + * @disk_bytenr: logical bytenr where the read/write will be + * @folio: the folio the block belongs to + * @pg_offset: the offset inside the folio * @read_em_generation: generation of the extent_map we are submitting * (only used for read) * - * The will either add the page into the existing @bio_ctrl->bbio, or allocate a + * This will either add the block into the existing @bio_ctrl->bbio, or allocate a * new one in @bio_ctrl->bbio. * The mirror number for this IO should already be initialized in * @bio_ctrl->mirror_num. * - * Return the number of bytes that are queued into a bio. - * If the returned bytes is smaller than @size, it means we hit a critical error - * for data write, where there is no ordered extent for the range. + * Return 0 if the block is queued or submitted. + * Return <0 for error. */ -static unsigned int submit_extent_folio(struct btrfs_bio_ctrl *bio_ctrl, - u64 disk_bytenr, struct folio *folio, - size_t size, unsigned long pg_offset, - u64 read_em_generation) +static int submit_one_block(struct btrfs_bio_ctrl *bio_ctrl, + u64 disk_bytenr, struct folio *folio, + unsigned long pg_offset, u64 read_em_generation) { struct btrfs_inode *inode = folio_to_inode(folio); + const struct btrfs_fs_info *fs_info = inode->root->fs_info; + const u32 blocksize = fs_info->sectorsize; loff_t file_offset = folio_pos(folio) + pg_offset; - unsigned int queued = 0; - ASSERT(pg_offset + size <= folio_size(folio)); + ASSERT(pg_offset + blocksize <= folio_size(folio)); ASSERT(bio_ctrl->end_io_func); if (bio_ctrl->bbio && !btrfs_bio_is_contig(bio_ctrl, disk_bytenr, file_offset)) submit_one_bio(bio_ctrl); - do { - u32 len = size; - - /* Allocate new bio if needed */ - if (!bio_ctrl->bbio) { - int ret; - - ret = alloc_new_bio(inode, bio_ctrl, disk_bytenr, file_offset); - if (ret < 0) - break; - } +again: + /* Allocate new bio if needed */ + if (!bio_ctrl->bbio) { + int ret; - /* Cap to the current ordered extent boundary if there is one. */ - if (len > bio_ctrl->len_to_oe_boundary) { - ASSERT(bio_ctrl->compress_type == BTRFS_COMPRESS_NONE); - ASSERT(is_data_inode(inode)); - len = bio_ctrl->len_to_oe_boundary; - } + ret = alloc_new_bio(inode, bio_ctrl, disk_bytenr, file_offset); + if (ret < 0) + return ret; + } - if (!bio_add_folio(&bio_ctrl->bbio->bio, folio, len, pg_offset)) { - /* bio full: move on to a new one */ - submit_one_bio(bio_ctrl); - continue; - } - /* - * Now that the folio is definitely added to the bio, include its - * generation in the max generation calculation. - */ - bio_ctrl->generation = max(bio_ctrl->generation, read_em_generation); - bio_ctrl->next_file_offset += len; + if (!bio_add_folio(&bio_ctrl->bbio->bio, folio, blocksize, pg_offset)) { + /* bio full: move on to a new one */ + submit_one_bio(bio_ctrl); + goto again; + } - if (bio_ctrl->wbc) - wbc_account_cgroup_owner(bio_ctrl->wbc, folio, len); + /* + * Now that the folio is definitely added to the bio, include its + * generation in the max generation calculation. + */ + bio_ctrl->generation = max(bio_ctrl->generation, read_em_generation); + bio_ctrl->next_file_offset += blocksize; - size -= len; - pg_offset += len; - disk_bytenr += len; - file_offset += len; - queued += len; + if (bio_ctrl->wbc) + wbc_account_cgroup_owner(bio_ctrl->wbc, folio, blocksize); - /* - * len_to_oe_boundary defaults to U32_MAX, which isn't folio or - * sector aligned. alloc_new_bio() then sets it to the end of - * our ordered extent for writes into zoned devices. - * - * When len_to_oe_boundary is tracking an ordered extent, we - * trust the ordered extent code to align things properly, and - * the check above to cap our write to the ordered extent - * boundary is correct. - * - * When len_to_oe_boundary is U32_MAX, the cap above would - * result in a 4095 byte IO for the last folio right before - * we hit the bio limit of UINT_MAX. bio_add_folio() has all - * the checks required to make sure we don't overflow the bio, - * and we should just ignore len_to_oe_boundary completely - * unless we're using it to track an ordered extent. - * - * It's pretty hard to make a bio sized U32_MAX, but it can - * happen when the page cache is able to feed us contiguous - * folios for large extents. - */ - if (bio_ctrl->len_to_oe_boundary != U32_MAX) - bio_ctrl->len_to_oe_boundary -= len; - - /* Ordered extent boundary: move on to a new bio. */ - if (bio_ctrl->len_to_oe_boundary == 0) - submit_one_bio(bio_ctrl); - /* - * If we have accumulated decent amount of IO, send it to the - * block layer so that IO can run while we are accumulating - * more folios to write. - */ - else if (bio_ctrl->wbc && - bio_ctrl->bbio->bio.bi_iter.bi_size >= - inode->root->fs_info->writeback_bio_size) - submit_one_bio(bio_ctrl); + /* + * len_to_oe_boundary defaults to U32_MAX, which isn't folio or sector + * aligned. alloc_new_bio() then sets it to the end of our ordered + * extent for writes into zoned devices. + * + * When len_to_oe_boundary is tracking an ordered extent, the + * len_to_oe_boundary should follow that OE and never go beyond the max + * extent size (128MiB). + * + * When len_to_oe_boundary is U32_MAX, decreasing the length by + * blocksize will never make it reach 0, thus skipping the later + * submit_one_bio() call. So if len_to_oe_boundary() is not tracking + * an OE, do not decrease it. + * + * It's pretty hard to make a bio sized U32_MAX, but it can happen when + * the page cache is able to feed us contiguous folios for large + * extents. + */ + if (bio_ctrl->len_to_oe_boundary != U32_MAX) + bio_ctrl->len_to_oe_boundary -= blocksize; - } while (size); - return queued; + /* Ordered extent boundary: move on to a new bio. */ + if (bio_ctrl->len_to_oe_boundary == 0) + submit_one_bio(bio_ctrl); + /* + * If we have accumulated decent amount of IO, send it to the block + * layer so that IO can run while we are accumulating more folios to + * write. + */ + else if (bio_ctrl->wbc && + bio_ctrl->bbio->bio.bi_iter.bi_size >= fs_info->writeback_bio_size) + submit_one_bio(bio_ctrl); + return 0; } static int attach_extent_buffer_folio(struct extent_buffer *eb, @@ -1092,7 +1069,6 @@ static int btrfs_do_readpage(struct folio *folio, struct extent_map **em_cached, u64 disk_bytenr; u64 block_start; u64 em_gen; - unsigned int queued; ASSERT(IS_ALIGNED(cur, fs_info->sectorsize)); if (cur >= last_byte) { @@ -1206,10 +1182,9 @@ static int btrfs_do_readpage(struct folio *folio, struct extent_map **em_cached, if (force_bio_submit) submit_one_bio(bio_ctrl); - queued = submit_extent_folio(bio_ctrl, disk_bytenr, folio, blocksize, - pg_offset, em_gen); + ret = submit_one_block(bio_ctrl, disk_bytenr, folio, pg_offset, em_gen); /* Read submission should not fail. */ - ASSERT(queued == blocksize); + ASSERT(ret == 0); } return 0; } @@ -1830,10 +1805,10 @@ static struct btrfs_ordered_extent *get_oe_from_bbio(const struct btrfs_bio *bbi * * Caller should make sure filepos < i_size and handle filepos >= i_size case. */ -static int submit_one_sector(struct btrfs_inode *inode, - struct folio *folio, - u64 filepos, struct btrfs_bio_ctrl *bio_ctrl, - loff_t i_size) +static int submit_write_sector(struct btrfs_inode *inode, + struct folio *folio, + u64 filepos, struct btrfs_bio_ctrl *bio_ctrl, + loff_t i_size) { struct btrfs_fs_info *fs_info = inode->root->fs_info; struct btrfs_ordered_extent *oe; @@ -1841,7 +1816,7 @@ static int submit_one_sector(struct btrfs_inode *inode, u64 disk_bytenr; u64 extent_offset; const u32 sectorsize = fs_info->sectorsize; - unsigned int queued; + int ret; ASSERT(IS_ALIGNED(filepos, sectorsize)); @@ -1904,17 +1879,17 @@ static int submit_one_sector(struct btrfs_inode *inode, */ ASSERT(folio_test_writeback(folio)); - queued = submit_extent_folio(bio_ctrl, disk_bytenr, folio, - sectorsize, filepos - folio_pos(folio), 0); - if (unlikely(queued < sectorsize)) { + ret = submit_one_block(bio_ctrl, disk_bytenr, folio, + offset_in_folio(folio, filepos), 0); + if (unlikely(ret < 0)) { btrfs_folio_clear_writeback(fs_info, folio, filepos, sectorsize); btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize, false); btrfs_err_rl(fs_info, - "failed to queue sector for root %lld ino %llu filepos %llu", + "failed to queue sector for root %lld ino %llu filepos %llu: %pe", btrfs_root_id(inode->root), - btrfs_ino(inode), filepos); - return -EUCLEAN; + btrfs_ino(inode), filepos, ERR_PTR(ret)); + return ret; } return 0; } @@ -1999,7 +1974,7 @@ static noinline_for_stack int extent_writepage_io(struct btrfs_inode *inode, btrfs_folio_clear_dirty(fs_info, folio, cur, fs_info->sectorsize); continue; } - ret = submit_one_sector(inode, folio, cur, bio_ctrl, i_size); + ret = submit_write_sector(inode, folio, cur, bio_ctrl, i_size); if (unlikely(ret < 0)) { if (!found_error) found_error = ret; From 9dc38f249a02e99124058caf6d4926fa0e0032fe Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 10 Sep 2026 11:35:52 +0930 Subject: [PATCH 0044/1012] btrfs: fix off-by-one end related to inode_need_compress() In most cases btrfs uses @end as the inclusive end bytenr for a range, and this applies to inode_need_compress(). However we have several sites not following the inclusive bytenr: - run_delalloc_inline() Which assigned @blocksize as @end for inode_need_compress() This makes inode_need_compress() always skip the disk_i_size check. - heuristic_collect_sample() Which assigned "start + BTRFS_MAX_UNCOMPRESSED" to @end, which is the exclusive bytenr. Neither is really causing any real problem, as heuristic_collect_sample() has proper checks to avoid reading anything beyond @end, and the sampling read size is 16 bytes, so it has enough headroom to handle that off-by-one problem. But still I do not like anything out of the common scheme, so fix the off-by-one @end for both call sites, and add extra ASSERT()s to catch such unaligned parameters. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/compression.c | 15 +++++++-------- fs/btrfs/inode.c | 5 ++++- 2 files changed, 11 insertions(+), 9 deletions(-) diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index 228d1cdd7c2950..7c1c018dbd6fd7 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -1488,11 +1488,14 @@ static bool sample_repeated_patterns(struct heuristic_ws *ws) static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, struct heuristic_ws *ws) { + const u32 blocksize = BTRFS_I(inode)->root->fs_info->sectorsize; struct folio *folio; u64 index, index_end; u32 i, curr_sample_pos; u8 *in_data; + ASSERT(IS_ALIGNED(start, blocksize) && IS_ALIGNED(end + 1, blocksize)); + /* * Compression handles the input data by chunks of 128KiB * (defined by BTRFS_MAX_UNCOMPRESSED) @@ -1502,18 +1505,14 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, * MAX_SAMPLE_SIZE - calculated under assumption that heuristic will * process no more than BTRFS_MAX_UNCOMPRESSED at a time. */ - if (end - start > BTRFS_MAX_UNCOMPRESSED) - end = start + BTRFS_MAX_UNCOMPRESSED; + if (end + 1 - start > BTRFS_MAX_UNCOMPRESSED) + end = start + BTRFS_MAX_UNCOMPRESSED - 1; index = start >> PAGE_SHIFT; index_end = end >> PAGE_SHIFT; - /* Don't miss unaligned end */ - if (!PAGE_ALIGNED(end)) - index_end++; - curr_sample_pos = 0; - while (index < index_end) { + while (index <= index_end) { folio = filemap_get_folio(inode->i_mapping, index); ASSERT(!IS_ERR(folio)); in_data = kmap_local_folio(folio, @@ -1522,7 +1521,7 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, i = start % PAGE_SIZE; while (i < PAGE_SIZE - SAMPLING_READ_SIZE) { /* Don't sample any garbage from the last page */ - if (start > end - SAMPLING_READ_SIZE) + if (start > end + 1 - SAMPLING_READ_SIZE) break; memcpy(&ws->sample[curr_sample_pos], &in_data[i], SAMPLING_READ_SIZE); diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index a85a7c561cf8e9..79f2181dc6310a 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -730,6 +730,9 @@ static inline int inode_need_compress(struct btrfs_inode *inode, u64 start, u64 end, bool check_inline) { struct btrfs_fs_info *fs_info = inode->root->fs_info; + const u32 blocksize = fs_info->sectorsize; + + ASSERT(IS_ALIGNED(start, blocksize) && IS_ALIGNED(end + 1, blocksize)); if (unlikely(!btrfs_inode_can_compress(inode))) { DEBUG_WARN("BTRFS: unexpected compression for ino %llu", btrfs_ino(inode)); @@ -2331,7 +2334,7 @@ static int run_delalloc_inline(struct btrfs_inode *inode, struct folio *locked_f btrfs_check_folio_write_protected(locked_folio); if (btrfs_inode_can_compress(inode) && - inode_need_compress(inode, 0, blocksize, true)) { + inode_need_compress(inode, 0, blocksize - 1, true)) { if (inode->defrag_compress > 0 && inode->defrag_compress < BTRFS_NR_COMPRESS_TYPES) { compress_type = inode->defrag_compress; From e894e7cf00f7cf62252645bb907d5172cca8fede Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Thu, 10 Sep 2026 11:35:53 +0930 Subject: [PATCH 0045/1012] btrfs: simplify heuristic_collect_sample() to handle large folios better Currently heuristic_collect_sample() is purely page size based, and it has a lot of extra handling just inside the page. However we already have large folio support, there is no need to look up the same folio repeatedly. Simplify the handling by: - Use @cur as the iterator instead of page index - Handle the sample copying on a per-folio basis Although kmap_local_folio() requires an offset to handle HIGHMEM page mapping, we have rejected large folios for HIGHMEM systems completely. So we can safely handle all sample copying inside the folio in one go. - Remove unnecessary unaligned range handling All the range passed in should be block aligned, thus there is no need to handle cases where sample crosses the block boundary. Reviewed-by: Boris Burkov Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/compression.c | 40 ++++++++++++++++------------------------ 1 file changed, 16 insertions(+), 24 deletions(-) diff --git a/fs/btrfs/compression.c b/fs/btrfs/compression.c index 7c1c018dbd6fd7..ad90e14032dbb3 100644 --- a/fs/btrfs/compression.c +++ b/fs/btrfs/compression.c @@ -1489,10 +1489,8 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, struct heuristic_ws *ws) { const u32 blocksize = BTRFS_I(inode)->root->fs_info->sectorsize; - struct folio *folio; - u64 index, index_end; - u32 i, curr_sample_pos; - u8 *in_data; + u64 cur = start; + u32 curr_sample_pos = 0; ASSERT(IS_ALIGNED(start, blocksize) && IS_ALIGNED(end + 1, blocksize)); @@ -1508,33 +1506,27 @@ static void heuristic_collect_sample(struct inode *inode, u64 start, u64 end, if (end + 1 - start > BTRFS_MAX_UNCOMPRESSED) end = start + BTRFS_MAX_UNCOMPRESSED - 1; - index = start >> PAGE_SHIFT; - index_end = end >> PAGE_SHIFT; + while (cur < end) { + struct folio *folio; + void *in_data; + u64 next_pos; - curr_sample_pos = 0; - while (index <= index_end) { - folio = filemap_get_folio(inode->i_mapping, index); + folio = filemap_get_folio(inode->i_mapping, cur >> PAGE_SHIFT); + /* All folios inside the range should exist and be locked. */ ASSERT(!IS_ERR(folio)); - in_data = kmap_local_folio(folio, - offset_in_folio(folio, index << PAGE_SHIFT)); - /* Handle case where the start is not aligned to PAGE_SIZE */ - i = start % PAGE_SIZE; - while (i < PAGE_SIZE - SAMPLING_READ_SIZE) { - /* Don't sample any garbage from the last page */ - if (start > end + 1 - SAMPLING_READ_SIZE) - break; - memcpy(&ws->sample[curr_sample_pos], &in_data[i], - SAMPLING_READ_SIZE); - i += SAMPLING_INTERVAL; - start += SAMPLING_INTERVAL; + next_pos = min_t(u64, end + 1, folio_next_pos(folio)); + in_data = kmap_local_folio(folio, 0); + + for (; cur < next_pos; cur += SAMPLING_INTERVAL) { + memcpy(&ws->sample[curr_sample_pos], + in_data + offset_in_folio(folio, cur), + SAMPLING_READ_SIZE); curr_sample_pos += SAMPLING_READ_SIZE; } kunmap_local(in_data); folio_put(folio); - - index++; + cur = next_pos; } - ws->sample_size = curr_sample_pos; } From fa1951b6a67fe23865bcea47915a8da6c408ddb4 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Mon, 14 Sep 2026 18:11:30 +0100 Subject: [PATCH 0046/1012] btrfs: fix creation of compressed inline extents that don't save space If the compressed data of an inline extent is larger than or equals to the size of the uncompressed data, we are still allowing the creation of the compressed inline extent, which does not result in any benefits, quite the contrary as we waste metadata space and have to decompress when reading. This is a recent regression introduced in commit 3eaf5f082c4c ("btrfs: extract inlined creation into a dedicated delalloc helper"). It happens because we are passing the block size to btrfs_compress_bio(), so we don't get -E2BIG from the compression code anymore, but we can not pass i_size either, because if i_size is smaller than sector size, we end up never creating lzo compressed inline extent for such small i_size values. So refuse the compressed result at run_delalloc_inline() if its size is not smaller than the uncompressed size (i_size). Reported-by: Hanabishi Link: https://lore.kernel.org/linux-btrfs/c97652a5-ac6b-4de6-aa23-3cdebc01d00b@gmail.com/ Fixes: 3eaf5f082c4c ("btrfs: extract inlined creation into a dedicated delalloc helper") CC: stable@vger.kernel.org # 7.1+ Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/inode.c | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 79f2181dc6310a..766dbdbf6e7d74 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -2342,12 +2342,27 @@ static int run_delalloc_inline(struct btrfs_inode *inode, struct folio *locked_f } else if (inode->prop_compress) { compress_type = inode->prop_compress; } + /* + * We need to pass blocksize and not i_size, otherwise we can't + * create compressed inline extents for data smaller than sector + * size with lzo. + */ cb = btrfs_compress_bio(inode, 0, blocksize, compress_type, compress_level, 0); if (IS_ERR(cb)) { cb = NULL; /* Just fall back to non-compressed case. */ } else { compressed_size = cb->bbio.bio.bi_iter.bi_size; + /* + * If we did not save space, it's pointless and wasteful + * to have an inline compressed extent, so fallback to + * an uncompressed inline extent. + */ + if (compressed_size >= i_size) { + cleanup_compressed_bio(cb); + cb = NULL; + compressed_size = 0; + } } } if (!can_cow_file_range_inline(inode, 0, i_size, compressed_size)) { From 4804b60c97bb94a6b8f954f1656c9907097bbf32 Mon Sep 17 00:00:00 2001 From: Daniel Linjama Date: Wed, 16 Sep 2026 09:15:56 +0300 Subject: [PATCH 0047/1012] btrfs: handle lack of space when cleaning up verity items When enable_verity() hits the qgroup limit, rollback_verity() needs its own metadata reservation. When the qgroup limit or lack of space refuses the rollback, the whole filesystem is forced read-only even though the qgroup limit was for one subvolume only. Also orphan cleanup at the next mount fails the same way, so the leftover items are never removed: with -EDQUOT the subvolume stays unreachable, and with -ENOSPC on a full filesystem the next read-write mount fails. Start transactions with btrfs_start_transaction_fallback_global_rsv() in btrfs_orphan_cleanup(), drop_verity_items() and rollback_verity(). Those calls only delete items and free the space in the end, so they may use the global reserve and skip the qgroup limit, which avoids -ENOSPC and -EDQUOT. Fixes: 146054090b08 ("btrfs: initial fsverity support") Reviewed-by: Qu Wenruo Signed-off-by: Daniel Linjama Signed-off-by: David Sterba --- fs/btrfs/inode.c | 3 ++- fs/btrfs/verity.c | 18 ++++++++++++++++-- 2 files changed, 18 insertions(+), 3 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 766dbdbf6e7d74..53f4532593b36c 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3857,7 +3857,8 @@ int btrfs_orphan_cleanup(struct btrfs_root *root) if (ret) goto out; } - trans = btrfs_start_transaction(root, 1); + /* Only deletes the orphan. */ + trans = btrfs_start_transaction_fallback_global_rsv(root, 1); if (IS_ERR(trans)) { ret = PTR_ERR(trans); goto out; diff --git a/fs/btrfs/verity.c b/fs/btrfs/verity.c index d432fec21c15bb..83dc7dee14cfb3 100644 --- a/fs/btrfs/verity.c +++ b/fs/btrfs/verity.c @@ -93,6 +93,20 @@ static loff_t merkle_file_pos(const struct inode *inode) return rounded; } +/* + * Start a transaction for removing verity items or the verity orphan. + * + * Like unlink, this only deletes items and frees space in the end, so the + * reservation may come from the global reserve when the filesystem is full + * (-ENOSPC) and is not subject to the qgroup limit (-EDQUOT). Otherwise a + * failed enable could never be cleaned up in either situation. + */ +static struct btrfs_trans_handle *start_verity_cleanup_trans(struct btrfs_root *root, + unsigned int num_items) +{ + return btrfs_start_transaction_fallback_global_rsv(root, num_items); +} + /* * Drop all the items for this inode with this key_type. * @@ -120,7 +134,7 @@ static int drop_verity_items(struct btrfs_inode *inode, u8 key_type) while (1) { /* 1 for the item being dropped */ - trans = btrfs_start_transaction(root, 1); + trans = start_verity_cleanup_trans(root, 1); if (IS_ERR(trans)) return PTR_ERR(trans); @@ -460,7 +474,7 @@ static int rollback_verity(struct btrfs_inode *inode) * 1 for updating the inode flag * 1 for deleting the orphan */ - trans = btrfs_start_transaction(root, 2); + trans = start_verity_cleanup_trans(root, 2); if (IS_ERR(trans)) { ret = PTR_ERR(trans); trans = NULL; From b351dbcd5a1c7b53552c51c7175d9b1c112cd7db Mon Sep 17 00:00:00 2001 From: Guanghui Yang <3497809730@qq.com> Date: Wed, 16 Sep 2026 05:16:38 +0000 Subject: [PATCH 0048/1012] btrfs: clear free space tree creation state on rebuild failure btrfs_rebuild_free_space_tree() sets BTRFS_FS_CREATING_FREE_SPACE_TREE before rebuilding the free space tree. Several error paths return without clearing this flag. The transaction restart failure path can leave the flag set on a live filesystem, causing delayed reference processing to be skipped. Clear it on all free space tree rebuild failure paths. Keep BTRFS_FS_FREE_SPACE_TREE_UNTRUSTED set, since a failed rebuild leaves the free space tree untrusted. Callers must fall back to extent-tree caching. Fixes: 882af9f13e83 ("btrfs: handle free space tree rebuild in multiple transactions") CC: stable@vger.kernel.org # 6.14+ Assisted-by: LLM Reviewed-by: Boris Burkov Reviewed-by: Qu Wenruo Signed-off-by: Guanghui Yang <3497809730@qq.com> Signed-off-by: David Sterba --- fs/btrfs/free-space-tree.c | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/free-space-tree.c b/fs/btrfs/free-space-tree.c index 1b3d82ae3de80e..b7a4a6ade30f0e 100644 --- a/fs/btrfs/free-space-tree.c +++ b/fs/btrfs/free-space-tree.c @@ -1353,7 +1353,7 @@ int btrfs_rebuild_free_space_tree(struct btrfs_fs_info *fs_info) if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); btrfs_end_transaction(trans); - return ret; + goto out_clear; } node = rb_first_cached(&fs_info->block_group_cache_tree); @@ -1371,14 +1371,16 @@ int btrfs_rebuild_free_space_tree(struct btrfs_fs_info *fs_info) if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); btrfs_end_transaction(trans); - return ret; + goto out_clear; } next: if (btrfs_should_end_transaction(trans)) { btrfs_end_transaction(trans); trans = btrfs_start_transaction(free_space_root, 1); - if (IS_ERR(trans)) - return PTR_ERR(trans); + if (IS_ERR(trans)) { + ret = PTR_ERR(trans); + goto out_clear; + } } node = rb_next(node); } @@ -1390,6 +1392,10 @@ int btrfs_rebuild_free_space_tree(struct btrfs_fs_info *fs_info) ret = btrfs_commit_transaction(trans); clear_bit(BTRFS_FS_FREE_SPACE_TREE_UNTRUSTED, &fs_info->flags); return ret; + +out_clear: + clear_bit(BTRFS_FS_CREATING_FREE_SPACE_TREE, &fs_info->flags); + return ret; } static int __add_block_group_free_space(struct btrfs_trans_handle *trans, From 5321707948bfcf3a9d13073b310ae3095a04e523 Mon Sep 17 00:00:00 2001 From: Qu Wenruo Date: Sat, 12 Sep 2026 18:12:06 +0930 Subject: [PATCH 0049/1012] btrfs: add "/dev/root" exception for device path update [BEHAVIOR CHANGE] Since commit 108cc8733989 ("btrfs: fix a lockdep caused by path resolution during device scan"), users with btrfs rootfs but without an initramfs are complaining that grub2 can no longer detect the rootfs device: /usr/sbin/grub-probe: error: cannot find a device for / (is /dev mounted?). [CAUSE] Although using btrfs without an initramfs is not recommended (if a new device is added to the rootfs, the system can no longer boot, as there is no way to register all devices), there is still a minority of users doing this. If there is no initramfs but the rootfs is on a block-device-based filesystem, the kernel boot sequence initializes a minimal ramfs/tmpfs, creates "/dev/root" with the proper device number for the rootfs, and then invokes mount using "/dev/root". That's why the end user will get the mount output: /dev/root on / rw To be honest, this is a user space problem: no one should trust the device path shown in mount, only the device number. E.g. one can even use "/proc/self/fd/*" to mount an fs, and that proc path will be registered, and no one else can mount that fs using that path. Before commit 108cc8733989 ("btrfs: fix a lockdep caused by path resolution during device scan"), btrfs had an internal path lookup workaround to address such weird paths, it works by checking if the existing device path can still resolve to the device number. But that path resolution is deadlock prone, thus it's replaced by a simple devt check. This works fine in most cases, as a btrfs device is registered by udev at boot time, thus all paths are sane. However this will not work for systems without an initramfs, causing the unreachable "/dev/root" path to exist forever without a way to rename it. [WORKAROUND] Despite updating the docs to discourage root btrfs without an initramfs, add an exception to the device path rename requirement. If the device has the name "/dev/root", we know it's booted without an initramfs, and only for that case we allow device path update. And if someone intentionally created "/dev/root" after boot, the existing devt checks will reject that weird name as usual. This should satisfy the minority of users, and still keep most of the existing guards preventing unexpected/unnecessary device path updates. But still, I prefer grub2 to implement a more robust device number based probing, and no one should use btrfs as rootfs without an initramfs. Fixes: 108cc8733989 ("btrfs: fix a lockdep caused by path resolution during device scan") Link: https://lore.kernel.org/linux-btrfs/CAKLYgeL7nrA4nXcewdv9Fqg_s=3GS=vmoypnEiZBKQ7rySZFuQ@mail.gmail.com/ Link: https://lore.kernel.org/linux-btrfs/dfbe1e27-dab8-4d55-8cf3-0b28eeac5df4@gmail.com/ Signed-off-by: Qu Wenruo Signed-off-by: David Sterba --- fs/btrfs/volumes.c | 33 ++++++++++++++++++++++++++++++++- 1 file changed, 32 insertions(+), 1 deletion(-) diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c index 4ddabadc918852..7c040f22dbc3e5 100644 --- a/fs/btrfs/volumes.c +++ b/fs/btrfs/volumes.c @@ -749,6 +749,36 @@ const u8 *btrfs_sb_fsid_ptr(const struct btrfs_super_block *sb) return has_metadata_uuid ? sb->metadata_uuid : sb->fsid; } +static bool should_rename_device(const struct btrfs_device *dev) +{ + bool ret; + const char *old_name; + + rcu_read_lock(); + old_name = rcu_dereference(dev->name); + /* + * For systems booted without an initramfs, the rootfs has the device + * name "/dev/root". + * + * Although using btrfs without an initramfs is not recommended (if a + * new device is added to the rootfs, the system can no longer boot, as + * there is no way to register all devices), there is still a minority + * of users doing this. + * + * And after the system is up, a later device scan on the real block + * device file will never get this device's name updated, as the + * device->devt is still the same. + * + * Here we add one and only one exception for "/dev/root", to allow the + * device name to be updated even if the new path points to the same + * block device. + */ + ret = (strcmp(old_name, "/dev/root") == 0); + rcu_read_unlock(); + + return ret; +} + /* * Add new device to list of registered devices * @@ -869,7 +899,8 @@ static noinline struct btrfs_device *device_list_add(const char *path, MAJOR(path_devt), MINOR(path_devt), current->comm, task_pid_nr(current)); - } else if (!device->name || device->devt != path_devt) { + } else if (!device->name || device->devt != path_devt || + should_rename_device(device)) { const char *old_name; /* From d9abc55d70e71b5d2876948043ba70c0f6200c4c Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Wed, 16 Sep 2026 15:43:41 +0100 Subject: [PATCH 0050/1012] btrfs: abort transaction on failure to update inode for hole punching and reflinking If we fail to update the inode we error out without aborting the transaction, which can result in a persistent inconsistency if after the failure the transaction is committed, as we have dropped file extent items from a range and either punched a hole or insert a new file extent item for that range (for reflinks). So add the missing transaction abort. Fixes: 2aaa66558172 ("Btrfs: add hole punching") Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/file.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/fs/btrfs/file.c b/fs/btrfs/file.c index 20e15dc30bfb55..f978c6524aa03f 100644 --- a/fs/btrfs/file.c +++ b/fs/btrfs/file.c @@ -2509,8 +2509,10 @@ int btrfs_replace_file_extents(struct btrfs_inode *inode, inode_set_ctime_current(&inode->vfs_inode)); ret = btrfs_update_inode(trans, inode); - if (ret) + if (unlikely(ret)) { + btrfs_abort_transaction(trans, ret); break; + } btrfs_end_transaction(trans); btrfs_btree_balance_dirty(fs_info); From 7b9a6cadd783aa04273cb29bdaea9ba9358b2b3a Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Wed, 16 Sep 2026 16:49:37 +0100 Subject: [PATCH 0051/1012] btrfs: check if there is space for chunk item when validating sys chunk array We checked if have enough remaining space for a key before dereferencing a key, but we then dereference a chunk item, to get the number of stripes, without checking if there is space for the item. So add a check to see if there is enough space for a chunk item before dereferencing the item to extract the stripe count. Fixes: 2a9bb78cfd36 ("btrfs: validate system chunk array at btrfs_validate_super()") Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index a1d83ad9a4c00c..94a7e9059a7b3f 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -2370,6 +2370,10 @@ static int validate_sys_chunk_array(const struct btrfs_fs_info *fs_info, key.type, cur); return -EUCLEAN; } + + if (unlikely(cur + sizeof(*chunk) > sys_array_size)) + goto short_read; + chunk = (struct btrfs_chunk *)(sb->sys_chunk_array + cur); num_stripes = btrfs_stack_chunk_num_stripes(chunk); if (unlikely(cur + btrfs_chunk_item_size(num_stripes) > sys_array_size)) From 4e5743baa0965f89b1c3763678d7fe1d397db02e Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Wed, 16 Sep 2026 17:11:51 +0100 Subject: [PATCH 0052/1012] btrfs: add missing unlikely to a couple error checks during sys chunk array validation It's unexpected to find errors during sys chunk array validation and all checks use the unlikely tag except for two of them, so add it. Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 2 +- fs/btrfs/tree-checker.c | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 94a7e9059a7b3f..21b72c90bd90f9 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -2387,7 +2387,7 @@ static int validate_sys_chunk_array(const struct btrfs_fs_info *fs_info, } ret = btrfs_check_chunk_valid(fs_info, NULL, chunk, key.offset, sectorsize); - if (ret < 0) + if (unlikely(ret < 0)) return ret; cur += btrfs_chunk_item_size(num_stripes); } diff --git a/fs/btrfs/tree-checker.c b/fs/btrfs/tree-checker.c index 0f4f1b347c9e8e..9cd79d97b9b5e7 100644 --- a/fs/btrfs/tree-checker.c +++ b/fs/btrfs/tree-checker.c @@ -1121,9 +1121,9 @@ int btrfs_check_chunk_valid(const struct btrfs_fs_info *fs_info, return -EUCLEAN; } - if (!remapped && - !valid_stripe_count(type & BTRFS_BLOCK_GROUP_PROFILE_MASK, - num_stripes, sub_stripes)) { + if (unlikely(!remapped && + !valid_stripe_count(type & BTRFS_BLOCK_GROUP_PROFILE_MASK, + num_stripes, sub_stripes))) { chunk_err(fs_info, leaf, chunk, logical, "invalid num_stripes:sub_stripes %u:%u for profile %llu", num_stripes, sub_stripes, From c41f076e570e760c2c30295b9abc15f1eb0dbfdb Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Wed, 16 Sep 2026 17:20:52 +0100 Subject: [PATCH 0053/1012] btrfs: remove redundant eb generation check in btrfs_buffer_uptodate() It's pointless to check if the extent buffer's generation does not match the value of 'parent_transid' because if it does, then we have already entered the previous if statement and returned from the function. Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 14 ++++++-------- 1 file changed, 6 insertions(+), 8 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 21b72c90bd90f9..161a7b9bb27a5d 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -125,15 +125,13 @@ int btrfs_buffer_uptodate(struct extent_buffer *eb, u64 parent_transid, return 1; } - if (btrfs_header_generation(eb) != parent_transid) { - btrfs_err_rl(eb->fs_info, + btrfs_err_rl(eb->fs_info, "parent transid verify failed on logical %llu mirror %u wanted %llu found %llu", - eb->start, eb->read_mirror, - parent_transid, btrfs_header_generation(eb)); - clear_extent_buffer_uptodate(eb); - return 0; - } - return 1; + eb->start, eb->read_mirror, + parent_transid, btrfs_header_generation(eb)); + clear_extent_buffer_uptodate(eb); + + return 0; } static bool btrfs_supported_super_csum(u16 csum_type) From 4a8aae3d14fc6c8022d0196853be3e46c4037d39 Mon Sep 17 00:00:00 2001 From: Filipe Manana Date: Wed, 16 Sep 2026 17:36:11 +0100 Subject: [PATCH 0054/1012] btrfs: remove duplicate error message when writing super blocks If the total error count is greater the maximum allowed number of errors, we print exactly the same error message twice. Remove one of the messages. Reviewed-by: Qu Wenruo Signed-off-by: Filipe Manana Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 161a7b9bb27a5d..a8535e309b5760 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -4228,8 +4228,6 @@ int write_all_supers(struct btrfs_trans_handle *trans) total_errors++; } if (unlikely(total_errors > max_errors)) { - btrfs_err(fs_info, "%d errors while writing supers", - total_errors); mutex_unlock(&fs_info->fs_devices->device_list_mutex); /* FUA is masked off if unsupported and can't be the reason */ From e8d38dd63712671a2d1da52988d6a5e6ef541d84 Mon Sep 17 00:00:00 2001 From: Boris Burkov Date: Wed, 16 Sep 2026 15:17:24 -0700 Subject: [PATCH 0055/1012] btrfs: keep unused block groups queued when a pass fails Once any block_group sets ret!=0 in the main loop of btrfs_delete_unused_bgs(), the check if (ret || btrfs_mixed_space_info(space_info)) { btrfs_put_block_group(block_group); continue; } skips the rest of the unused bgs while unlinking them from fs_info->unused_bgs. There is no "level triggered" re-queueing of empty block groups onto fs_info->unused_bgs so it is possible to leak quite a bit of space this way and unless we happen to get a balance or re-use/re-empty one of these bgs, they are leaked for good, which can lead to a spurious enospc later. While I have observed such leaked blocked groups that are empty but not on the unused_bgs list on production systems, I have not observed that it is definitely due to this issue. I also reproduced this behavior by injecting an ENOSPC error from btrfs_start_trans_remove_block_group which can also fail with ENOMEM, so this feels like a legitimate injection point. To fix it, instead of checking ret in the loop, just break out of the loop when ret != 0. Also, link the bg to the retry list at the individual failure sites so that the failing bg is not leaked. Assisted-by: LLM (reproducer/error injection) Reviewed-by: Qu Wenruo Signed-off-by: Boris Burkov Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index ee182369254c08..2eb09c9901c9e1 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -1612,7 +1612,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) space_info = block_group->space_info; - if (ret || btrfs_mixed_space_info(space_info)) { + if (btrfs_mixed_space_info(space_info)) { btrfs_put_block_group(block_group); continue; } @@ -1727,6 +1727,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) ret = inc_block_group_ro(block_group, false); up_write(&space_info->groups_sem); if (ret < 0) { + btrfs_link_bg_list(block_group, &retry_list); ret = 0; goto next; } @@ -1749,6 +1750,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) block_group->start); if (IS_ERR(trans)) { btrfs_dec_block_group_ro(block_group); + btrfs_link_bg_list(block_group, &retry_list); ret = PTR_ERR(trans); goto next; } @@ -1759,6 +1761,7 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) */ if (!clean_pinned_extents(trans, block_group)) { btrfs_dec_block_group_ro(block_group); + btrfs_link_bg_list(block_group, &retry_list); goto end_trans; } @@ -1845,6 +1848,8 @@ void btrfs_delete_unused_bgs(struct btrfs_fs_info *fs_info) next: btrfs_put_block_group(block_group); spin_lock(&fs_info->unused_bgs_lock); + if (ret) + break; } list_splice_tail(&retry_list, &fs_info->unused_bgs); spin_unlock(&fs_info->unused_bgs_lock); From 3484117f798eafa89f38d3171f112e5adde9958f Mon Sep 17 00:00:00 2001 From: David Sterba Date: Wed, 21 Feb 2024 15:50:10 +0100 Subject: [PATCH 0056/1012] btrfs: === misc-next on b-for-next === Any commits after this one are for testing and evaluation only. Signed-off-by: David Sterba --- fs/btrfs/Kconfig | 1 + 1 file changed, 1 insertion(+) diff --git a/fs/btrfs/Kconfig b/fs/btrfs/Kconfig index 4b10d78ed99b16..0b8d8905e38e25 100644 --- a/fs/btrfs/Kconfig +++ b/fs/btrfs/Kconfig @@ -1,4 +1,5 @@ # SPDX-License-Identifier: GPL-2.0 +# misc-next marker config BTRFS_FS tristate "Btrfs filesystem support" From 7d5739b25dcbdf9b6cd9e679fce2318be7344e02 Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:32 +0800 Subject: [PATCH 0057/1012] btrfs: always return -EIOCBQUEUED after btrfs_uring_read_extent_endio If all bios finish before btrfs_encoded_read_regular_fill_pages() returns, it calls btrfs_uring_read_extent_endio() and previously returned the I/O status. A negative errno then made btrfs_uring_read_extent() unlock and free while btrfs_uring_read_finished() did the same again. Return -EIOCBQUEUED so only the deferred path cleans up. Reported-by: Yue Sun Closes: https://lore.kernel.org/linux-btrfs/20260630091609.3414-1-samsun1006219@gmail.com/ Suggested-by: Jens Axboe Fixes: 34310c442e17 ("btrfs: add io_uring command for encoded reads (ENCODED_READ ioctl)") Signed-off-by: Yang Xiuwei Signed-off-by: David Sterba --- fs/btrfs/inode.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 53f4532593b36c..32c4dbcd3a4c86 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -9631,7 +9631,6 @@ int btrfs_encoded_read_regular_fill_pages(struct btrfs_inode *inode, struct completion sync_reads; unsigned long i = 0; struct btrfs_bio *bbio; - int ret; /* * Fast path for synchronous reads which completes in this call, io_uring @@ -9678,10 +9677,10 @@ int btrfs_encoded_read_regular_fill_pages(struct btrfs_inode *inode, if (uring_ctx) { if (refcount_dec_and_test(&priv->pending_refs)) { - ret = blk_status_to_errno(READ_ONCE(priv->status)); - btrfs_uring_read_extent_endio(uring_ctx, ret); + int err = blk_status_to_errno(READ_ONCE(priv->status)); + + btrfs_uring_read_extent_endio(uring_ctx, err); kfree(priv); - return ret; } return -EIOCBQUEUED; From bb67833f80ab7c7759fa6425a43e0d0b1db4df2b Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:33 +0800 Subject: [PATCH 0058/1012] btrfs: free iov when btrfs_uring_read_extent fails After btrfs_uring_read_extent(), the caller always jumped to out_acct. That skips kfree(data->iov), which is only correct for -EIOCBQUEUED where the deferred path owns the iov. On failure, fall through to out_free instead. Fixes: 34310c442e17 ("btrfs: add io_uring command for encoded reads (ENCODED_READ ioctl)") Signed-off-by: Yang Xiuwei Signed-off-by: David Sterba --- fs/btrfs/ioctl.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 54960351fbd171..64ccfa5fefdabf 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -4865,8 +4865,8 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue cached_state, disk_bytenr, disk_io_size, count, data->args.compression, data->iov, cmd); - - goto out_acct; + if (ret == -EIOCBQUEUED) + goto out_acct; } out_free: From a8031b21b1140ae6a12ca196d98f432ec1d30723 Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:34 +0800 Subject: [PATCH 0059/1012] btrfs: unlock inode and extent in caller when uring read extent fails btrfs_uring_read_extent() runs only after btrfs_encoded_read() has taken the inode shared lock and the extent lock. On failure it used to unlock in out_fail, and a pages-array allocation failure returned -ENOMEM without unlocking at all. Unlock in the caller instead on all failure returns, matching the copy_to_user() error path. The deferred -EIOCBQUEUED path still unlocks in btrfs_uring_read_finished(). Fixes: 34310c442e17 ("btrfs: add io_uring command for encoded reads (ENCODED_READ ioctl)") Suggested-by: Qu Wenruo Reviewed-by: Qu Wenruo Signed-off-by: Yang Xiuwei Signed-off-by: David Sterba --- fs/btrfs/ioctl.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 64ccfa5fefdabf..588c25d1968879 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -4601,7 +4601,7 @@ static void btrfs_uring_read_finished(struct io_tw_req tw_req, io_tw_token_t tw) size_t page_offset; ssize_t ret; - /* The inode lock has already been acquired in btrfs_uring_read_extent. */ + /* The inode lock has already been acquired in btrfs_encoded_read(). */ btrfs_lockdep_inode_acquire(inode, i_rwsem); if (priv->err) { @@ -4667,7 +4667,6 @@ static int btrfs_uring_read_extent(struct kiocb *iocb, struct iov_iter *iter, struct iovec *iov, struct io_uring_cmd *cmd) { struct btrfs_inode *inode = BTRFS_I(file_inode(iocb->ki_filp)); - struct extent_io_tree *io_tree = &inode->io_tree; struct page **pages = NULL; struct btrfs_uring_priv *priv = NULL; unsigned long nr_pages; @@ -4723,8 +4722,6 @@ static int btrfs_uring_read_extent(struct kiocb *iocb, struct iov_iter *iter, return -EIOCBQUEUED; out_fail: - btrfs_unlock_extent(io_tree, start, lockend, &cached_state); - btrfs_inode_unlock(inode, BTRFS_ILOCK_SHARED); kfree(priv); for (int i = 0; i < nr_pages; i++) { if (pages[i]) @@ -4867,6 +4864,8 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue data->iov, cmd); if (ret == -EIOCBQUEUED) goto out_acct; + btrfs_unlock_extent(io_tree, start, lockend, &cached_state); + btrfs_inode_unlock(inode, BTRFS_ILOCK_SHARED); } out_free: From 75df5ed2749daceace54a0b85c12c642122b1073 Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:35 +0800 Subject: [PATCH 0060/1012] btrfs: don't stash uring encoded data across -EAGAIN Returning -EAGAIN while leaving btrfs_uring_encoded_data in the cmd PDU leaks if the request is cancelled or the ring exits before reissue. io_uring does not free driver PDU allocations on cleanup. Write: io_queue_sqe() always issues with IO_URING_F_NONBLOCK first, so return -EAGAIN before allocating and free data on every exit. Read: free on nowait -EAGAIN too; only -EIOCBQUEUED keeps the allocation for btrfs_uring_read_finished(). Fixes: 34310c442e17 ("btrfs: add io_uring command for encoded reads (ENCODED_READ ioctl)") Fixes: e32dcdb0af9f ("btrfs: add io_uring interface for encoded writes") Signed-off-by: Yang Xiuwei Signed-off-by: David Sterba --- fs/btrfs/ioctl.c | 20 +++++++++++--------- 1 file changed, 11 insertions(+), 9 deletions(-) diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 588c25d1968879..25d9feaa90030f 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -4834,7 +4834,7 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue ret = btrfs_encoded_read(&kiocb, &data->iter, &data->args, &cached_state, &disk_bytenr, &disk_io_size); if (ret == -EAGAIN) - goto out_acct; + goto out_free; if (ret < 0 && ret != -EIOCBQUEUED) goto out_free; @@ -4876,8 +4876,10 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue add_rchar(current, ret); inc_syscr(current); - if (ret != -EIOCBQUEUED && ret != -EAGAIN) + if (ret != -EIOCBQUEUED) { kfree(data); + bc->data = NULL; + } return ret; } @@ -4906,6 +4908,11 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu goto out_acct; } + if (issue_flags & IO_URING_F_NONBLOCK) { + ret = -EAGAIN; + goto out_acct; + } + if (!data) { data = kzalloc_obj(*data, GFP_NOFS); if (!data) { @@ -4974,11 +4981,6 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu } } - if (issue_flags & IO_URING_F_NONBLOCK) { - ret = -EAGAIN; - goto out_acct; - } - pos = data->args.offset; ret = rw_verify_area(WRITE, file, &pos, data->args.len); if (ret < 0) @@ -5004,8 +5006,8 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu add_wchar(current, ret); inc_syscw(current); - if (ret != -EAGAIN) - kfree(data); + kfree(data); + bc->data = NULL; return ret; } From 438d99f9a81f2c06296feb6831ef72780d6506cf Mon Sep 17 00:00:00 2001 From: Yang Xiuwei Date: Wed, 19 Aug 2026 10:54:36 +0800 Subject: [PATCH 0061/1012] btrfs: drop unused uring encoded IO REISSUE stash helpers After not keeping state across -EAGAIN, restoring bc->data on REISSUE is dead. Remove it, stop using the cmd PDU on the write path, and fold the read -EAGAIN check into the existing error path. Signed-off-by: Yang Xiuwei Signed-off-by: David Sterba --- fs/btrfs/ioctl.c | 12 ------------ 1 file changed, 12 deletions(-) diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c index 25d9feaa90030f..52aab510aea0d1 100644 --- a/fs/btrfs/ioctl.c +++ b/fs/btrfs/ioctl.c @@ -4749,9 +4749,6 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue struct io_btrfs_cmd *bc = io_uring_cmd_to_pdu(cmd, struct io_btrfs_cmd); struct btrfs_uring_encoded_data *data = NULL; - if (cmd->flags & IORING_URING_CMD_REISSUE) - data = bc->data; - if (!capable(CAP_SYS_ADMIN)) { ret = -EPERM; goto out_acct; @@ -4833,8 +4830,6 @@ static int btrfs_uring_encoded_read(struct io_uring_cmd *cmd, unsigned int issue ret = btrfs_encoded_read(&kiocb, &data->iter, &data->args, &cached_state, &disk_bytenr, &disk_io_size); - if (ret == -EAGAIN) - goto out_free; if (ret < 0 && ret != -EIOCBQUEUED) goto out_free; @@ -4891,12 +4886,8 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu struct kiocb kiocb; ssize_t ret; void __user *sqe_addr; - struct io_btrfs_cmd *bc = io_uring_cmd_to_pdu(cmd, struct io_btrfs_cmd); struct btrfs_uring_encoded_data *data = NULL; - if (cmd->flags & IORING_URING_CMD_REISSUE) - data = bc->data; - if (!capable(CAP_SYS_ADMIN)) { ret = -EPERM; goto out_acct; @@ -4920,8 +4911,6 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu goto out_acct; } - bc->data = data; - if (issue_flags & IO_URING_F_COMPAT) { #if defined(CONFIG_64BIT) && defined(CONFIG_COMPAT) struct btrfs_ioctl_encoded_io_args_32 args32; @@ -5007,7 +4996,6 @@ static int btrfs_uring_encoded_write(struct io_uring_cmd *cmd, unsigned int issu inc_syscw(current); kfree(data); - bc->data = NULL; return ret; } From 008137ce11439e69527212afad480c92cfef1eed Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Wed, 16 Sep 2026 23:59:57 -0400 Subject: [PATCH 0062/1012] btrfs: stop enabling the v1 space cache from the on-disk state Since commit 545e560a5b0f ("btrfs: disable v1 space cache") the mount options can no longer request the v1 space cache, but a filesystem with an active v1 cache and no free space tree still enables it from cache_generation, and remount does the same. Drop both, so SPACE_CACHE can never be set. btrfs_start_pre_rw_mount() then sees the on-disk cache as active but unwanted and cleans it up, as -o nospace_cache does today. That covers the read-only to read-write remount as well, so drop the toggle in btrfs_remount_cleanup(), which would otherwise start a transaction on remounts of a read-only filesystem with an old cache. The cleanup is now unconditional, and the first read-write mount fails if it fails, as it did with -o nospace_cache. This also lets an old filesystem mount without options when the page size is larger than the sector size, which btrfs_check_features() rejected once SPACE_CACHE was set from the superblock. Assisted-by: Claude:claude-fable-5-1 Reviewed-by: Qu Wenruo Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/super.c | 16 +++------------- 1 file changed, 3 insertions(+), 13 deletions(-) diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c index 464129b1b0d4cb..77443ded6db393 100644 --- a/fs/btrfs/super.c +++ b/fs/btrfs/super.c @@ -759,12 +759,12 @@ void btrfs_set_free_space_cache_settings(struct btrfs_fs_info *fs_info) /* * At this point we don't have explicit options set by the user, set - * them ourselves based on the state of the file system. + * them ourselves based on the state of the file system. An existing + * v1 space cache is no longer used and gets cleaned up once the + * filesystem is mounted read-write. */ if (btrfs_fs_compat_ro(fs_info, FREE_SPACE_TREE)) btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE); - else if (btrfs_free_space_cache_v1_active(fs_info)) - btrfs_set_opt(fs_info->mount_opt, SPACE_CACHE); } static void set_device_specific_options(struct btrfs_fs_info *fs_info) @@ -1264,8 +1264,6 @@ static inline void btrfs_remount_begin(struct btrfs_fs_info *fs_info, static inline void btrfs_remount_cleanup(struct btrfs_fs_info *fs_info, unsigned long long old_opts) { - const bool cache_opt = btrfs_test_opt(fs_info, SPACE_CACHE); - /* * We need to cleanup all defraggable inodes if the autodefragment is * close or the filesystem is read only. @@ -1282,10 +1280,6 @@ static inline void btrfs_remount_cleanup(struct btrfs_fs_info *fs_info, else if (btrfs_raw_test_opt(old_opts, DISCARD_ASYNC) && !btrfs_test_opt(fs_info, DISCARD_ASYNC)) btrfs_discard_cleanup(fs_info); - - /* If we toggled space cache */ - if (cache_opt != btrfs_free_space_cache_v1_active(fs_info)) - btrfs_set_free_space_cache_v1_active(fs_info, cache_opt); } static int btrfs_remount_rw(struct btrfs_fs_info *fs_info) @@ -1535,10 +1529,6 @@ static int btrfs_reconfigure(struct fs_context *fc) btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE); btrfs_clear_opt(fs_info->mount_opt, SPACE_CACHE); } - if (btrfs_free_space_cache_v1_active(fs_info)) { - btrfs_clear_opt(fs_info->mount_opt, FREE_SPACE_TREE); - btrfs_set_opt(fs_info->mount_opt, SPACE_CACHE); - } } ret = 0; From 1c38cbb3d2c51fdbc96b67c98349802f617720f5 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Wed, 16 Sep 2026 23:59:58 -0400 Subject: [PATCH 0063/1012] btrfs: remove the v1 space cache writeout from the transaction commit Nothing sets SPACE_CACHE anymore, so the dirty block group writers never have a cache to write out or wait for. Remove cache_save_setup(), btrfs_setup_space_cache(), the io_list handling, the io_bgs list and BTRFS_TRANS_CACHE_ENOSPC, and the abort-time cleanup of in-flight cache IO. The -ENOENT retry in btrfs_write_dirty_block_groups() handled a free space endio worker creating a block group during the commit critical section, so drop it too. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 397 +++-------------------------------------- fs/btrfs/block-group.h | 1 - fs/btrfs/disk-io.c | 44 ----- fs/btrfs/transaction.c | 9 +- fs/btrfs/transaction.h | 18 -- 5 files changed, 26 insertions(+), 443 deletions(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index 2eb09c9901c9e1..985141e4f133ba 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -1194,29 +1194,10 @@ int btrfs_remove_block_group(struct btrfs_trans_handle *trans, goto out; } - /* - * get the inode first so any iput calls done for the io_list - * aren't the final iput (no unlinks allowed now) - */ inode = lookup_free_space_inode(block_group, path); mutex_lock(&trans->transaction->cache_write_mutex); - /* - * Make sure our free space cache IO is done before removing the - * free space inode - */ spin_lock(&trans->transaction->dirty_bgs_lock); - if (!list_empty(&block_group->io_list)) { - list_del_init(&block_group->io_list); - - WARN_ON(!IS_ERR(inode) && inode != block_group->io_ctl.inode); - - spin_unlock(&trans->transaction->dirty_bgs_lock); - btrfs_wait_cache_io(trans, block_group, path); - btrfs_put_block_group(block_group); - spin_lock(&trans->transaction->dirty_bgs_lock); - } - if (!list_empty(&block_group->dirty_list)) { list_del_init(&block_group->dirty_list); remove_rsv = true; @@ -3378,197 +3359,6 @@ static int update_block_group_item(struct btrfs_trans_handle *trans, } -static void cache_save_setup(struct btrfs_block_group *block_group, - struct btrfs_trans_handle *trans, - struct btrfs_path *path) -{ - struct btrfs_fs_info *fs_info = block_group->fs_info; - struct inode *inode = NULL; - struct extent_changeset *data_reserved = NULL; - u64 alloc_hint = 0; - int dcs = BTRFS_DC_ERROR; - u64 cache_size = 0; - int retries = 0; - int ret = 0; - - if (!btrfs_test_opt(fs_info, SPACE_CACHE)) - return; - - /* - * If this block group is smaller than 100 megs don't bother caching the - * block group. - */ - if (block_group->length < (100 * SZ_1M)) { - spin_lock(&block_group->lock); - block_group->disk_cache_state = BTRFS_DC_WRITTEN; - spin_unlock(&block_group->lock); - return; - } - - if (TRANS_ABORTED(trans)) - return; -again: - inode = lookup_free_space_inode(block_group, path); - if (IS_ERR(inode) && PTR_ERR(inode) != -ENOENT) { - ret = PTR_ERR(inode); - btrfs_release_path(path); - goto out; - } - - if (IS_ERR(inode)) { - if (retries) { - ret = PTR_ERR(inode); - btrfs_err(fs_info, - "failed to lookup free space inode after creation for block group %llu: %d", - block_group->start, ret); - goto out_free; - } - retries++; - - if (block_group->ro) - goto out_free; - - ret = create_free_space_inode(trans, block_group, path); - if (ret) - goto out_free; - goto again; - } - - /* - * We want to set the generation to 0, that way if anything goes wrong - * from here on out we know not to trust this cache when we load up next - * time. - */ - BTRFS_I(inode)->generation = 0; - ret = btrfs_update_inode(trans, BTRFS_I(inode)); - if (unlikely(ret)) { - /* - * So theoretically we could recover from this, simply set the - * super cache generation to 0 so we know to invalidate the - * cache, but then we'd have to keep track of the block groups - * that fail this way so we know we _have_ to reset this cache - * before the next commit or risk reading stale cache. So to - * limit our exposure to horrible edge cases lets just abort the - * transaction, this only happens in really bad situations - * anyway. - */ - btrfs_abort_transaction(trans, ret); - goto out_put; - } - - /* We've already setup this transaction, go ahead and exit */ - if (block_group->cache_generation == trans->transid && - i_size_read(inode)) { - dcs = BTRFS_DC_SETUP; - goto out_put; - } - - if (i_size_read(inode) > 0) { - ret = btrfs_check_trunc_cache_free_space(fs_info, - &fs_info->global_block_rsv); - if (ret) - goto out_put; - - ret = btrfs_truncate_free_space_cache(trans, NULL, inode); - if (ret) - goto out_put; - } - - spin_lock(&block_group->lock); - if (block_group->cached != BTRFS_CACHE_FINISHED || - !btrfs_test_opt(fs_info, SPACE_CACHE)) { - /* - * don't bother trying to write stuff out _if_ - * a) we're not cached, - * b) we're with nospace_cache mount option, - * c) we're with v2 space_cache (FREE_SPACE_TREE). - */ - dcs = BTRFS_DC_WRITTEN; - spin_unlock(&block_group->lock); - goto out_put; - } - spin_unlock(&block_group->lock); - - /* - * We hit an ENOSPC when setting up the cache in this transaction, just - * skip doing the setup, we've already cleared the cache so we're safe. - */ - if (test_bit(BTRFS_TRANS_CACHE_ENOSPC, &trans->transaction->flags)) - goto out_put; - - /* - * Try to preallocate enough space based on how big the block group is. - * Keep in mind this has to include any pinned space which could end up - * taking up quite a bit since it's not folded into the other space - * cache. - */ - cache_size = div_u64(block_group->length, SZ_256M); - if (!cache_size) - cache_size = 1; - - cache_size *= 16; - cache_size *= fs_info->sectorsize; - - ret = btrfs_check_data_free_space(BTRFS_I(inode), &data_reserved, 0, - cache_size, false); - if (ret) - goto out_put; - - ret = btrfs_prealloc_file_range_trans(inode, trans, 0, 0, cache_size, - cache_size, cache_size, - &alloc_hint); - /* - * Our cache requires contiguous chunks so that we don't modify a bunch - * of metadata or split extents when writing the cache out, which means - * we can enospc if we are heavily fragmented in addition to just normal - * out of space conditions. So if we hit this just skip setting up any - * other block groups for this transaction, maybe we'll unpin enough - * space the next time around. - */ - if (!ret) - dcs = BTRFS_DC_SETUP; - else if (ret == -ENOSPC) - set_bit(BTRFS_TRANS_CACHE_ENOSPC, &trans->transaction->flags); - -out_put: - iput(inode); -out_free: - btrfs_release_path(path); -out: - spin_lock(&block_group->lock); - if (!ret && dcs == BTRFS_DC_SETUP) - block_group->cache_generation = trans->transid; - block_group->disk_cache_state = dcs; - spin_unlock(&block_group->lock); - - extent_changeset_free(data_reserved); -} - -int btrfs_setup_space_cache(struct btrfs_trans_handle *trans) -{ - struct btrfs_fs_info *fs_info = trans->fs_info; - struct btrfs_block_group *cache, *tmp; - struct btrfs_transaction *cur_trans = trans->transaction; - BTRFS_PATH_AUTO_FREE(path); - - if (list_empty(&cur_trans->dirty_bgs) || - !btrfs_test_opt(fs_info, SPACE_CACHE)) - return 0; - - path = btrfs_alloc_path(); - if (!path) - return -ENOMEM; - - /* Could add new block groups, use _safe just in case */ - list_for_each_entry_safe(cache, tmp, &cur_trans->dirty_bgs, - dirty_list) { - if (cache->disk_cache_state == BTRFS_DC_CLEAR) - cache_save_setup(cache, trans, path); - } - - return 0; -} - /* * Transaction commit does final block group cache writeback during a critical * section where nothing is allowed to change the FS. This is required in @@ -3587,10 +3377,8 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) struct btrfs_block_group *cache; struct btrfs_transaction *cur_trans = trans->transaction; int ret = 0; - int should_put; BTRFS_PATH_AUTO_FREE(path); LIST_HEAD(dirty); - struct list_head *io = &cur_trans->io_bgs; int loops = 0; spin_lock(&cur_trans->dirty_bgs_lock); @@ -3616,7 +3404,7 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) /* * cache_write_mutex is here only to save us from balance or automatic * removal of empty block groups deleting this block group while we are - * writing out the cache + * updating its item */ mutex_lock(&trans->transaction->cache_write_mutex); while (!list_empty(&dirty)) { @@ -3624,23 +3412,8 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) cache = list_first_entry(&dirty, struct btrfs_block_group, dirty_list); - /* - * This can happen if something re-dirties a block group that - * is already under IO. Just wait for it to finish and then do - * it all again - */ - if (!list_empty(&cache->io_list)) { - list_del_init(&cache->io_list); - btrfs_wait_cache_io(trans, cache, path); - btrfs_put_block_group(cache); - } - /* - * btrfs_wait_cache_io uses the cache->dirty_list to decide if - * it should update the cache_state. Don't delete until after - * we wait. - * * Since we're not running in the commit critical section * we need the dirty_bgs_lock to protect from update_block_group */ @@ -3648,66 +3421,33 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) list_del_init(&cache->dirty_list); spin_unlock(&cur_trans->dirty_bgs_lock); - should_put = 1; - - cache_save_setup(cache, trans, path); - - if (cache->disk_cache_state == BTRFS_DC_SETUP) { - cache->io_ctl.inode = NULL; - ret = btrfs_write_out_cache(trans, cache, path); - if (ret == 0 && cache->io_ctl.inode) { - should_put = 0; - - /* - * The cache_write_mutex is protecting the - * io_list, also refer to the definition of - * btrfs_transaction::io_bgs for more details - */ - list_add_tail(&cache->io_list, io); - } else { - /* - * If we failed to write the cache, the - * generation will be bad and life goes on - */ - ret = 0; - } - } - if (!ret) { - ret = update_block_group_item(trans, path, cache); - /* - * Our block group might still be attached to the list - * of new block groups in the transaction handle of some - * other task (struct btrfs_trans_handle->new_bgs). This - * means its block group item isn't yet in the extent - * tree. If this happens ignore the error, as we will - * try again later in the critical section of the - * transaction commit. - */ - if (ret == -ENOENT) { - ret = 0; - spin_lock(&cur_trans->dirty_bgs_lock); - if (list_empty(&cache->dirty_list)) { - list_add_tail(&cache->dirty_list, - &cur_trans->dirty_bgs); - btrfs_get_block_group(cache); - drop_reserve = false; - } - spin_unlock(&cur_trans->dirty_bgs_lock); - } else if (ret) { - btrfs_abort_transaction(trans, ret); + ret = update_block_group_item(trans, path, cache); + /* + * Our block group might still be attached to the list of new + * block groups in the transaction handle of some other task + * (struct btrfs_trans_handle->new_bgs). This means its block + * group item isn't yet in the extent tree. If this happens + * ignore the error, as we will try again later in the critical + * section of the transaction commit. + */ + if (ret == -ENOENT) { + ret = 0; + spin_lock(&cur_trans->dirty_bgs_lock); + if (list_empty(&cache->dirty_list)) { + list_add_tail(&cache->dirty_list, + &cur_trans->dirty_bgs); + btrfs_get_block_group(cache); + drop_reserve = false; } + spin_unlock(&cur_trans->dirty_bgs_lock); + } else if (ret) { + btrfs_abort_transaction(trans, ret); } - /* If it's not on the io list, we need to put the block group */ - if (should_put) - btrfs_put_block_group(cache); + btrfs_put_block_group(cache); if (drop_reserve) btrfs_dec_delayed_refs_rsv_bg_updates(fs_info); - /* - * Avoid blocking other tasks for too long. It might even save - * us from writing caches for block groups that are going to be - * removed. - */ + /* Avoid blocking other tasks for too long. */ mutex_unlock(&trans->transaction->cache_write_mutex); if (ret) goto out; @@ -3752,121 +3492,34 @@ int btrfs_write_dirty_block_groups(struct btrfs_trans_handle *trans) struct btrfs_block_group *cache; struct btrfs_transaction *cur_trans = trans->transaction; int ret = 0; - int should_put; BTRFS_PATH_AUTO_FREE(path); - struct list_head *io = &cur_trans->io_bgs; path = btrfs_alloc_path(); if (!path) return -ENOMEM; - /* - * Even though we are in the critical section of the transaction commit, - * we can still have concurrent tasks adding elements to this - * transaction's list of dirty block groups. These tasks correspond to - * endio free space workers started when writeback finishes for a - * space cache, which run inode.c:btrfs_finish_ordered_io(), and can - * allocate new block groups as a result of COWing nodes of the root - * tree when updating the free space inode. The writeback for the space - * caches is triggered by an earlier call to - * btrfs_start_dirty_block_groups() and iterations of the following - * loop. - * Also we want to do the cache_save_setup first and then run the - * delayed refs to make sure we have the best chance at doing this all - * in one shot. - */ spin_lock(&cur_trans->dirty_bgs_lock); while (!list_empty(&cur_trans->dirty_bgs)) { cache = list_first_entry(&cur_trans->dirty_bgs, struct btrfs_block_group, dirty_list); - - /* - * This can happen if cache_save_setup re-dirties a block group - * that is already under IO. Just wait for it to finish and - * then do it all again - */ - if (!list_empty(&cache->io_list)) { - spin_unlock(&cur_trans->dirty_bgs_lock); - list_del_init(&cache->io_list); - btrfs_wait_cache_io(trans, cache, path); - btrfs_put_block_group(cache); - spin_lock(&cur_trans->dirty_bgs_lock); - } - - /* - * Don't remove from the dirty list until after we've waited on - * any pending IO - */ list_del_init(&cache->dirty_list); spin_unlock(&cur_trans->dirty_bgs_lock); - should_put = 1; - - cache_save_setup(cache, trans, path); if (!ret) ret = btrfs_run_delayed_refs(trans, U64_MAX); - - if (!ret && cache->disk_cache_state == BTRFS_DC_SETUP) { - cache->io_ctl.inode = NULL; - ret = btrfs_write_out_cache(trans, cache, path); - if (ret == 0 && cache->io_ctl.inode) { - should_put = 0; - list_add_tail(&cache->io_list, io); - } else { - /* - * If we failed to write the cache, the - * generation will be bad and life goes on - */ - ret = 0; - } - } if (!ret) { ret = update_block_group_item(trans, path, cache); - /* - * One of the free space endio workers might have - * created a new block group while updating a free space - * cache's inode (at inode.c:btrfs_finish_ordered_io()) - * and hasn't released its transaction handle yet, in - * which case the new block group is still attached to - * its transaction handle and its creation has not - * finished yet (no block group item in the extent tree - * yet, etc). If this is the case, wait for all free - * space endio workers to finish and retry. This is a - * very rare case so no need for a more efficient and - * complex approach. - */ - if (ret == -ENOENT) { - wait_event(cur_trans->writer_wait, - atomic_read(&cur_trans->num_writers) == 1); - ret = update_block_group_item(trans, path, cache); - if (ret) - btrfs_abort_transaction(trans, ret); - } else if (ret) { + if (ret) btrfs_abort_transaction(trans, ret); - } } - /* If its not on the io list, we need to put the block group */ - if (should_put) - btrfs_put_block_group(cache); + btrfs_put_block_group(cache); btrfs_dec_delayed_refs_rsv_bg_updates(fs_info); spin_lock(&cur_trans->dirty_bgs_lock); } spin_unlock(&cur_trans->dirty_bgs_lock); - /* - * Refer to the definition of io_bgs member for details why it's safe - * to use it without any locking - */ - while (!list_empty(io)) { - cache = list_first_entry(io, struct btrfs_block_group, - io_list); - list_del_init(&cache->io_list); - btrfs_wait_cache_io(trans, cache, path); - btrfs_put_block_group(cache); - } - return ret; } diff --git a/fs/btrfs/block-group.h b/fs/btrfs/block-group.h index b349f94cf929ab..d69432b236ec02 100644 --- a/fs/btrfs/block-group.h +++ b/fs/btrfs/block-group.h @@ -368,7 +368,6 @@ int btrfs_inc_block_group_ro(struct btrfs_block_group *cache, void btrfs_dec_block_group_ro(struct btrfs_block_group *cache); int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans); int btrfs_write_dirty_block_groups(struct btrfs_trans_handle *trans); -int btrfs_setup_space_cache(struct btrfs_trans_handle *trans); int btrfs_update_block_group(struct btrfs_trans_handle *trans, u64 bytenr, u64 num_bytes, bool alloc); int btrfs_add_reserved_bytes(struct btrfs_block_group *cache, diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index a8535e309b5760..2acd62b325ea0f 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -4873,26 +4873,6 @@ static void btrfs_destroy_pinned_extent(struct btrfs_fs_info *fs_info, } } -static void btrfs_cleanup_bg_io(struct btrfs_block_group *cache) -{ - struct inode *inode; - - inode = cache->io_ctl.inode; - if (inode) { - unsigned int nofs_flag; - - nofs_flag = memalloc_nofs_save(); - invalidate_inode_pages2(inode->i_mapping); - memalloc_nofs_restore(nofs_flag); - - BTRFS_I(inode)->generation = 0; - cache->io_ctl.inode = NULL; - iput(inode); - } - ASSERT(cache->io_ctl.pages == NULL); - btrfs_put_block_group(cache); -} - void btrfs_cleanup_dirty_bgs(struct btrfs_transaction *cur_trans, struct btrfs_fs_info *fs_info) { @@ -4904,13 +4884,6 @@ void btrfs_cleanup_dirty_bgs(struct btrfs_transaction *cur_trans, struct btrfs_block_group, dirty_list); - if (!list_empty(&cache->io_list)) { - spin_unlock(&cur_trans->dirty_bgs_lock); - list_del_init(&cache->io_list); - btrfs_cleanup_bg_io(cache); - spin_lock(&cur_trans->dirty_bgs_lock); - } - list_del_init(&cache->dirty_list); spin_lock(&cache->lock); cache->disk_cache_state = BTRFS_DC_ERROR; @@ -4922,22 +4895,6 @@ void btrfs_cleanup_dirty_bgs(struct btrfs_transaction *cur_trans, spin_lock(&cur_trans->dirty_bgs_lock); } spin_unlock(&cur_trans->dirty_bgs_lock); - - /* - * Refer to the definition of io_bgs member for details why it's safe - * to use it without any locking - */ - while (!list_empty(&cur_trans->io_bgs)) { - cache = list_first_entry(&cur_trans->io_bgs, - struct btrfs_block_group, - io_list); - - list_del_init(&cache->io_list); - spin_lock(&cache->lock); - cache->disk_cache_state = BTRFS_DC_ERROR; - spin_unlock(&cache->lock); - btrfs_cleanup_bg_io(cache); - } } static void btrfs_free_all_qgroup_pertrans(struct btrfs_fs_info *fs_info) @@ -4973,7 +4930,6 @@ void btrfs_cleanup_one_transaction(struct btrfs_transaction *cur_trans) btrfs_cleanup_dirty_bgs(cur_trans, fs_info); ASSERT(list_empty(&cur_trans->dirty_bgs)); - ASSERT(list_empty(&cur_trans->io_bgs)); list_for_each_entry_safe(dev, tmp, &cur_trans->dev_update_list, post_commit_list) { diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index ca114235bbe1e1..a5097714578b2a 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -379,7 +379,6 @@ static noinline int join_transaction(struct btrfs_fs_info *fs_info, INIT_LIST_HEAD(&cur_trans->dev_update_list); INIT_LIST_HEAD(&cur_trans->switch_commits); INIT_LIST_HEAD(&cur_trans->dirty_bgs); - INIT_LIST_HEAD(&cur_trans->io_bgs); INIT_LIST_HEAD(&cur_trans->dropped_roots); mutex_init(&cur_trans->cache_write_mutex); spin_lock_init(&cur_trans->dirty_bgs_lock); @@ -1363,7 +1362,6 @@ static noinline int commit_cowonly_roots(struct btrfs_trans_handle *trans) { struct btrfs_fs_info *fs_info = trans->fs_info; struct list_head *dirty_bgs = &trans->transaction->dirty_bgs; - struct list_head *io_bgs = &trans->transaction->io_bgs; struct extent_buffer *eb; int ret; @@ -1393,10 +1391,6 @@ static noinline int commit_cowonly_roots(struct btrfs_trans_handle *trans) if (ret) return ret; - ret = btrfs_setup_space_cache(trans); - if (ret) - return ret; - again: while (!list_empty(&fs_info->dirty_cowonly_roots)) { struct btrfs_root *root; @@ -1417,7 +1411,7 @@ static noinline int commit_cowonly_roots(struct btrfs_trans_handle *trans) if (ret) return ret; - while (!list_empty(dirty_bgs) || !list_empty(io_bgs)) { + while (!list_empty(dirty_bgs)) { ret = btrfs_write_dirty_block_groups(trans); if (ret) return ret; @@ -2542,7 +2536,6 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans) switch_commit_roots(trans); ASSERT(list_empty(&cur_trans->dirty_bgs)); - ASSERT(list_empty(&cur_trans->io_bgs)); update_super_roots(fs_info); btrfs_set_super_log_root(fs_info->super_copy, 0); diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h index 89153cd2259678..24b9be1833afb1 100644 --- a/fs/btrfs/transaction.h +++ b/fs/btrfs/transaction.h @@ -47,7 +47,6 @@ enum btrfs_trans_state { #define BTRFS_TRANS_HAVE_FREE_BGS 0 #define BTRFS_TRANS_DIRTY_BG_RUN 1 -#define BTRFS_TRANS_CACHE_ENOSPC 2 struct btrfs_transaction { u64 transid; @@ -78,23 +77,6 @@ struct btrfs_transaction { struct list_head dev_update_list; struct list_head switch_commits; struct list_head dirty_bgs; - - /* - * There is no explicit lock which protects io_bgs, rather its - * consistency is implied by the fact that all the sites which modify - * it do so under some form of transaction critical section, namely: - * - * - btrfs_start_dirty_block_groups - This function can only ever be - * run by one of the transaction committers. Refer to - * BTRFS_TRANS_DIRTY_BG_RUN usage in btrfs_commit_transaction - * - * - btrfs_write_dirty_blockgroups - this is called by - * commit_cowonly_roots from transaction critical section - * (TRANS_STATE_COMMIT_DOING) - * - * - btrfs_cleanup_dirty_bgs - called on transaction abort - */ - struct list_head io_bgs; struct list_head dropped_roots; struct extent_io_tree pinned_extents; From 79eb7449cd70675c0fdf5ed504f086de924da721 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Wed, 16 Sep 2026 23:59:59 -0400 Subject: [PATCH 0064/1012] btrfs: remove the free space cache endio workqueue Free space inodes are no longer written to, so nothing queues ordered extent completion on endio_freespace_worker. Remove it and always use endio_write_workers in btrfs_queue_ordered_fn(). Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 14 +++----------- fs/btrfs/fs.h | 1 - fs/btrfs/ordered-data.c | 7 ++----- fs/btrfs/super.c | 1 - 4 files changed, 5 insertions(+), 18 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 2acd62b325ea0f..2cf9638cd864e4 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -1780,7 +1780,6 @@ static void btrfs_stop_all_workers(struct btrfs_fs_info *fs_info) if (fs_info->rmw_workers) destroy_workqueue(fs_info->rmw_workers); btrfs_destroy_workqueue(fs_info->endio_write_workers); - btrfs_destroy_workqueue(fs_info->endio_freespace_worker); btrfs_destroy_workqueue(fs_info->delayed_workers); btrfs_destroy_workqueue(fs_info->caching_workers); btrfs_destroy_workqueue(fs_info->flush_workers); @@ -1991,9 +1990,6 @@ static int btrfs_init_workqueues(struct btrfs_fs_info *fs_info) fs_info->endio_write_workers = btrfs_alloc_workqueue(fs_info, "endio-write", flags, max_active, 2); - fs_info->endio_freespace_worker = - btrfs_alloc_workqueue(fs_info, "freespace-write", flags, - max_active, 0); fs_info->delayed_workers = btrfs_alloc_workqueue(fs_info, "delayed-meta", flags, max_active, 0); @@ -2006,8 +2002,7 @@ static int btrfs_init_workqueues(struct btrfs_fs_info *fs_info) if (!(fs_info->workers && fs_info->delalloc_workers && fs_info->flush_workers && fs_info->endio_workers && fs_info->endio_meta_workers && - fs_info->endio_write_workers && - fs_info->endio_freespace_worker && fs_info->rmw_workers && + fs_info->endio_write_workers && fs_info->rmw_workers && fs_info->caching_workers && fs_info->fixup_workers && fs_info->delayed_workers && fs_info->qgroup_rescan_workers && fs_info->discard_ctl.discard_workers)) { @@ -4455,9 +4450,8 @@ void __cold close_ctree(struct btrfs_fs_info *fs_info) * to finish an ordered extent - end_bbio_compressed_write() * calls btrfs_finish_ordered_extent() which in turns does a call to * btrfs_queue_ordered_fn(), and that queues the ordered extent - * completion either in the endio_write_workers work queue or in the - * fs_info->endio_freespace_worker work queue. We flush those queues - * below, so before we flush them we must flush this queue for the + * completion in the endio_write_workers work queue. We flush that + * queue below, so before we flush it we must flush this queue for the * workers of compressed writes. */ flush_workqueue(fs_info->endio_workers); @@ -4483,8 +4477,6 @@ void __cold close_ctree(struct btrfs_fs_info *fs_info) * btrfs_finish_ordered_io() when we are unmounting). */ btrfs_flush_workqueue(fs_info->endio_write_workers); - /* Ordered extents for free space inodes. */ - btrfs_flush_workqueue(fs_info->endio_freespace_worker); /* * Run delayed iputs in case an async reclaim worker is waiting for them * to be run as mentioned above. diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h index 3eba8438593cde..441b315e8a9893 100644 --- a/fs/btrfs/fs.h +++ b/fs/btrfs/fs.h @@ -712,7 +712,6 @@ struct btrfs_fs_info { struct workqueue_struct *endio_meta_workers; struct workqueue_struct *rmw_workers; struct btrfs_workqueue *endio_write_workers; - struct btrfs_workqueue *endio_freespace_worker; struct btrfs_workqueue *caching_workers; struct workqueue_struct *fixup_workers; diff --git a/fs/btrfs/ordered-data.c b/fs/btrfs/ordered-data.c index b32d4eabe0abe4..e9f1cbeb555a4c 100644 --- a/fs/btrfs/ordered-data.c +++ b/fs/btrfs/ordered-data.c @@ -417,13 +417,10 @@ static bool can_finish_ordered_extent(struct btrfs_ordered_extent *ordered, static void btrfs_queue_ordered_fn(struct btrfs_ordered_extent *ordered) { - struct btrfs_inode *inode = ordered->inode; - struct btrfs_fs_info *fs_info = inode->root->fs_info; - struct btrfs_workqueue *wq = btrfs_is_free_space_inode(inode) ? - fs_info->endio_freespace_worker : fs_info->endio_write_workers; + struct btrfs_fs_info *fs_info = ordered->inode->root->fs_info; btrfs_init_work(&ordered->work, finish_ordered_fn, NULL); - btrfs_queue_work(wq, &ordered->work); + btrfs_queue_work(fs_info->endio_write_workers, &ordered->work); } void btrfs_finish_ordered_extent(struct btrfs_ordered_extent *ordered, diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c index 77443ded6db393..b44b16970a6233 100644 --- a/fs/btrfs/super.c +++ b/fs/btrfs/super.c @@ -1243,7 +1243,6 @@ static void btrfs_resize_thread_pool(struct btrfs_fs_info *fs_info, workqueue_set_max_active(fs_info->endio_workers, new_pool_size); workqueue_set_max_active(fs_info->endio_meta_workers, new_pool_size); btrfs_workqueue_set_max(fs_info->endio_write_workers, new_pool_size); - btrfs_workqueue_set_max(fs_info->endio_freespace_worker, new_pool_size); btrfs_workqueue_set_max(fs_info->delayed_workers, new_pool_size); } From e3437ffad54d0220b3a897e9b89085ef809613e9 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:00 -0400 Subject: [PATCH 0065/1012] btrfs: remove the v1 space cache write path Nothing writes out a v1 space cache any more. Remove the writers and their io_ctl helpers, along with create_free_space_inode() and btrfs_prealloc_file_range_trans(), whose only user was the cache inode creation. The io_list and io_ctl block group fields were only used by the writers, so remove them too. btrfs_truncate_free_space_cache() only needed the block group to wait for and clear in-flight cache IO, so drop that parameter. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 4 - fs/btrfs/block-group.h | 3 - fs/btrfs/btrfs_inode.h | 4 - fs/btrfs/free-space-cache.c | 688 ------------------------------------ fs/btrfs/free-space-cache.h | 10 - fs/btrfs/inode.c | 9 - fs/btrfs/relocation.c | 2 +- 7 files changed, 1 insertion(+), 719 deletions(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index 985141e4f133ba..03d5779358ca29 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -1268,7 +1268,6 @@ int btrfs_remove_block_group(struct btrfs_trans_handle *trans, spin_lock(&trans->transaction->dirty_bgs_lock); WARN_ON(!list_empty(&block_group->dirty_list)); - WARN_ON(!list_empty(&block_group->io_list)); spin_unlock(&trans->transaction->dirty_bgs_lock); btrfs_remove_free_space_cache(block_group); @@ -2417,7 +2416,6 @@ static struct btrfs_block_group *btrfs_create_block_group( INIT_LIST_HEAD(&cache->ro_list); INIT_LIST_HEAD(&cache->discard_list); INIT_LIST_HEAD(&cache->dirty_list); - INIT_LIST_HEAD(&cache->io_list); INIT_LIST_HEAD(&cache->active_bg_list); btrfs_init_free_space_ctl(cache, cache->free_space_ctl); atomic_set(&cache->frozen, 0); @@ -4298,7 +4296,6 @@ void btrfs_put_block_group_cache(struct btrfs_fs_info *info) block_group->inode = NULL; spin_unlock(&block_group->lock); - ASSERT(block_group->io_ctl.inode == NULL); iput(&inode->vfs_inode); } else { spin_unlock(&block_group->lock); @@ -4435,7 +4432,6 @@ int btrfs_free_block_groups(struct btrfs_fs_info *info) btrfs_remove_free_space_cache(block_group); ASSERT(block_group->cached != BTRFS_CACHE_STARTED); ASSERT(list_empty(&block_group->dirty_list)); - ASSERT(list_empty(&block_group->io_list)); ASSERT(list_empty(&block_group->bg_list)); ASSERT(refcount_read(&block_group->refs) == 1); ASSERT(block_group->swap_extents == 0); diff --git a/fs/btrfs/block-group.h b/fs/btrfs/block-group.h index d69432b236ec02..f7346a0170fc46 100644 --- a/fs/btrfs/block-group.h +++ b/fs/btrfs/block-group.h @@ -228,9 +228,6 @@ struct btrfs_block_group { /* For dirty block groups */ struct list_head dirty_list; - struct list_head io_list; - - struct btrfs_io_ctl io_ctl; /* * Incremented when doing extent allocations and holding a read lock diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 46c62f980c24ac..389eadc5c28f22 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -592,10 +592,6 @@ int btrfs_wait_on_delayed_iputs(struct btrfs_fs_info *fs_info); int btrfs_prealloc_file_range(struct inode *inode, int mode, u64 start, u64 num_bytes, u64 min_size, loff_t actual_len, u64 *alloc_hint); -int btrfs_prealloc_file_range_trans(struct inode *inode, - struct btrfs_trans_handle *trans, int mode, - u64 start, u64 num_bytes, u64 min_size, - loff_t actual_len, u64 *alloc_hint); int btrfs_run_delalloc_range(struct btrfs_inode *inode, struct folio *locked_folio, u64 start, u64 end, struct writeback_control *wbc); void btrfs_queue_writepage_fixup(struct btrfs_inode *inode, struct folio *folio); diff --git a/fs/btrfs/free-space-cache.c b/fs/btrfs/free-space-cache.c index e2af75a205ea53..336b546b0a9467 100644 --- a/fs/btrfs/free-space-cache.c +++ b/fs/btrfs/free-space-cache.c @@ -164,78 +164,6 @@ struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, return inode; } -static int __create_free_space_inode(struct btrfs_root *root, - struct btrfs_trans_handle *trans, - struct btrfs_path *path, - u64 ino, u64 offset) -{ - struct btrfs_key key; - struct btrfs_disk_key disk_key; - struct btrfs_free_space_header *header; - struct btrfs_inode_item *inode_item; - struct extent_buffer *leaf; - /* We inline CRCs for the free disk space cache */ - const u64 flags = BTRFS_INODE_NOCOMPRESS | BTRFS_INODE_PREALLOC | - BTRFS_INODE_NODATASUM | BTRFS_INODE_NODATACOW; - int ret; - - ret = btrfs_insert_empty_inode(trans, root, path, ino); - if (ret) - return ret; - - leaf = path->nodes[0]; - inode_item = btrfs_item_ptr(leaf, path->slots[0], - struct btrfs_inode_item); - btrfs_item_key(leaf, &disk_key, path->slots[0]); - memzero_extent_buffer(leaf, (unsigned long)inode_item, - sizeof(*inode_item)); - btrfs_set_inode_generation(leaf, inode_item, trans->transid); - btrfs_set_inode_size(leaf, inode_item, 0); - btrfs_set_inode_nbytes(leaf, inode_item, 0); - btrfs_set_inode_uid(leaf, inode_item, 0); - btrfs_set_inode_gid(leaf, inode_item, 0); - btrfs_set_inode_mode(leaf, inode_item, S_IFREG | 0600); - btrfs_set_inode_flags(leaf, inode_item, flags); - btrfs_set_inode_nlink(leaf, inode_item, 1); - btrfs_set_inode_transid(leaf, inode_item, trans->transid); - btrfs_set_inode_block_group(leaf, inode_item, offset); - btrfs_release_path(path); - - key.objectid = BTRFS_FREE_SPACE_OBJECTID; - key.type = 0; - key.offset = offset; - ret = btrfs_insert_empty_item(trans, root, path, &key, - sizeof(struct btrfs_free_space_header)); - if (ret < 0) { - btrfs_release_path(path); - return ret; - } - - leaf = path->nodes[0]; - header = btrfs_item_ptr(leaf, path->slots[0], - struct btrfs_free_space_header); - memzero_extent_buffer(leaf, (unsigned long)header, sizeof(*header)); - btrfs_set_free_space_key(leaf, header, &disk_key); - btrfs_release_path(path); - - return 0; -} - -int create_free_space_inode(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path) -{ - int ret; - u64 ino; - - ret = btrfs_get_free_objectid(trans->fs_info->tree_root, &ino); - if (ret < 0) - return ret; - - return __create_free_space_inode(trans->fs_info->tree_root, trans, path, - ino, block_group->start); -} - /* * inode is an optional sink: if it is NULL, btrfs_remove_free_space_inode * handles lookup, otherwise it takes ownership and iputs the inode. @@ -292,7 +220,6 @@ int btrfs_remove_free_space_inode(struct btrfs_trans_handle *trans, } int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, struct inode *vfs_inode) { struct btrfs_truncate_control control = { @@ -306,33 +233,6 @@ int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, struct btrfs_root *root = inode->root; struct extent_state *cached_state = NULL; int ret = 0; - bool locked = false; - - if (block_group) { - BTRFS_PATH_AUTO_FREE(path); - - path = btrfs_alloc_path(); - if (!path) { - ret = -ENOMEM; - goto fail; - } - locked = true; - mutex_lock(&trans->transaction->cache_write_mutex); - if (!list_empty(&block_group->io_list)) { - list_del_init(&block_group->io_list); - - btrfs_wait_cache_io(trans, block_group, path); - btrfs_put_block_group(block_group); - } - - /* - * now that we've truncated the cache away, its no longer - * setup or written - */ - spin_lock(&block_group->lock); - block_group->disk_cache_state = BTRFS_DC_CLEAR; - spin_unlock(&block_group->lock); - } btrfs_i_size_write(inode, 0); truncate_pagecache(vfs_inode, 0); @@ -356,8 +256,6 @@ int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, ret = btrfs_update_inode(trans, inode); fail: - if (locked) - mutex_unlock(&trans->transaction->cache_write_mutex); if (ret) btrfs_abort_transaction(trans, ret); @@ -490,21 +388,6 @@ static int io_ctl_prepare_pages(struct btrfs_io_ctl *io_ctl, bool uptodate) return 0; } -static void io_ctl_set_generation(struct btrfs_io_ctl *io_ctl, u64 generation) -{ - io_ctl_map_page(io_ctl, 1); - - /* - * Skip the csum areas. If we don't check crcs then we just have a - * 64bit chunk at the front of the first page. - */ - io_ctl->cur += (sizeof(u32) * io_ctl->num_pages); - io_ctl->size -= sizeof(u64) + (sizeof(u32) * io_ctl->num_pages); - - put_unaligned_le64(generation, io_ctl->cur); - io_ctl->cur += sizeof(u64); -} - static int io_ctl_check_generation(struct btrfs_io_ctl *io_ctl, u64 generation) { u64 cache_gen; @@ -528,23 +411,6 @@ static int io_ctl_check_generation(struct btrfs_io_ctl *io_ctl, u64 generation) return 0; } -static void io_ctl_set_crc(struct btrfs_io_ctl *io_ctl, int index) -{ - u32 *tmp; - u32 crc = ~(u32)0; - unsigned offset = 0; - - if (index == 0) - offset = sizeof(u32) * io_ctl->num_pages; - - crc = crc32c(crc, io_ctl->orig + offset, PAGE_SIZE - offset); - btrfs_crc32c_final(crc, (u8 *)&crc); - io_ctl_unmap_page(io_ctl); - tmp = page_address(io_ctl->pages[0]); - tmp += index; - *tmp = crc; -} - static int io_ctl_check_crc(struct btrfs_io_ctl *io_ctl, int index) { u32 *tmp, val; @@ -574,76 +440,6 @@ static int io_ctl_check_crc(struct btrfs_io_ctl *io_ctl, int index) return 0; } -static int io_ctl_add_entry(struct btrfs_io_ctl *io_ctl, u64 offset, u64 bytes, - void *bitmap) -{ - struct btrfs_free_space_entry *entry; - - if (!io_ctl->cur) - return -ENOSPC; - - entry = io_ctl->cur; - put_unaligned_le64(offset, &entry->offset); - put_unaligned_le64(bytes, &entry->bytes); - entry->type = (bitmap) ? BTRFS_FREE_SPACE_BITMAP : - BTRFS_FREE_SPACE_EXTENT; - io_ctl->cur += sizeof(struct btrfs_free_space_entry); - io_ctl->size -= sizeof(struct btrfs_free_space_entry); - - if (io_ctl->size >= sizeof(struct btrfs_free_space_entry)) - return 0; - - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - - /* No more pages to map */ - if (io_ctl->index >= io_ctl->num_pages) - return 0; - - /* map the next page */ - io_ctl_map_page(io_ctl, 1); - return 0; -} - -static int io_ctl_add_bitmap(struct btrfs_io_ctl *io_ctl, void *bitmap) -{ - if (!io_ctl->cur) - return -ENOSPC; - - /* - * If we aren't at the start of the current page, unmap this one and - * map the next one if there is any left. - */ - if (io_ctl->cur != io_ctl->orig) { - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - if (io_ctl->index >= io_ctl->num_pages) - return -ENOSPC; - io_ctl_map_page(io_ctl, 0); - } - - copy_page(io_ctl->cur, bitmap); - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - if (io_ctl->index < io_ctl->num_pages) - io_ctl_map_page(io_ctl, 0); - return 0; -} - -static void io_ctl_zero_remaining_pages(struct btrfs_io_ctl *io_ctl) -{ - /* - * If we're not on the boundary we know we've modified the page and we - * need to crc the page. - */ - if (io_ctl->cur != io_ctl->orig) - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - else - io_ctl_unmap_page(io_ctl); - - while (io_ctl->index < io_ctl->num_pages) { - io_ctl_map_page(io_ctl, 1); - io_ctl_set_crc(io_ctl, io_ctl->index - 1); - } -} - static int io_ctl_read_entry(struct btrfs_io_ctl *io_ctl, struct btrfs_free_space *entry, u8 *type) { @@ -1065,490 +861,6 @@ int load_free_space_cache(struct btrfs_block_group *block_group) return ret; } -static noinline_for_stack -int write_cache_extent_entries(struct btrfs_io_ctl *io_ctl, - struct btrfs_block_group *block_group, - int *entries, int *bitmaps, - struct list_head *bitmap_list) -{ - int ret; - struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl; - struct btrfs_free_cluster *cluster = NULL; - struct btrfs_free_cluster *cluster_locked = NULL; - struct rb_node *node = rb_first(&ctl->free_space_offset); - struct btrfs_trim_range *trim_entry; - - /* Get the cluster for this block_group if it exists */ - if (!list_empty(&block_group->cluster_list)) { - cluster = list_first_entry(&block_group->cluster_list, - struct btrfs_free_cluster, block_group_list); - } - - if (!node && cluster) { - cluster_locked = cluster; - spin_lock(&cluster_locked->lock); - node = rb_first(&cluster->root); - cluster = NULL; - } - - /* Write out the extent entries */ - while (node) { - struct btrfs_free_space *e; - - e = rb_entry(node, struct btrfs_free_space, offset_index); - *entries += 1; - - ret = io_ctl_add_entry(io_ctl, e->offset, e->bytes, - e->bitmap); - if (ret) - goto fail; - - if (e->bitmap) { - list_add_tail(&e->list, bitmap_list); - *bitmaps += 1; - } - node = rb_next(node); - if (!node && cluster) { - node = rb_first(&cluster->root); - cluster_locked = cluster; - spin_lock(&cluster_locked->lock); - cluster = NULL; - } - } - if (cluster_locked) { - spin_unlock(&cluster_locked->lock); - cluster_locked = NULL; - } - - /* - * Make sure we don't miss any range that was removed from our rbtree - * because trimming is running. Otherwise after a umount+mount (or crash - * after committing the transaction) we would leak free space and get - * an inconsistent free space cache report from fsck. - */ - list_for_each_entry(trim_entry, &ctl->trimming_ranges, list) { - ret = io_ctl_add_entry(io_ctl, trim_entry->start, - trim_entry->bytes, NULL); - if (ret) - goto fail; - *entries += 1; - } - - return 0; -fail: - if (cluster_locked) - spin_unlock(&cluster_locked->lock); - return -ENOSPC; -} - -static noinline_for_stack int -update_cache_item(struct btrfs_trans_handle *trans, - struct btrfs_root *root, - struct inode *inode, - struct btrfs_path *path, u64 offset, - int entries, int bitmaps) -{ - struct btrfs_key key; - struct btrfs_free_space_header *header; - struct extent_buffer *leaf; - int ret; - - key.objectid = BTRFS_FREE_SPACE_OBJECTID; - key.type = 0; - key.offset = offset; - - ret = btrfs_search_slot(trans, root, &key, path, 0, 1); - if (ret < 0) { - btrfs_clear_extent_bit(&BTRFS_I(inode)->io_tree, 0, inode->i_size - 1, - EXTENT_DELALLOC, NULL); - return ret; - } - leaf = path->nodes[0]; - if (ret > 0) { - struct btrfs_key found_key; - ASSERT(path->slots[0]); - path->slots[0]--; - btrfs_item_key_to_cpu(leaf, &found_key, path->slots[0]); - if (found_key.objectid != BTRFS_FREE_SPACE_OBJECTID || - found_key.offset != offset) { - btrfs_clear_extent_bit(&BTRFS_I(inode)->io_tree, 0, - inode->i_size - 1, EXTENT_DELALLOC, - NULL); - btrfs_release_path(path); - return -ENOENT; - } - } - - BTRFS_I(inode)->generation = trans->transid; - header = btrfs_item_ptr(leaf, path->slots[0], - struct btrfs_free_space_header); - btrfs_set_free_space_entries(leaf, header, entries); - btrfs_set_free_space_bitmaps(leaf, header, bitmaps); - btrfs_set_free_space_generation(leaf, header, trans->transid); - btrfs_release_path(path); - - return 0; -} - -static noinline_for_stack int write_pinned_extent_entries( - struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_io_ctl *io_ctl, - int *entries) -{ - u64 start, extent_start, extent_end, len; - const u64 block_group_end = btrfs_block_group_end(block_group); - struct extent_io_tree *unpin = NULL; - int ret; - - /* - * We want to add any pinned extents to our free space cache - * so we don't leak the space - * - * We shouldn't have switched the pinned extents yet so this is the - * right one - */ - unpin = &trans->transaction->pinned_extents; - - start = block_group->start; - - while (start < block_group_end) { - if (!btrfs_find_first_extent_bit(unpin, start, - &extent_start, &extent_end, - EXTENT_DIRTY, NULL)) - return 0; - - /* This pinned extent is out of our range */ - if (extent_start >= block_group_end) - return 0; - - extent_start = max(extent_start, start); - extent_end = min(block_group_end, extent_end + 1); - len = extent_end - extent_start; - - *entries += 1; - ret = io_ctl_add_entry(io_ctl, extent_start, len, NULL); - if (ret) - return -ENOSPC; - - start = extent_end; - } - - return 0; -} - -static noinline_for_stack int -write_bitmap_entries(struct btrfs_io_ctl *io_ctl, struct list_head *bitmap_list) -{ - struct btrfs_free_space *entry, *next; - int ret; - - /* Write out the bitmaps */ - list_for_each_entry_safe(entry, next, bitmap_list, list) { - ret = io_ctl_add_bitmap(io_ctl, entry->bitmap); - if (ret) - return -ENOSPC; - list_del_init(&entry->list); - } - - return 0; -} - -static int flush_dirty_cache(struct inode *inode) -{ - int ret; - - ret = btrfs_wait_ordered_range(BTRFS_I(inode), 0, (u64)-1); - if (ret) - btrfs_clear_extent_bit(&BTRFS_I(inode)->io_tree, 0, inode->i_size - 1, - EXTENT_DELALLOC, NULL); - - return ret; -} - -static void noinline_for_stack -cleanup_bitmap_list(struct list_head *bitmap_list) -{ - struct btrfs_free_space *entry, *next; - - list_for_each_entry_safe(entry, next, bitmap_list, list) - list_del_init(&entry->list); -} - -static void noinline_for_stack -cleanup_write_cache_enospc(struct inode *inode, - struct btrfs_io_ctl *io_ctl, - struct extent_state **cached_state) -{ - io_ctl_drop_pages(io_ctl); - btrfs_unlock_extent(&BTRFS_I(inode)->io_tree, 0, i_size_read(inode) - 1, - cached_state); -} - -static int __btrfs_wait_cache_io(struct btrfs_root *root, - struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_io_ctl *io_ctl, - struct btrfs_path *path, u64 offset) -{ - int ret; - struct inode *inode = io_ctl->inode; - - if (!inode) - return 0; - - /* Flush the dirty pages in the cache file. */ - ret = flush_dirty_cache(inode); - if (ret) - goto out; - - /* Update the cache item to tell everyone this cache file is valid. */ - ret = update_cache_item(trans, root, inode, path, offset, - io_ctl->entries, io_ctl->bitmaps); -out: - if (ret) { - invalidate_inode_pages2(inode->i_mapping); - BTRFS_I(inode)->generation = 0; - if (block_group) - btrfs_debug(root->fs_info, - "failed to write free space cache for block group %llu error %d", - block_group->start, ret); - } - btrfs_update_inode(trans, BTRFS_I(inode)); - - if (block_group) { - /* the dirty list is protected by the dirty_bgs_lock */ - spin_lock(&trans->transaction->dirty_bgs_lock); - - /* the disk_cache_state is protected by the block group lock */ - spin_lock(&block_group->lock); - - /* - * only mark this as written if we didn't get put back on - * the dirty list while waiting for IO. Otherwise our - * cache state won't be right, and we won't get written again - */ - if (!ret && list_empty(&block_group->dirty_list)) - block_group->disk_cache_state = BTRFS_DC_WRITTEN; - else if (ret) - block_group->disk_cache_state = BTRFS_DC_ERROR; - - spin_unlock(&block_group->lock); - spin_unlock(&trans->transaction->dirty_bgs_lock); - io_ctl->inode = NULL; - iput(inode); - } - - return ret; - -} - -int btrfs_wait_cache_io(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path) -{ - return __btrfs_wait_cache_io(block_group->fs_info->tree_root, trans, - block_group, &block_group->io_ctl, - path, block_group->start); -} - -/* - * Write out cached info to an inode. - * - * @inode: freespace inode we are writing out - * @ctl: free space cache we are going to write out - * @block_group: block_group for this cache if it belongs to a block_group - * @io_ctl: holds context for the io - * @trans: the trans handle - * - * This function writes out a free space cache struct to disk for quick recovery - * on mount. This will return 0 if it was successful in writing the cache out, - * or an errno if it was not. - */ -static int __btrfs_write_out_cache(struct inode *inode, - struct btrfs_block_group *block_group, - struct btrfs_trans_handle *trans) -{ - struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl; - struct btrfs_io_ctl *io_ctl = &block_group->io_ctl; - struct extent_state *cached_state = NULL; - LIST_HEAD(bitmap_list); - int entries = 0; - int bitmaps = 0; - int ret; - bool must_iput = false; - int i_size; - - if (!i_size_read(inode)) - return -EIO; - - WARN_ON(io_ctl->pages); - ret = io_ctl_init(io_ctl, inode, 1); - if (ret) - return ret; - - if (block_group->flags & BTRFS_BLOCK_GROUP_DATA) { - down_write(&block_group->data_rwsem); - spin_lock(&block_group->lock); - if (block_group->delalloc_bytes) { - block_group->disk_cache_state = BTRFS_DC_WRITTEN; - spin_unlock(&block_group->lock); - up_write(&block_group->data_rwsem); - BTRFS_I(inode)->generation = 0; - ret = 0; - must_iput = true; - goto out; - } - spin_unlock(&block_group->lock); - } - - /* Lock all pages first so we can lock the extent safely. */ - ret = io_ctl_prepare_pages(io_ctl, false); - if (ret) - goto out_unlock; - - btrfs_lock_extent(&BTRFS_I(inode)->io_tree, 0, i_size_read(inode) - 1, - &cached_state); - - io_ctl_set_generation(io_ctl, trans->transid); - - mutex_lock(&ctl->cache_writeout_mutex); - /* Write out the extent entries in the free space cache */ - spin_lock(&ctl->tree_lock); - ret = write_cache_extent_entries(io_ctl, block_group, &entries, &bitmaps, - &bitmap_list); - if (ret) - goto out_nospc_locked; - - /* - * Some spaces that are freed in the current transaction are pinned, - * they will be added into free space cache after the transaction is - * committed, we shouldn't lose them. - * - * If this changes while we are working we'll get added back to - * the dirty list and redo it. No locking needed - */ - ret = write_pinned_extent_entries(trans, block_group, io_ctl, &entries); - if (ret) - goto out_nospc_locked; - - /* - * At last, we write out all the bitmaps and keep cache_writeout_mutex - * locked while doing it because a concurrent trim can be manipulating - * or freeing the bitmap. - */ - ret = write_bitmap_entries(io_ctl, &bitmap_list); - spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); - if (ret) - goto out_nospc; - - /* Zero out the rest of the pages just to make sure */ - io_ctl_zero_remaining_pages(io_ctl); - - /* Everything is written out, now we dirty the pages in the file. */ - i_size = i_size_read(inode); - for (int i = 0; i < round_up(i_size, PAGE_SIZE) / PAGE_SIZE; i++) { - u64 dirty_start = i * PAGE_SIZE; - u64 dirty_len = min_t(u64, dirty_start + PAGE_SIZE, i_size) - dirty_start; - - ret = btrfs_dirty_folio(BTRFS_I(inode), page_folio(io_ctl->pages[i]), - dirty_start, dirty_len, &cached_state, false); - if (ret < 0) - goto out_nospc; - } - - if (block_group->flags & BTRFS_BLOCK_GROUP_DATA) - up_write(&block_group->data_rwsem); - /* - * Release the pages and unlock the extent, we will flush - * them out later - */ - io_ctl_drop_pages(io_ctl); - io_ctl_free(io_ctl); - - btrfs_unlock_extent(&BTRFS_I(inode)->io_tree, 0, i_size_read(inode) - 1, - &cached_state); - - /* - * at this point the pages are under IO and we're happy, - * The caller is responsible for waiting on them and updating - * the cache and the inode - */ - io_ctl->entries = entries; - io_ctl->bitmaps = bitmaps; - - ret = btrfs_fdatawrite_range(BTRFS_I(inode), 0, (u64)-1); - if (ret) - goto out; - - return 0; - -out_nospc_locked: - cleanup_bitmap_list(&bitmap_list); - spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); - -out_nospc: - cleanup_write_cache_enospc(inode, io_ctl, &cached_state); - -out_unlock: - if (block_group->flags & BTRFS_BLOCK_GROUP_DATA) - up_write(&block_group->data_rwsem); - -out: - io_ctl->inode = NULL; - io_ctl_free(io_ctl); - if (ret) { - invalidate_inode_pages2(inode->i_mapping); - BTRFS_I(inode)->generation = 0; - } - btrfs_update_inode(trans, BTRFS_I(inode)); - if (must_iput) - iput(inode); - return ret; -} - -int btrfs_write_out_cache(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path) -{ - struct btrfs_fs_info *fs_info = trans->fs_info; - struct inode *inode; - int ret = 0; - - spin_lock(&block_group->lock); - if (block_group->disk_cache_state < BTRFS_DC_SETUP) { - spin_unlock(&block_group->lock); - return 0; - } - spin_unlock(&block_group->lock); - - inode = lookup_free_space_inode(block_group, path); - if (IS_ERR(inode)) - return 0; - - ret = __btrfs_write_out_cache(inode, block_group, trans); - if (ret) { - btrfs_debug(fs_info, - "failed to write free space cache for block group %llu error %d", - block_group->start, ret); - spin_lock(&block_group->lock); - block_group->disk_cache_state = BTRFS_DC_ERROR; - spin_unlock(&block_group->lock); - - block_group->io_ctl.inode = NULL; - iput(inode); - } - - /* - * if ret == 0 the caller is expected to call btrfs_wait_cache_io - * to wait for IO and put the inode - */ - - return ret; -} - static inline unsigned long offset_to_bit(u64 bitmap_start, u32 unit, u64 offset) { diff --git a/fs/btrfs/free-space-cache.h b/fs/btrfs/free-space-cache.h index 53fe8e293af1d5..2432f1783f47cd 100644 --- a/fs/btrfs/free-space-cache.h +++ b/fs/btrfs/free-space-cache.h @@ -105,23 +105,13 @@ int __init btrfs_free_space_init(void); void __cold btrfs_free_space_exit(void); struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, struct btrfs_path *path); -int create_free_space_inode(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path); int btrfs_remove_free_space_inode(struct btrfs_trans_handle *trans, struct inode *inode, struct btrfs_block_group *block_group); int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, struct inode *inode); int load_free_space_cache(struct btrfs_block_group *block_group); -int btrfs_wait_cache_io(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path); -int btrfs_write_out_cache(struct btrfs_trans_handle *trans, - struct btrfs_block_group *block_group, - struct btrfs_path *path); void btrfs_init_free_space_ctl(struct btrfs_block_group *block_group, struct btrfs_free_space_ctl *ctl); diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 32c4dbcd3a4c86..8598792f1ae425 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -9388,15 +9388,6 @@ int btrfs_prealloc_file_range(struct inode *inode, int mode, NULL); } -int btrfs_prealloc_file_range_trans(struct inode *inode, - struct btrfs_trans_handle *trans, int mode, - u64 start, u64 num_bytes, u64 min_size, - loff_t actual_len, u64 *alloc_hint) -{ - return __btrfs_prealloc_file_range(inode, mode, start, num_bytes, - min_size, actual_len, alloc_hint, trans); -} - /* * NOTE: in case you are adding MAY_EXEC check for directories: * we are marking them with IOP_FASTPERM_MAY_EXEC, allowing path lookup to diff --git a/fs/btrfs/relocation.c b/fs/btrfs/relocation.c index da54db75e7a9b9..630a7ad8f8e1ac 100644 --- a/fs/btrfs/relocation.c +++ b/fs/btrfs/relocation.c @@ -3357,7 +3357,7 @@ static int delete_block_group_cache(struct btrfs_block_group *block_group, goto out; } - ret = btrfs_truncate_free_space_cache(trans, block_group, inode); + ret = btrfs_truncate_free_space_cache(trans, inode); btrfs_end_transaction(trans); btrfs_btree_balance_dirty(fs_info); From 8edbb950f8abdf20cf27511bcef79b4df5b45fc6 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:01 -0400 Subject: [PATCH 0066/1012] btrfs: rename cache_write_mutex to dirty_bgs_update_mutex The v1 space cache writeout is gone, but the mutex is still needed. It keeps btrfs_remove_block_group() from deleting a block group item while btrfs_start_dirty_block_groups() is updating it outside the commit critical section. Rename it to reflect what it protects, and update the comments around the dirty block group writeout that still refer to the space cache. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 46 +++++++++++++++++++++++------------------- fs/btrfs/transaction.c | 25 +++++++++-------------- fs/btrfs/transaction.h | 8 ++++---- 3 files changed, 39 insertions(+), 40 deletions(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index 03d5779358ca29..228ceea58b10ec 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -1196,7 +1196,11 @@ int btrfs_remove_block_group(struct btrfs_trans_handle *trans, inode = lookup_free_space_inode(block_group, path); - mutex_lock(&trans->transaction->cache_write_mutex); + /* + * Do not delete the block group item while + * btrfs_start_dirty_block_groups() is updating it. + */ + mutex_lock(&trans->transaction->dirty_bgs_update_mutex); spin_lock(&trans->transaction->dirty_bgs_lock); if (!list_empty(&block_group->dirty_list)) { list_del_init(&block_group->dirty_list); @@ -1204,7 +1208,7 @@ int btrfs_remove_block_group(struct btrfs_trans_handle *trans, btrfs_put_block_group(block_group); } spin_unlock(&trans->transaction->dirty_bgs_lock); - mutex_unlock(&trans->transaction->cache_write_mutex); + mutex_unlock(&trans->transaction->dirty_bgs_update_mutex); ret = btrfs_remove_free_space_inode(trans, inode, block_group); if (unlikely(ret)) { @@ -3358,15 +3362,15 @@ static int update_block_group_item(struct btrfs_trans_handle *trans, } /* - * Transaction commit does final block group cache writeback during a critical + * Transaction commit does the final block group item updates during a critical * section where nothing is allowed to change the FS. This is required in - * order for the cache to actually match the block group, but can introduce a + * order for the items to actually match the block groups, but can introduce a * lot of latency into the commit. * - * So, btrfs_start_dirty_block_groups is here to kick off block group cache IO. - * There's a chance we'll have to redo some of it if the block group changes - * again during the commit, but it greatly reduces the commit latency by - * getting rid of the easy block groups while we're still allowing others to + * So, btrfs_start_dirty_block_groups is here to update the block group items + * early. There's a chance we'll have to redo some of it if the block group + * changes again during the commit, but it greatly reduces the commit latency + * by getting rid of the easy block groups while we're still allowing others to * join the commit. */ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) @@ -3400,11 +3404,11 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) } /* - * cache_write_mutex is here only to save us from balance or automatic - * removal of empty block groups deleting this block group while we are - * updating its item + * dirty_bgs_update_mutex is here only to save us from balance or + * automatic removal of empty block groups deleting this block group + * while we are updating its item */ - mutex_lock(&trans->transaction->cache_write_mutex); + mutex_lock(&trans->transaction->dirty_bgs_update_mutex); while (!list_empty(&dirty)) { bool drop_reserve = true; @@ -3446,12 +3450,12 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) if (drop_reserve) btrfs_dec_delayed_refs_rsv_bg_updates(fs_info); /* Avoid blocking other tasks for too long. */ - mutex_unlock(&trans->transaction->cache_write_mutex); + mutex_unlock(&trans->transaction->dirty_bgs_update_mutex); if (ret) goto out; - mutex_lock(&trans->transaction->cache_write_mutex); + mutex_lock(&trans->transaction->dirty_bgs_update_mutex); } - mutex_unlock(&trans->transaction->cache_write_mutex); + mutex_unlock(&trans->transaction->dirty_bgs_update_mutex); /* * Go through delayed refs for all the stuff we've just kicked off @@ -3465,7 +3469,7 @@ int btrfs_start_dirty_block_groups(struct btrfs_trans_handle *trans) list_splice_init(&cur_trans->dirty_bgs, &dirty); /* * dirty_bgs_lock protects us from concurrent block group - * deletes too (not just cache_write_mutex). + * deletes too (not just dirty_bgs_update_mutex). */ if (!list_empty(&dirty)) { spin_unlock(&cur_trans->dirty_bgs_lock); @@ -3561,10 +3565,10 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans, factor = btrfs_bg_type_to_factor(cache->flags); /* - * If this block group has free space cache written out, we need to make - * sure to load it if we are removing space. This is because we need - * the unpinning stage to actually add the space back to the block group, - * otherwise we will leak space. + * Make sure the free space of this block group is loaded if we are + * removing space. This is because we need the unpinning stage to + * actually add the space back to the block group, otherwise we will + * leak space. */ if (!alloc && !btrfs_block_group_done(cache)) btrfs_cache_block_group(cache, true); @@ -3620,7 +3624,7 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans, /* * No longer have used bytes in this block group, queue it for deletion. * We do this after adding the block group to the dirty list to avoid - * races between cleaner kthread and space cache writeout. + * races between the cleaner kthread and the dirty block group writeout. */ if (!alloc && old_val == 0) { if (!btrfs_test_opt(info, DISCARD_ASYNC)) diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index a5097714578b2a..61fe4a889c9b6f 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -380,7 +380,7 @@ static noinline int join_transaction(struct btrfs_fs_info *fs_info, INIT_LIST_HEAD(&cur_trans->switch_commits); INIT_LIST_HEAD(&cur_trans->dirty_bgs); INIT_LIST_HEAD(&cur_trans->dropped_roots); - mutex_init(&cur_trans->cache_write_mutex); + mutex_init(&cur_trans->dirty_bgs_update_mutex); spin_lock_init(&cur_trans->dirty_bgs_lock); INIT_LIST_HEAD(&cur_trans->deleted_bgs); spin_lock_init(&cur_trans->dropped_roots_lock); @@ -2267,18 +2267,16 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans) if (!test_bit(BTRFS_TRANS_DIRTY_BG_RUN, &cur_trans->flags)) { bool run_it = false; - /* this mutex is also taken before trying to set - * block groups readonly. We need to make sure - * that nobody has set a block group readonly - * after a extents from that block group have been - * allocated for cache files. btrfs_set_block_group_ro - * will wait for the transaction to commit if it - * finds BTRFS_TRANS_DIRTY_BG_RUN set. + /* + * This mutex is also taken before trying to set block groups + * readonly. btrfs_inc_block_group_ro() will wait for the + * transaction to commit if it finds BTRFS_TRANS_DIRTY_BG_RUN + * set. * * The BTRFS_TRANS_DIRTY_BG_RUN flag is also used to make sure - * only one process starts all the block group IO. It wouldn't - * hurt to have more than one go through, but there's no - * real advantage to it either. + * only one process starts all the block group item updates. It + * wouldn't hurt to have more than one go through, but there's + * no real advantage to it either. */ mutex_lock(&fs_info->ro_block_group_mutex); if (!test_and_set_bit(BTRFS_TRANS_DIRTY_BG_RUN, @@ -2512,10 +2510,7 @@ int btrfs_commit_transaction(struct btrfs_trans_handle *trans) if (unlikely(ret)) goto unlock_reloc; - /* - * The tasks which save the space cache and inode cache may also - * update ->aborted, check it. - */ + /* Other tasks may also have updated ->aborted, check it. */ if (TRANS_ABORTED(cur_trans)) { ret = cur_trans->aborted; goto unlock_reloc; diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h index 24b9be1833afb1..17d136675d49db 100644 --- a/fs/btrfs/transaction.h +++ b/fs/btrfs/transaction.h @@ -81,11 +81,11 @@ struct btrfs_transaction { struct extent_io_tree pinned_extents; /* - * we need to make sure block group deletion doesn't race with - * free space cache writeout. This mutex keeps them from stomping - * on each other + * We need to make sure block group deletion doesn't race with the + * dirty block group item updates done outside the commit critical + * section. This mutex keeps them from stomping on each other. */ - struct mutex cache_write_mutex; + struct mutex dirty_bgs_update_mutex; spinlock_t dirty_bgs_lock; /* Protected by spin lock fs_info->unused_bgs_lock. */ struct list_head deleted_bgs; From d4676aeeb64faca1a162b1cffa88097e7e599a4b Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:02 -0400 Subject: [PATCH 0067/1012] btrfs: drop the transaction handle from the prealloc helpers The v1 space cache created its inode during the transaction commit, and btrfs_prealloc_file_range_trans() existed so that preallocation could reuse the open handle. It was the only caller passing a transaction, so __btrfs_prealloc_file_range() and insert_prealloc_file_extent() now always start their own. Fold the wrapper into btrfs_prealloc_file_range() and drop the parameter. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/inode.c | 47 ++++++++++------------------------------------- 1 file changed, 10 insertions(+), 37 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 8598792f1ae425..142ce180e7c933 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -9156,14 +9156,13 @@ static int btrfs_symlink(struct mnt_idmap *idmap, struct inode *dir, } static struct btrfs_trans_handle *insert_prealloc_file_extent( - struct btrfs_trans_handle *trans_in, struct btrfs_inode *inode, struct btrfs_key *ins, u64 file_offset) { struct btrfs_file_extent_item stack_fi; struct btrfs_replace_extent_info extent_info; - struct btrfs_trans_handle *trans = trans_in; + struct btrfs_trans_handle *trans; struct btrfs_path *path; u64 start = ins->objectid; u64 len = ins->offset; @@ -9184,15 +9183,6 @@ static struct btrfs_trans_handle *insert_prealloc_file_extent( if (ret < 0) return ERR_PTR(ret); - if (trans) { - ret = insert_reserved_file_extent(trans, inode, - file_offset, &stack_fi, - true, qgroup_released); - if (ret) - goto free_qgroup; - return trans; - } - extent_info.disk_offset = start; extent_info.disk_len = len; extent_info.data_offset = 0; @@ -9232,12 +9222,12 @@ static struct btrfs_trans_handle *insert_prealloc_file_extent( return ERR_PTR(ret); } -static int __btrfs_prealloc_file_range(struct inode *inode, int mode, - u64 start, u64 num_bytes, u64 min_size, - loff_t actual_len, u64 *alloc_hint, - struct btrfs_trans_handle *trans) +int btrfs_prealloc_file_range(struct inode *inode, int mode, + u64 start, u64 num_bytes, u64 min_size, + loff_t actual_len, u64 *alloc_hint) { struct btrfs_fs_info *fs_info = inode_to_fs_info(inode); + struct btrfs_trans_handle *trans; struct extent_map *em; struct btrfs_root *root = BTRFS_I(inode)->root; struct btrfs_key ins; @@ -9247,11 +9237,8 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode, u64 cur_bytes; u64 last_alloc = (u64)-1; int ret = 0; - bool own_trans = true; u64 end = start + num_bytes - 1; - if (trans) - own_trans = false; while (num_bytes > 0) { cur_bytes = min_t(u64, num_bytes, SZ_256M); cur_bytes = max(cur_bytes, min_size); @@ -9277,8 +9264,8 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode, clear_offset += ins.offset; last_alloc = ins.offset; - trans = insert_prealloc_file_extent(trans, BTRFS_I(inode), - &ins, cur_offset); + trans = insert_prealloc_file_extent(BTRFS_I(inode), &ins, + cur_offset); /* * Now that we inserted the prealloc extent we can finally * decrement the number of reservations in the block group. @@ -9350,8 +9337,7 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode, range_start, range_end - range_start); if (ret) { btrfs_abort_transaction(trans, ret); - if (own_trans) - btrfs_end_transaction(trans); + btrfs_end_transaction(trans); break; } @@ -9363,15 +9349,11 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode, if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); - if (own_trans) - btrfs_end_transaction(trans); + btrfs_end_transaction(trans); break; } - if (own_trans) { - btrfs_end_transaction(trans); - trans = NULL; - } + btrfs_end_transaction(trans); } if (clear_offset < end) btrfs_free_reserved_data_space(BTRFS_I(inode), NULL, clear_offset, @@ -9379,15 +9361,6 @@ static int __btrfs_prealloc_file_range(struct inode *inode, int mode, return ret; } -int btrfs_prealloc_file_range(struct inode *inode, int mode, - u64 start, u64 num_bytes, u64 min_size, - loff_t actual_len, u64 *alloc_hint) -{ - return __btrfs_prealloc_file_range(inode, mode, start, num_bytes, - min_size, actual_len, alloc_hint, - NULL); -} - /* * NOTE: in case you are adding MAY_EXEC check for directories: * we are marking them with IOP_FASTPERM_MAY_EXEC, allowing path lookup to From a2908b4cfb851e79624ee4445717c689b2523695 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:03 -0400 Subject: [PATCH 0068/1012] btrfs: remove the v1 space cache load path Nothing writes a v1 space cache any more, and since commit 545e560a5b0f ("btrfs: disable v1 space cache") the mount option can't be enabled to read one either. Remove load_free_space_cache(), its io_ctl helpers and struct btrfs_io_ctl. Drop the gfp constraint on the inode mapping as well, it only covered the cache's page cache allocations. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 18 +- fs/btrfs/free-space-cache.c | 565 ------------------------------------ fs/btrfs/free-space-cache.h | 15 - 3 files changed, 1 insertion(+), 597 deletions(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index 228ceea58b10ec..997ee27c89f7c5 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -904,22 +904,6 @@ static noinline void caching_thread(struct btrfs_work *work) down_read(&fs_info->commit_root_sem); load_block_group_size_class(caching_ctl); - if (btrfs_test_opt(fs_info, SPACE_CACHE)) { - ret = load_free_space_cache(block_group); - if (ret == 1) { - ret = 0; - goto done; - } - - /* - * We failed to load the space cache, set ourselves to - * CACHE_STARTED and carry on. - */ - spin_lock(&block_group->lock); - block_group->cached = BTRFS_CACHE_STARTED; - spin_unlock(&block_group->lock); - wake_up(&caching_ctl->wait); - } /* * If we are in the transaction that populated the free space tree we @@ -933,7 +917,7 @@ static noinline void caching_thread(struct btrfs_work *work) ret = btrfs_load_free_space_tree(caching_ctl); else ret = load_extent_tree_free(caching_ctl); -done: + spin_lock(&block_group->lock); block_group->caching_ctl = NULL; block_group->cached = ret ? BTRFS_CACHE_ERROR : BTRFS_CACHE_FINISHED; diff --git a/fs/btrfs/free-space-cache.c b/fs/btrfs/free-space-cache.c index 336b546b0a9467..a25c4db561b4bd 100644 --- a/fs/btrfs/free-space-cache.c +++ b/fs/btrfs/free-space-cache.c @@ -9,7 +9,6 @@ #include #include #include -#include #include #include #include "extent-tree.h" @@ -23,7 +22,6 @@ #include "space-info.h" #include "block-group.h" #include "discard.h" -#include "subpage.h" #include "inode-item.h" #include "accessors.h" #include "file-item.h" @@ -57,11 +55,6 @@ static void bitmap_clear_bits(struct btrfs_free_space_ctl *ctl, struct btrfs_free_space *info, u64 offset, u64 bytes, bool update_stats); -static void btrfs_crc32c_final(u32 crc, u8 *result) -{ - put_unaligned_le32(~crc, result); -} - static void __btrfs_remove_free_space_cache(struct btrfs_free_space_ctl *ctl) { struct btrfs_free_space *info; @@ -123,10 +116,6 @@ static struct inode *__lookup_free_space_inode(struct btrfs_root *root, if (IS_ERR(inode)) return ERR_CAST(inode); - mapping_set_gfp_mask(inode->vfs_inode.i_mapping, - mapping_gfp_constraint(inode->vfs_inode.i_mapping, - ~(__GFP_FS | __GFP_HIGHMEM))); - return &inode->vfs_inode; } @@ -262,226 +251,6 @@ int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, return ret; } -static void readahead_cache(struct inode *inode) -{ - struct file_ra_state ra; - pgoff_t last_index; - - file_ra_state_init(&ra, inode->i_mapping); - last_index = (i_size_read(inode) - 1) >> PAGE_SHIFT; - - page_cache_sync_readahead(inode->i_mapping, &ra, NULL, 0, last_index); -} - -static int io_ctl_init(struct btrfs_io_ctl *io_ctl, struct inode *inode, - int write) -{ - int num_pages; - - num_pages = DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE); - - /* Make sure we can fit our crcs and generation into the first page */ - if (write && (num_pages * sizeof(u32) + sizeof(u64)) > PAGE_SIZE) - return -ENOSPC; - - memset(io_ctl, 0, sizeof(struct btrfs_io_ctl)); - - io_ctl->pages = kzalloc_objs(struct page *, num_pages, GFP_NOFS); - if (!io_ctl->pages) - return -ENOMEM; - - io_ctl->num_pages = num_pages; - io_ctl->fs_info = inode_to_fs_info(inode); - io_ctl->inode = inode; - - return 0; -} -ALLOW_ERROR_INJECTION(io_ctl_init, ERRNO); - -static void io_ctl_free(struct btrfs_io_ctl *io_ctl) -{ - kfree(io_ctl->pages); - io_ctl->pages = NULL; -} - -static void io_ctl_unmap_page(struct btrfs_io_ctl *io_ctl) -{ - if (io_ctl->cur) { - io_ctl->cur = NULL; - io_ctl->orig = NULL; - } -} - -static void io_ctl_map_page(struct btrfs_io_ctl *io_ctl, int clear) -{ - ASSERT(io_ctl->index < io_ctl->num_pages); - io_ctl->page = io_ctl->pages[io_ctl->index++]; - io_ctl->cur = page_address(io_ctl->page); - io_ctl->orig = io_ctl->cur; - io_ctl->size = PAGE_SIZE; - if (clear) - clear_page(io_ctl->cur); -} - -static void io_ctl_drop_pages(struct btrfs_io_ctl *io_ctl) -{ - int i; - - io_ctl_unmap_page(io_ctl); - - for (i = 0; i < io_ctl->num_pages; i++) { - if (io_ctl->pages[i]) { - unlock_page(io_ctl->pages[i]); - put_page(io_ctl->pages[i]); - } - } -} - -static int io_ctl_prepare_pages(struct btrfs_io_ctl *io_ctl, bool uptodate) -{ - struct folio *folio; - struct inode *inode = io_ctl->inode; - gfp_t mask = btrfs_alloc_write_mask(inode->i_mapping); - int i; - - for (i = 0; i < io_ctl->num_pages; i++) { - int ret; - - folio = __filemap_get_folio(inode->i_mapping, i, - FGP_LOCK | FGP_ACCESSED | FGP_CREAT, - mask); - if (IS_ERR(folio)) { - io_ctl_drop_pages(io_ctl); - return PTR_ERR(folio); - } - - ret = set_folio_extent_mapped(folio); - if (ret < 0) { - folio_unlock(folio); - folio_put(folio); - io_ctl_drop_pages(io_ctl); - return ret; - } - - io_ctl->pages[i] = &folio->page; - if (uptodate && !folio_test_uptodate(folio)) { - btrfs_read_folio(NULL, folio); - folio_lock(folio); - if (folio->mapping != inode->i_mapping) { - btrfs_err(BTRFS_I(inode)->root->fs_info, - "free space cache page truncated"); - io_ctl_drop_pages(io_ctl); - return -EIO; - } - if (!folio_test_uptodate(folio)) { - btrfs_err(BTRFS_I(inode)->root->fs_info, - "error reading free space cache"); - io_ctl_drop_pages(io_ctl); - return -EIO; - } - } - } - - for (i = 0; i < io_ctl->num_pages; i++) - clear_page_dirty_for_io(io_ctl->pages[i]); - - return 0; -} - -static int io_ctl_check_generation(struct btrfs_io_ctl *io_ctl, u64 generation) -{ - u64 cache_gen; - - /* - * Skip the crc area. If we don't check crcs then we just have a 64bit - * chunk at the front of the first page. - */ - io_ctl->cur += sizeof(u32) * io_ctl->num_pages; - io_ctl->size -= sizeof(u64) + (sizeof(u32) * io_ctl->num_pages); - - cache_gen = get_unaligned_le64(io_ctl->cur); - if (cache_gen != generation) { - btrfs_err_rl(io_ctl->fs_info, - "space cache generation (%llu) does not match inode (%llu)", - cache_gen, generation); - io_ctl_unmap_page(io_ctl); - return -EIO; - } - io_ctl->cur += sizeof(u64); - return 0; -} - -static int io_ctl_check_crc(struct btrfs_io_ctl *io_ctl, int index) -{ - u32 *tmp, val; - u32 crc = ~(u32)0; - unsigned offset = 0; - - if (index >= io_ctl->num_pages) - return -EIO; - - if (index == 0) - offset = sizeof(u32) * io_ctl->num_pages; - - tmp = page_address(io_ctl->pages[0]); - tmp += index; - val = *tmp; - - io_ctl_map_page(io_ctl, 0); - crc = crc32c(crc, io_ctl->orig + offset, PAGE_SIZE - offset); - btrfs_crc32c_final(crc, (u8 *)&crc); - if (val != crc) { - btrfs_err_rl(io_ctl->fs_info, - "csum mismatch on free space cache"); - io_ctl_unmap_page(io_ctl); - return -EIO; - } - - return 0; -} - -static int io_ctl_read_entry(struct btrfs_io_ctl *io_ctl, - struct btrfs_free_space *entry, u8 *type) -{ - struct btrfs_free_space_entry *e; - int ret; - - if (!io_ctl->cur) { - ret = io_ctl_check_crc(io_ctl, io_ctl->index); - if (ret) - return ret; - } - - e = io_ctl->cur; - entry->offset = get_unaligned_le64(&e->offset); - entry->bytes = get_unaligned_le64(&e->bytes); - *type = e->type; - io_ctl->cur += sizeof(struct btrfs_free_space_entry); - io_ctl->size -= sizeof(struct btrfs_free_space_entry); - - if (io_ctl->size >= sizeof(struct btrfs_free_space_entry)) - return 0; - - io_ctl_unmap_page(io_ctl); - - return 0; -} - -static int io_ctl_read_bitmap(struct btrfs_io_ctl *io_ctl, - struct btrfs_free_space *entry) -{ - int ret; - - ret = io_ctl_check_crc(io_ctl, io_ctl->index); - if (ret) - return ret; - - copy_page(entry->bitmap, io_ctl->cur); - io_ctl_unmap_page(io_ctl); - - return 0; -} - static void recalculate_thresholds(struct btrfs_free_space_ctl *ctl) { struct btrfs_block_group *block_group = ctl->block_group; @@ -527,340 +296,6 @@ static void recalculate_thresholds(struct btrfs_free_space_ctl *ctl) div_u64(extent_bytes, sizeof(struct btrfs_free_space)); } -static int __load_free_space_cache(struct btrfs_root *root, struct inode *inode, - struct btrfs_free_space_ctl *ctl, - struct btrfs_path *path, u64 offset) -{ - struct btrfs_fs_info *fs_info = root->fs_info; - struct btrfs_free_space_header *header; - struct extent_buffer *leaf; - struct btrfs_io_ctl io_ctl; - struct btrfs_key key; - struct btrfs_free_space *e, *n; - LIST_HEAD(bitmaps); - u64 num_entries; - u64 num_bitmaps; - u64 generation; - u8 type; - int ret = 0; - - /* Nothing in the space cache, goodbye */ - if (!i_size_read(inode)) - return 0; - - key.objectid = BTRFS_FREE_SPACE_OBJECTID; - key.type = 0; - key.offset = offset; - - ret = btrfs_search_slot(NULL, root, &key, path, 0, 0); - if (ret < 0) - return 0; - else if (ret > 0) { - btrfs_release_path(path); - return 0; - } - - ret = -1; - - leaf = path->nodes[0]; - header = btrfs_item_ptr(leaf, path->slots[0], - struct btrfs_free_space_header); - num_entries = btrfs_free_space_entries(leaf, header); - num_bitmaps = btrfs_free_space_bitmaps(leaf, header); - generation = btrfs_free_space_generation(leaf, header); - btrfs_release_path(path); - - if (!BTRFS_I(inode)->generation) { - btrfs_info(fs_info, - "the free space cache file (%llu) is invalid, skip it", - offset); - return 0; - } - - if (BTRFS_I(inode)->generation != generation) { - btrfs_err(fs_info, - "free space inode generation (%llu) did not match free space cache generation (%llu)", - BTRFS_I(inode)->generation, generation); - return 0; - } - - if (!num_entries) - return 0; - - ret = io_ctl_init(&io_ctl, inode, 0); - if (ret) - return ret; - - readahead_cache(inode); - - ret = io_ctl_prepare_pages(&io_ctl, true); - if (ret) - goto out; - - ret = io_ctl_check_crc(&io_ctl, 0); - if (ret) - goto free_cache; - - ret = io_ctl_check_generation(&io_ctl, generation); - if (ret) - goto free_cache; - - while (num_entries) { - e = kmem_cache_zalloc(btrfs_free_space_cachep, - GFP_NOFS); - if (!e) { - ret = -ENOMEM; - goto free_cache; - } - - ret = io_ctl_read_entry(&io_ctl, e, &type); - if (ret) { - kmem_cache_free(btrfs_free_space_cachep, e); - goto free_cache; - } - - if (!e->bytes) { - ret = -1; - kmem_cache_free(btrfs_free_space_cachep, e); - goto free_cache; - } - - if (type == BTRFS_FREE_SPACE_EXTENT) { - spin_lock(&ctl->tree_lock); - ret = link_free_space(ctl, e); - spin_unlock(&ctl->tree_lock); - if (ret) { - btrfs_err(fs_info, - "Duplicate entries in free space cache, dumping"); - kmem_cache_free(btrfs_free_space_cachep, e); - goto free_cache; - } - } else { - ASSERT(num_bitmaps); - num_bitmaps--; - e->bitmap = kmem_cache_zalloc( - btrfs_free_space_bitmap_cachep, GFP_NOFS); - if (!e->bitmap) { - ret = -ENOMEM; - kmem_cache_free( - btrfs_free_space_cachep, e); - goto free_cache; - } - spin_lock(&ctl->tree_lock); - ret = link_free_space(ctl, e); - if (ret) { - spin_unlock(&ctl->tree_lock); - btrfs_err(fs_info, - "Duplicate entries in free space cache, dumping"); - kmem_cache_free(btrfs_free_space_bitmap_cachep, e->bitmap); - kmem_cache_free(btrfs_free_space_cachep, e); - goto free_cache; - } - ctl->total_bitmaps++; - recalculate_thresholds(ctl); - spin_unlock(&ctl->tree_lock); - list_add_tail(&e->list, &bitmaps); - } - - num_entries--; - } - - io_ctl_unmap_page(&io_ctl); - - /* - * We add the bitmaps at the end of the entries in order that - * the bitmap entries are added to the cache. - */ - list_for_each_entry_safe(e, n, &bitmaps, list) { - list_del_init(&e->list); - ret = io_ctl_read_bitmap(&io_ctl, e); - if (ret) - goto free_cache; - } - - io_ctl_drop_pages(&io_ctl); - ret = 1; -out: - io_ctl_free(&io_ctl); - return ret; -free_cache: - io_ctl_drop_pages(&io_ctl); - - spin_lock(&ctl->tree_lock); - __btrfs_remove_free_space_cache(ctl); - spin_unlock(&ctl->tree_lock); - goto out; -} - -static int copy_free_space_cache(struct btrfs_free_space_ctl *ctl) -{ - struct btrfs_free_space *info; - struct rb_node *n; - int ret = 0; - - while (!ret && (n = rb_first(&ctl->free_space_offset)) != NULL) { - info = rb_entry(n, struct btrfs_free_space, offset_index); - if (!info->bitmap) { - const u64 offset = info->offset; - const u64 bytes = info->bytes; - - unlink_free_space(ctl, info, true); - spin_unlock(&ctl->tree_lock); - kmem_cache_free(btrfs_free_space_cachep, info); - ret = btrfs_add_free_space(ctl->block_group, offset, bytes); - spin_lock(&ctl->tree_lock); - } else { - u64 offset = info->offset; - u64 bytes = ctl->block_group->fs_info->sectorsize; - - ret = search_bitmap(ctl, info, &offset, &bytes, false); - if (ret == 0) { - bitmap_clear_bits(ctl, info, offset, bytes, true); - spin_unlock(&ctl->tree_lock); - ret = btrfs_add_free_space(ctl->block_group, offset, - bytes); - spin_lock(&ctl->tree_lock); - } else { - free_bitmap(ctl, info); - ret = 0; - } - } - cond_resched_lock(&ctl->tree_lock); - } - return ret; -} - -static struct lock_class_key btrfs_free_space_inode_key; - -int load_free_space_cache(struct btrfs_block_group *block_group) -{ - struct btrfs_fs_info *fs_info = block_group->fs_info; - struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl; - struct btrfs_free_space_ctl tmp_ctl = {}; - struct inode *inode; - struct btrfs_path *path; - int ret = 0; - bool matched; - u64 used = block_group->used; - - /* - * Because we could potentially discard our loaded free space, we want - * to load everything into a temporary structure first, and then if it's - * valid copy it all into the actual free space ctl. - */ - btrfs_init_free_space_ctl(block_group, &tmp_ctl); - - /* - * If this block group has been marked to be cleared for one reason or - * another then we can't trust the on disk cache, so just return. - */ - spin_lock(&block_group->lock); - if (block_group->disk_cache_state != BTRFS_DC_WRITTEN) { - spin_unlock(&block_group->lock); - return 0; - } - spin_unlock(&block_group->lock); - - path = btrfs_alloc_path(); - if (!path) - return 0; - path->search_commit_root = true; - path->skip_locking = true; - - /* - * We must pass a path with search_commit_root set to btrfs_iget in - * order to avoid a deadlock when allocating extents for the tree root. - * - * When we are COWing an extent buffer from the tree root, when looking - * for a free extent, at extent-tree.c:find_free_extent(), we can find - * block group without its free space cache loaded. When we find one - * we must load its space cache which requires reading its free space - * cache's inode item from the root tree. If this inode item is located - * in the same leaf that we started COWing before, then we end up in - * deadlock on the extent buffer (trying to read lock it when we - * previously write locked it). - * - * It's safe to read the inode item using the commit root because - * block groups, once loaded, stay in memory forever (until they are - * removed) as well as their space caches once loaded. New block groups - * once created get their ->cached field set to BTRFS_CACHE_FINISHED so - * we will never try to read their inode item while the fs is mounted. - */ - inode = lookup_free_space_inode(block_group, path); - if (IS_ERR(inode)) { - btrfs_free_path(path); - return 0; - } - - /* We may have converted the inode and made the cache invalid. */ - spin_lock(&block_group->lock); - if (block_group->disk_cache_state != BTRFS_DC_WRITTEN) { - spin_unlock(&block_group->lock); - btrfs_free_path(path); - goto out; - } - spin_unlock(&block_group->lock); - - /* - * Reinitialize the class of struct inode's mapping->invalidate_lock for - * free space inodes to prevent false positives related to locks for normal - * inodes. - */ - lockdep_set_class(&(&inode->i_data)->invalidate_lock, - &btrfs_free_space_inode_key); - - ret = __load_free_space_cache(fs_info->tree_root, inode, &tmp_ctl, - path, block_group->start); - btrfs_free_path(path); - if (ret <= 0) - goto out; - - matched = (tmp_ctl.free_space == (block_group->length - used - - block_group->bytes_super)); - - if (matched) { - spin_lock(&tmp_ctl.tree_lock); - ret = copy_free_space_cache(&tmp_ctl); - spin_unlock(&tmp_ctl.tree_lock); - /* - * ret == 1 means we successfully loaded the free space cache, - * so we need to re-set it here. - */ - if (ret == 0) - ret = 1; - } else { - /* - * We need to call the _locked variant so we don't try to update - * the discard counters. - */ - spin_lock(&tmp_ctl.tree_lock); - __btrfs_remove_free_space_cache(&tmp_ctl); - spin_unlock(&tmp_ctl.tree_lock); - btrfs_warn(fs_info, - "block group %llu has wrong amount of free space", - block_group->start); - ret = -1; - } -out: - if (ret < 0) { - /* This cache is bogus, make sure it gets cleared */ - spin_lock(&block_group->lock); - block_group->disk_cache_state = BTRFS_DC_CLEAR; - spin_unlock(&block_group->lock); - ret = 0; - - btrfs_warn(fs_info, - "failed to load free space cache for block group %llu, rebuilding it now", - block_group->start); - } - - spin_lock(&ctl->tree_lock); - btrfs_discard_update_discardable(block_group); - spin_unlock(&ctl->tree_lock); - iput(inode); - return ret; -} - static inline unsigned long offset_to_bit(u64 bitmap_start, u32 unit, u64 offset) { diff --git a/fs/btrfs/free-space-cache.h b/fs/btrfs/free-space-cache.h index 2432f1783f47cd..29166cc09b9012 100644 --- a/fs/btrfs/free-space-cache.h +++ b/fs/btrfs/free-space-cache.h @@ -14,7 +14,6 @@ #include "fs.h" struct inode; -struct page; struct btrfs_fs_info; struct btrfs_path; struct btrfs_trans_handle; @@ -88,19 +87,6 @@ struct btrfs_free_space_ctl { struct list_head trimming_ranges; }; -struct btrfs_io_ctl { - void *cur, *orig; - struct page *page; - struct page **pages; - struct btrfs_fs_info *fs_info; - struct inode *inode; - unsigned long size; - int index; - int num_pages; - int entries; - int bitmaps; -}; - int __init btrfs_free_space_init(void); void __cold btrfs_free_space_exit(void); struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, @@ -111,7 +97,6 @@ int btrfs_remove_free_space_inode(struct btrfs_trans_handle *trans, int btrfs_truncate_free_space_cache(struct btrfs_trans_handle *trans, struct inode *inode); -int load_free_space_cache(struct btrfs_block_group *block_group); void btrfs_init_free_space_ctl(struct btrfs_block_group *block_group, struct btrfs_free_space_ctl *ctl); From e24297027dd04324a6b37e88e8117b66fe9e83e9 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:04 -0400 Subject: [PATCH 0069/1012] btrfs: remove btrfs_disk_cache_state With neither the writer nor the loader left, nothing acts on disk_cache_state. Remove it, the need_clear handling when reading block groups, and the enum. While at it, drop the unused cache_generation field from struct btrfs_block_group. lookup_free_space_inode() converted old style space inodes by clearing disk_cache_state so the cache would be rewritten with the new inode flags. Without that it only sets flags on the in-memory inode, which every remaining caller truncates or deletes right after, so drop the conversion too. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/block-group.c | 32 ++------------------------------ fs/btrfs/block-group.h | 10 ---------- fs/btrfs/disk-io.c | 4 ---- fs/btrfs/free-space-cache.c | 8 -------- 4 files changed, 2 insertions(+), 52 deletions(-) diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c index 997ee27c89f7c5..e6080cb47c895c 100644 --- a/fs/btrfs/block-group.c +++ b/fs/btrfs/block-group.c @@ -2459,8 +2459,7 @@ static int check_chunk_block_group_mappings(struct btrfs_fs_info *fs_info) static int read_one_block_group(struct btrfs_fs_info *info, struct btrfs_block_group_item_v2 *bgi, - const struct btrfs_key *key, - bool need_clear) + const struct btrfs_key *key) { struct btrfs_block_group *cache; const bool mixed = btrfs_fs_incompat(info, MIXED_GROUPS); @@ -2486,20 +2485,6 @@ static int read_one_block_group(struct btrfs_fs_info *info, btrfs_set_free_space_tree_thresholds(cache); - if (need_clear) { - /* - * When we mount with old space cache, we need to - * set BTRFS_DC_CLEAR and set dirty flag. - * - * a) Setting 'BTRFS_DC_CLEAR' makes sure that we - * truncate the old free space cache inode and - * setup a new one. - * b) Setting 'dirty flag' makes sure that we flush - * the new space cache info onto disk. - */ - if (btrfs_test_opt(info, SPACE_CACHE)) - cache->disk_cache_state = BTRFS_DC_CLEAR; - } if (!mixed && ((cache->flags & BTRFS_BLOCK_GROUP_METADATA) && (cache->flags & BTRFS_BLOCK_GROUP_DATA))) { btrfs_err(info, @@ -2640,8 +2625,6 @@ int btrfs_read_block_groups(struct btrfs_fs_info *info) struct btrfs_block_group *cache; struct btrfs_space_info *space_info; struct btrfs_key key; - bool need_clear = false; - u64 cache_gen; /* * Either no extent root (with ibadroots rescue option) or we have @@ -2662,13 +2645,6 @@ int btrfs_read_block_groups(struct btrfs_fs_info *info) if (!path) return -ENOMEM; - cache_gen = btrfs_super_cache_generation(info->super_copy); - if (btrfs_test_opt(info, SPACE_CACHE) && - btrfs_super_generation(info->super_copy) != cache_gen) - need_clear = true; - if (btrfs_test_opt(info, CLEAR_CACHE)) - need_clear = true; - while (1) { struct btrfs_block_group_item_v2 bgi; struct extent_buffer *leaf; @@ -2697,7 +2673,7 @@ int btrfs_read_block_groups(struct btrfs_fs_info *info) btrfs_item_key_to_cpu(leaf, &key, slot); btrfs_release_path(path); - ret = read_one_block_group(info, &bgi, &key, need_clear); + ret = read_one_block_group(info, &bgi, &key); if (ret < 0) goto error; key.objectid += key.offset; @@ -3560,10 +3536,6 @@ int btrfs_update_block_group(struct btrfs_trans_handle *trans, spin_lock(&space_info->lock); spin_lock(&cache->lock); - if (btrfs_test_opt(info, SPACE_CACHE) && - cache->disk_cache_state < BTRFS_DC_CLEAR) - cache->disk_cache_state = BTRFS_DC_CLEAR; - old_val = cache->used; if (alloc) { old_val += num_bytes; diff --git a/fs/btrfs/block-group.h b/fs/btrfs/block-group.h index f7346a0170fc46..d567ed822e55da 100644 --- a/fs/btrfs/block-group.h +++ b/fs/btrfs/block-group.h @@ -20,13 +20,6 @@ struct btrfs_fs_info; struct btrfs_inode; struct btrfs_trans_handle; -enum btrfs_disk_cache_state { - BTRFS_DC_WRITTEN, - BTRFS_DC_ERROR, - BTRFS_DC_CLEAR, - BTRFS_DC_SETUP, -}; - enum btrfs_block_group_size_class { /* Unset */ BTRFS_BG_SZ_NONE, @@ -131,7 +124,6 @@ struct btrfs_block_group { u64 delalloc_bytes; u64 bytes_super; u64 flags; - u64 cache_generation; u64 global_root_id; u64 remap_bytes; u32 identity_remap_count; @@ -171,8 +163,6 @@ struct btrfs_block_group { unsigned long full_stripe_len; unsigned long runtime_flags; - enum btrfs_disk_cache_state disk_cache_state; - /* Cache tracking stuff */ enum btrfs_caching_type cached; struct btrfs_caching_control *caching_ctl; diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 2cf9638cd864e4..140577a7f7974f 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -4877,10 +4877,6 @@ void btrfs_cleanup_dirty_bgs(struct btrfs_transaction *cur_trans, dirty_list); list_del_init(&cache->dirty_list); - spin_lock(&cache->lock); - cache->disk_cache_state = BTRFS_DC_ERROR; - spin_unlock(&cache->lock); - spin_unlock(&cur_trans->dirty_bgs_lock); btrfs_put_block_group(cache); btrfs_dec_delayed_refs_rsv_bg_updates(fs_info); diff --git a/fs/btrfs/free-space-cache.c b/fs/btrfs/free-space-cache.c index a25c4db561b4bd..3ba9ed4a39d0da 100644 --- a/fs/btrfs/free-space-cache.c +++ b/fs/btrfs/free-space-cache.c @@ -124,7 +124,6 @@ struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, { struct btrfs_fs_info *fs_info = block_group->fs_info; struct inode *inode = NULL; - u32 flags = BTRFS_INODE_NODATASUM | BTRFS_INODE_NODATACOW; spin_lock(&block_group->lock); if (block_group->inode) @@ -139,13 +138,6 @@ struct inode *lookup_free_space_inode(struct btrfs_block_group *block_group, return inode; spin_lock(&block_group->lock); - if (!((BTRFS_I(inode)->flags & flags) == flags)) { - btrfs_info(fs_info, "Old style space inode found, converting."); - BTRFS_I(inode)->flags |= BTRFS_INODE_NODATASUM | - BTRFS_INODE_NODATACOW; - block_group->disk_cache_state = BTRFS_DC_CLEAR; - } - if (!test_and_set_bit(BLOCK_GROUP_FLAG_IREF, &block_group->runtime_flags)) block_group->inode = BTRFS_I(igrab(inode)); spin_unlock(&block_group->lock); From 6d1a4b5b37e70266e4e4039449da993c9d13e83b Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:05 -0400 Subject: [PATCH 0070/1012] btrfs: remove the SPACE_CACHE mount option flag Nothing sets BTRFS_MOUNT_SPACE_CACHE anymore, so every test of it is false. Remove the flag, the checks rejecting the v1 cache on zoned filesystems and for sector sizes other than the page size, and the deprecation warning. Show a read-only filesystem that still has an old cache as nospace_cache, since that's what's in effect. space_cache and space_cache=v1 keep falling back to no space cache with a warning. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 19 ++----------------- fs/btrfs/fs.h | 1 - fs/btrfs/super.c | 31 ++----------------------------- fs/btrfs/transaction.c | 4 +--- fs/btrfs/zoned.c | 9 --------- 5 files changed, 5 insertions(+), 59 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 140577a7f7974f..2563e06f29d1b1 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -3069,7 +3069,6 @@ static int btrfs_cleanup_fs_roots(struct btrfs_fs_info *fs_info) int btrfs_start_pre_rw_mount(struct btrfs_fs_info *fs_info) { int ret; - const bool cache_opt = btrfs_test_opt(fs_info, SPACE_CACHE); bool rebuild_free_space_tree = false; if (btrfs_test_opt(fs_info, CLEAR_CACHE) && @@ -3164,8 +3163,8 @@ int btrfs_start_pre_rw_mount(struct btrfs_fs_info *fs_info) } } - if (cache_opt != btrfs_free_space_cache_v1_active(fs_info)) { - ret = btrfs_set_free_space_cache_v1_active(fs_info, cache_opt); + if (btrfs_free_space_cache_v1_active(fs_info)) { + ret = btrfs_set_free_space_cache_v1_active(fs_info, false); if (ret) return ret; } @@ -3281,20 +3280,6 @@ int btrfs_check_features(struct btrfs_fs_info *fs_info, bool is_rw_mount) return -EINVAL; } - /* - * Subpage/bs > ps runtime limitation on v1 cache. - * - * V1 space cache still has some hard coded PAGE_SIZE usage, while - * we're already defaulting to v2 cache, no need to bother v1 as it's - * going to be deprecated anyway. - */ - if (fs_info->sectorsize != PAGE_SIZE && btrfs_test_opt(fs_info, SPACE_CACHE)) { - btrfs_warn(fs_info, - "v1 space cache is not supported for page size %lu with sectorsize %u", - PAGE_SIZE, fs_info->sectorsize); - return -EINVAL; - } - /* This can be called by remount, we need to protect the super block. */ spin_lock(&fs_info->super_lock); btrfs_set_super_incompat_flags(disk_super, incompat); diff --git a/fs/btrfs/fs.h b/fs/btrfs/fs.h index 441b315e8a9893..79d0828c51c7d7 100644 --- a/fs/btrfs/fs.h +++ b/fs/btrfs/fs.h @@ -259,7 +259,6 @@ enum { BTRFS_MOUNT_NOSSD = (1ULL << 9), BTRFS_MOUNT_DISCARD_SYNC = (1ULL << 10), BTRFS_MOUNT_FORCE_COMPRESS = (1ULL << 11), - BTRFS_MOUNT_SPACE_CACHE = (1ULL << 12), BTRFS_MOUNT_CLEAR_CACHE = (1ULL << 13), BTRFS_MOUNT_USER_SUBVOL_RM_ALLOWED = (1ULL << 14), BTRFS_MOUNT_ENOSPC_DEBUG = (1ULL << 15), diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c index b44b16970a6233..6ddb7b35216625 100644 --- a/fs/btrfs/super.c +++ b/fs/btrfs/super.c @@ -515,7 +515,6 @@ static int btrfs_parse_param(struct fs_context *fc, struct fs_parameter *param) btrfs_warn(NULL, "v1 space cache is deprecated, falling back to no space cache"); btrfs_set_opt(ctx->mount_opt, NOSPACECACHE); - btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE); btrfs_clear_opt(ctx->mount_opt, FREE_SPACE_TREE); break; case Opt_space_cache_version: @@ -524,11 +523,9 @@ static int btrfs_parse_param(struct fs_context *fc, struct fs_parameter *param) btrfs_warn(NULL, "v1 space cache is deprecated, falling back to no space cache"); btrfs_set_opt(ctx->mount_opt, NOSPACECACHE); - btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE); btrfs_clear_opt(ctx->mount_opt, FREE_SPACE_TREE); break; case Opt_space_cache_v2: - btrfs_clear_opt(ctx->mount_opt, SPACE_CACHE); btrfs_set_opt(ctx->mount_opt, FREE_SPACE_TREE); break; default: @@ -705,13 +702,6 @@ bool btrfs_check_options(const struct btrfs_fs_info *info, if (btrfs_check_mountopts_zoned(info, mount_opt)) ret = false; - if (!test_bit(BTRFS_FS_STATE_REMOUNTING, &info->fs_state)) { - if (btrfs_raw_test_opt(*mount_opt, SPACE_CACHE)) { - btrfs_warn(info, -"space cache v1 is being deprecated and will be removed in a future release, please use -o space_cache=v2"); - } - } - return ret; } @@ -729,14 +719,6 @@ bool btrfs_check_options(const struct btrfs_fs_info *info, */ void btrfs_set_free_space_cache_settings(struct btrfs_fs_info *fs_info) { - if (fs_info->sectorsize != PAGE_SIZE && btrfs_test_opt(fs_info, SPACE_CACHE)) { - btrfs_info(fs_info, - "forcing free space tree for sector size %u with page size %lu", - fs_info->sectorsize, PAGE_SIZE); - btrfs_clear_opt(fs_info->mount_opt, SPACE_CACHE); - btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE); - } - /* * At this point our mount options are populated, so we only mess with * these settings if we don't have any settings already. @@ -751,9 +733,6 @@ void btrfs_set_free_space_cache_settings(struct btrfs_fs_info *fs_info) return; } - if (btrfs_test_opt(fs_info, SPACE_CACHE)) - return; - if (btrfs_test_opt(fs_info, NOSPACECACHE)) return; @@ -1107,9 +1086,7 @@ static int btrfs_show_options(struct seq_file *seq, struct dentry *dentry) seq_puts(seq, ",discard=async"); if (!(info->sb->s_flags & SB_POSIXACL)) seq_puts(seq, ",noacl"); - if (btrfs_free_space_cache_v1_active(info)) - seq_puts(seq, ",space_cache"); - else if (btrfs_fs_compat_ro(info, FREE_SPACE_TREE)) + if (btrfs_fs_compat_ro(info, FREE_SPACE_TREE)) seq_puts(seq, ",space_cache=v2"); else seq_puts(seq, ",nospace_cache"); @@ -1441,7 +1418,6 @@ static void btrfs_emit_options(struct btrfs_fs_info *info, btrfs_info_if_set(info, old, DISCARD_SYNC, "turning on sync discard"); btrfs_info_if_set(info, old, DISCARD_ASYNC, "turning on async discard"); btrfs_info_if_set(info, old, FREE_SPACE_TREE, "enabling free space tree"); - btrfs_info_if_set(info, old, SPACE_CACHE, "enabling disk space caching"); btrfs_info_if_set(info, old, CLEAR_CACHE, "force clearing of disk cache"); btrfs_info_if_set(info, old, AUTO_DEFRAG, "enabling auto defrag"); btrfs_info_if_set(info, old, FRAGMENT_DATA, "fragmenting data"); @@ -1459,7 +1435,6 @@ static void btrfs_emit_options(struct btrfs_fs_info *info, btrfs_info_if_unset(info, old, SSD_SPREAD, "not using spread ssd allocation scheme"); btrfs_info_if_unset(info, old, NOBARRIER, "turning on barriers"); btrfs_info_if_unset(info, old, NOTREELOG, "enabling tree log"); - btrfs_info_if_unset(info, old, SPACE_CACHE, "disabling disk space caching"); btrfs_info_if_unset(info, old, FREE_SPACE_TREE, "disabling free space tree"); btrfs_info_if_unset(info, old, AUTO_DEFRAG, "disabling auto defrag"); btrfs_info_if_unset(info, old, COMPRESS, "use no compression"); @@ -1524,10 +1499,8 @@ static int btrfs_reconfigure(struct fs_context *fc) btrfs_warn(fs_info, "remount supports changing free space tree only from RO to RW"); /* Make sure free space cache options match the state on disk. */ - if (btrfs_fs_compat_ro(fs_info, FREE_SPACE_TREE)) { + if (btrfs_fs_compat_ro(fs_info, FREE_SPACE_TREE)) btrfs_set_opt(fs_info->mount_opt, FREE_SPACE_TREE); - btrfs_clear_opt(fs_info->mount_opt, SPACE_CACHE); - } } ret = 0; diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index 61fe4a889c9b6f..b0e38039eca184 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -1983,9 +1983,7 @@ static void update_super_roots(struct btrfs_fs_info *fs_info) super->root = root_item->bytenr; super->generation = root_item->generation; super->root_level = root_item->level; - if (btrfs_test_opt(fs_info, SPACE_CACHE)) - super->cache_generation = root_item->generation; - else if (test_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags)) + if (test_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags)) super->cache_generation = 0; if (test_bit(BTRFS_FS_UPDATE_UUID_TREE_GEN, &fs_info->flags)) super->uuid_tree_generation = root_item->generation; diff --git a/fs/btrfs/zoned.c b/fs/btrfs/zoned.c index a1ef8caaacdab4..c04a9955f2d39b 100644 --- a/fs/btrfs/zoned.c +++ b/fs/btrfs/zoned.c @@ -804,15 +804,6 @@ int btrfs_check_mountopts_zoned(const struct btrfs_fs_info *info, if (!btrfs_is_zoned(info)) return 0; - /* - * Space cache writing is not COWed. Disable that to avoid write errors - * in sequential zones. - */ - if (btrfs_raw_test_opt(*mount_opt, SPACE_CACHE)) { - btrfs_err(info, "zoned: space cache v1 is not supported"); - return -EINVAL; - } - if (btrfs_raw_test_opt(*mount_opt, NODATACOW)) { btrfs_err(info, "zoned: NODATACOW not supported"); return -EINVAL; From 691151b8b3ce78244e92db9f10533468ec353981 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:06 -0400 Subject: [PATCH 0071/1012] btrfs: replace btrfs_set_free_space_cache_v1_active() with a cleanup helper The only caller passes active = false. Turn it into btrfs_cleanup_free_space_cache_v1() and fold the block group loop into it. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/disk-io.c | 2 +- fs/btrfs/free-space-cache.c | 42 +++++++++++-------------------------- fs/btrfs/free-space-cache.h | 2 +- 3 files changed, 14 insertions(+), 32 deletions(-) diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c index 2563e06f29d1b1..7fd8babd74a61e 100644 --- a/fs/btrfs/disk-io.c +++ b/fs/btrfs/disk-io.c @@ -3164,7 +3164,7 @@ int btrfs_start_pre_rw_mount(struct btrfs_fs_info *fs_info) } if (btrfs_free_space_cache_v1_active(fs_info)) { - ret = btrfs_set_free_space_cache_v1_active(fs_info, false); + ret = btrfs_cleanup_free_space_cache_v1(fs_info); if (ret) return ret; } diff --git a/fs/btrfs/free-space-cache.c b/fs/btrfs/free-space-cache.c index 3ba9ed4a39d0da..fb6ff3db1241d8 100644 --- a/fs/btrfs/free-space-cache.c +++ b/fs/btrfs/free-space-cache.c @@ -2891,47 +2891,29 @@ bool btrfs_free_space_cache_v1_active(struct btrfs_fs_info *fs_info) return btrfs_super_cache_generation(fs_info->super_copy); } -static int cleanup_free_space_cache_v1(struct btrfs_fs_info *fs_info, - struct btrfs_trans_handle *trans) +int btrfs_cleanup_free_space_cache_v1(struct btrfs_fs_info *fs_info) { - struct btrfs_block_group *block_group; + struct btrfs_trans_handle *trans; struct rb_node *node; + int ret; btrfs_info(fs_info, "cleaning free space cache v1"); - node = rb_first_cached(&fs_info->block_group_cache_tree); - while (node) { - int ret; - - block_group = rb_entry(node, struct btrfs_block_group, cache_node); - ret = btrfs_remove_free_space_inode(trans, NULL, block_group); - if (ret) - return ret; - node = rb_next(node); - } - return 0; -} - -int btrfs_set_free_space_cache_v1_active(struct btrfs_fs_info *fs_info, bool active) -{ - struct btrfs_trans_handle *trans; - int ret; - /* - * update_super_roots will appropriately set or unset - * super_copy->cache_generation based on SPACE_CACHE and - * BTRFS_FS_CLEANUP_SPACE_CACHE_V1. For this reason, we need a - * transaction commit whether we are enabling space cache v1 and don't - * have any other work to do, or are disabling it and removing free - * space inodes. + * update_super_roots() zeroes super_copy->cache_generation while + * BTRFS_FS_CLEANUP_SPACE_CACHE_V1 is set, so this needs a commit. */ trans = btrfs_start_transaction(fs_info->tree_root, 0); if (IS_ERR(trans)) return PTR_ERR(trans); - if (!active) { - set_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags); - ret = cleanup_free_space_cache_v1(fs_info, trans); + set_bit(BTRFS_FS_CLEANUP_SPACE_CACHE_V1, &fs_info->flags); + for (node = rb_first_cached(&fs_info->block_group_cache_tree); node; + node = rb_next(node)) { + struct btrfs_block_group *block_group; + + block_group = rb_entry(node, struct btrfs_block_group, cache_node); + ret = btrfs_remove_free_space_inode(trans, NULL, block_group); if (unlikely(ret)) { btrfs_abort_transaction(trans, ret); btrfs_end_transaction(trans); diff --git a/fs/btrfs/free-space-cache.h b/fs/btrfs/free-space-cache.h index 29166cc09b9012..f5f18e397b130e 100644 --- a/fs/btrfs/free-space-cache.h +++ b/fs/btrfs/free-space-cache.h @@ -136,7 +136,7 @@ int btrfs_trim_block_group_bitmaps(struct btrfs_block_group *block_group, void btrfs_trim_fully_remapped_block_group(struct btrfs_block_group *bg); bool btrfs_free_space_cache_v1_active(struct btrfs_fs_info *fs_info); -int btrfs_set_free_space_cache_v1_active(struct btrfs_fs_info *fs_info, bool active); +int btrfs_cleanup_free_space_cache_v1(struct btrfs_fs_info *fs_info); /* Support functions for running our sanity tests */ #ifdef CONFIG_BTRFS_FS_RUN_SANITY_TESTS bool btrfs_use_bitmap(struct btrfs_free_space_ctl *ctl, From dfce6462d321977cb86caa854a6950d2d50f7d7f Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:07 -0400 Subject: [PATCH 0072/1012] btrfs: remove the free space cache trimming ranges cache_writeout_mutex and trimming_ranges let the v1 cache writer see ranges that were unlinked from the free space tree while being discarded. Nothing consumes the list anymore, and the tree itself is protected by tree_lock, so remove them. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/free-space-cache.c | 41 +++---------------------------------- fs/btrfs/free-space-cache.h | 2 -- 2 files changed, 3 insertions(+), 40 deletions(-) diff --git a/fs/btrfs/free-space-cache.c b/fs/btrfs/free-space-cache.c index fb6ff3db1241d8..2a40167c3fb657 100644 --- a/fs/btrfs/free-space-cache.c +++ b/fs/btrfs/free-space-cache.c @@ -36,12 +36,6 @@ static struct kmem_cache *btrfs_free_space_cachep; static struct kmem_cache *btrfs_free_space_bitmap_cachep; -struct btrfs_trim_range { - u64 start; - u64 bytes; - struct list_head list; -}; - static int link_free_space(struct btrfs_free_space_ctl *ctl, struct btrfs_free_space *info); static void unlink_free_space(struct btrfs_free_space_ctl *ctl, @@ -1692,8 +1686,6 @@ void btrfs_init_free_space_ctl(struct btrfs_block_group *block_group, spin_lock_init(&ctl->tree_lock); ctl->block_group = block_group; ctl->free_space_bytes = RB_ROOT_CACHED; - INIT_LIST_HEAD(&ctl->trimming_ranges); - mutex_init(&ctl->cache_writeout_mutex); /* * we only want to have 32k of ram per block group for keeping @@ -2389,12 +2381,10 @@ void btrfs_init_free_cluster(struct btrfs_free_cluster *cluster) static int do_trimming(struct btrfs_block_group *block_group, u64 *total_trimmed, u64 start, u64 bytes, u64 reserved_start, u64 reserved_bytes, - enum btrfs_trim_state reserved_trim_state, - struct btrfs_trim_range *trim_entry) + enum btrfs_trim_state reserved_trim_state) { struct btrfs_space_info *space_info = block_group->space_info; struct btrfs_fs_info *fs_info = block_group->fs_info; - struct btrfs_free_space_ctl *ctl = block_group->free_space_ctl; int ret; bool bg_ro; const u64 end = start + bytes; @@ -2420,7 +2410,6 @@ static int do_trimming(struct btrfs_block_group *block_group, trim_state = BTRFS_TRIM_STATE_TRIMMED; } - mutex_lock(&ctl->cache_writeout_mutex); if (reserved_start < start) __btrfs_add_free_space(block_group, reserved_start, start - reserved_start, @@ -2429,8 +2418,6 @@ static int do_trimming(struct btrfs_block_group *block_group, __btrfs_add_free_space(block_group, end, reserved_end - end, reserved_trim_state); __btrfs_add_free_space(block_group, start, bytes, trim_state); - list_del(&trim_entry->list); - mutex_unlock(&ctl->cache_writeout_mutex); if (!bg_ro) { spin_lock(&space_info->lock); @@ -2468,9 +2455,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, const u64 max_discard_size = READ_ONCE(discard_ctl->max_discard_size); while (start < end) { - struct btrfs_trim_range trim_entry; - - mutex_lock(&ctl->cache_writeout_mutex); spin_lock(&ctl->tree_lock); if (ctl->free_space < minlen) @@ -2501,7 +2485,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, bytes = entry->bytes; if (bytes < minlen) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); goto next; } unlink_free_space(ctl, entry, true); @@ -2526,7 +2509,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, bytes = min(extent_start + extent_bytes, end) - start; if (bytes < minlen) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); goto next; } @@ -2535,14 +2517,9 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, } spin_unlock(&ctl->tree_lock); - trim_entry.start = extent_start; - trim_entry.bytes = extent_bytes; - list_add_tail(&trim_entry.list, &ctl->trimming_ranges); - mutex_unlock(&ctl->cache_writeout_mutex); ret = do_trimming(block_group, total_trimmed, start, bytes, - extent_start, extent_bytes, extent_trim_state, - &trim_entry); + extent_start, extent_bytes, extent_trim_state); if (ret) { block_group->discard_cursor = start + bytes; break; @@ -2566,7 +2543,6 @@ static int trim_no_bitmap(struct btrfs_block_group *block_group, out_unlock: block_group->discard_cursor = btrfs_block_group_end(block_group); spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); return ret; } @@ -2677,16 +2653,13 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, while (offset < end) { bool next_bitmap = false; - struct btrfs_trim_range trim_entry; - mutex_lock(&ctl->cache_writeout_mutex); spin_lock(&ctl->tree_lock); if (ctl->free_space < minlen) { block_group->discard_cursor = btrfs_block_group_end(block_group); spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); break; } @@ -2702,7 +2675,6 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, if (!entry || (async && minlen && start == offset && btrfs_free_space_trimmed(entry))) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); next_bitmap = true; goto next; } @@ -2728,7 +2700,6 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, else entry->trim_state = BTRFS_TRIM_STATE_UNTRIMMED; spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); next_bitmap = true; goto next; } @@ -2739,14 +2710,12 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, */ if (async && *total_trimmed) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); return ret; } bytes = min(bytes, end - start); if (bytes < minlen || (async && maxlen && bytes > maxlen)) { spin_unlock(&ctl->tree_lock); - mutex_unlock(&ctl->cache_writeout_mutex); goto next; } @@ -2766,13 +2735,9 @@ static int trim_bitmaps(struct btrfs_block_group *block_group, free_bitmap(ctl, entry); spin_unlock(&ctl->tree_lock); - trim_entry.start = start; - trim_entry.bytes = bytes; - list_add_tail(&trim_entry.list, &ctl->trimming_ranges); - mutex_unlock(&ctl->cache_writeout_mutex); ret = do_trimming(block_group, total_trimmed, start, bytes, - start, bytes, 0, &trim_entry); + start, bytes, 0); if (ret) { reset_trimming_bitmap(ctl, offset); block_group->discard_cursor = diff --git a/fs/btrfs/free-space-cache.h b/fs/btrfs/free-space-cache.h index f5f18e397b130e..e22443598b8efe 100644 --- a/fs/btrfs/free-space-cache.h +++ b/fs/btrfs/free-space-cache.h @@ -83,8 +83,6 @@ struct btrfs_free_space_ctl { s32 discardable_extents[BTRFS_STAT_NR_ENTRIES]; s64 discardable_bytes[BTRFS_STAT_NR_ENTRIES]; struct btrfs_block_group *block_group; - struct mutex cache_writeout_mutex; - struct list_head trimming_ranges; }; int __init btrfs_free_space_init(void); From 2dc5760ee98c86cdde9fbeaa26cf8df8594682a2 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:08 -0400 Subject: [PATCH 0073/1012] btrfs: remove BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE Free space inodes no longer reserve data or delalloc space, as nothing writes to them. Remove the flush mode and the special cases that selected it. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/delalloc-space.c | 13 ++----------- fs/btrfs/space-info.c | 2 -- fs/btrfs/space-info.h | 4 ---- 3 files changed, 2 insertions(+), 17 deletions(-) diff --git a/fs/btrfs/delalloc-space.c b/fs/btrfs/delalloc-space.c index d357ed7efd99bd..77781852e41782 100644 --- a/fs/btrfs/delalloc-space.c +++ b/fs/btrfs/delalloc-space.c @@ -132,9 +132,7 @@ int btrfs_alloc_data_chunk_ondemand(const struct btrfs_inode *inode, u64 bytes) /* Make sure bytes are sectorsize aligned */ bytes = ALIGN(bytes, fs_info->sectorsize); - if (btrfs_is_free_space_inode(inode)) - flush = BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE; - else if (btrfs_is_zoned(fs_info) && btrfs_is_data_reloc_root(root)) + if (btrfs_is_zoned(fs_info) && btrfs_is_data_reloc_root(root)) flush = BTRFS_RESERVE_FLUSH_ZONED_RELOCATION; return btrfs_reserve_data_bytes(data_sinfo_for_inode(inode), bytes, flush); @@ -155,8 +153,6 @@ int btrfs_check_data_free_space(struct btrfs_inode *inode, if (noflush) flush = BTRFS_RESERVE_NO_FLUSH; - else if (btrfs_is_free_space_inode(inode)) - flush = BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE; ret = btrfs_reserve_data_bytes(data_sinfo_for_inode(inode), len, flush); if (ret < 0) @@ -326,15 +322,10 @@ int btrfs_delalloc_reserve_metadata(struct btrfs_inode *inode, u64 num_bytes, int ret = 0; /* - * If we are a free space inode we need to not flush since we will be in - * the middle of a transaction commit. We also don't need the delalloc - * mutex since we won't race with anybody. We need this mostly to make - * lockdep shut its filthy mouth. - * * If we have a transaction open (can happen if we call truncate_block * from truncate), then we need FLUSH_LIMIT so we don't deadlock. */ - if (noflush || btrfs_is_free_space_inode(inode)) { + if (noflush) { flush = BTRFS_RESERVE_NO_FLUSH; } else { if (current->journal_info) diff --git a/fs/btrfs/space-info.c b/fs/btrfs/space-info.c index 39a28e1bec8ad8..01018152c054b0 100644 --- a/fs/btrfs/space-info.c +++ b/fs/btrfs/space-info.c @@ -1704,7 +1704,6 @@ static int handle_reserve_ticket(struct btrfs_space_info *space_info, evict_flush_states, ARRAY_SIZE(evict_flush_states)); break; - case BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE: case BTRFS_RESERVE_FLUSH_ZONED_RELOCATION: priority_reclaim_data_space(space_info, ticket); break; @@ -1968,7 +1967,6 @@ int btrfs_reserve_data_bytes(struct btrfs_space_info *space_info, u64 bytes, int ret; ASSERT(flush == BTRFS_RESERVE_FLUSH_DATA || - flush == BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE || flush == BTRFS_RESERVE_FLUSH_ZONED_RELOCATION || flush == BTRFS_RESERVE_NO_FLUSH, "flush=%d", flush); ASSERT(!current->journal_info || flush != BTRFS_RESERVE_FLUSH_DATA, diff --git a/fs/btrfs/space-info.h b/fs/btrfs/space-info.h index aa836e8a9d4a6f..d0130c8ba3ddaf 100644 --- a/fs/btrfs/space-info.h +++ b/fs/btrfs/space-info.h @@ -66,7 +66,6 @@ enum btrfs_reserve_flush_enum { * Can be interrupted by a fatal signal. */ BTRFS_RESERVE_FLUSH_DATA, - BTRFS_RESERVE_FLUSH_FREE_SPACE_INODE, BTRFS_RESERVE_FLUSH_ALL, /* @@ -82,9 +81,6 @@ enum btrfs_reserve_flush_enum { * priority flushing for this, because otherwise we can deadlock on * waiting for a ticket, that cannot be granted, because we cannot do * any allocations. - * - * Apart from being specific to zoned relocation, it is equal to - * BTRFS_FLUSH_FREE_SPACE_INODE. */ BTRFS_RESERVE_FLUSH_ZONED_RELOCATION, From 869ea0cc13a7c7dafa7af71718e9ec9161f470dc Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:09 -0400 Subject: [PATCH 0074/1012] btrfs: remove the free space inode ordered extent special cases Free space inodes never have ordered extents anymore. Drop the lockdep exceptions for them and btrfs_join_transaction_spacecache(), which was only used to finish their ordered extents during a commit. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/inode.c | 20 +++----------------- fs/btrfs/ordered-data.c | 20 ++------------------ fs/btrfs/transaction.c | 6 ------ fs/btrfs/transaction.h | 1 - 4 files changed, 5 insertions(+), 42 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 142ce180e7c933..c8c6f7bc1818f2 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -3221,7 +3221,6 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) int compress_type = 0; int ret = 0; u64 logical_len = ordered_extent->num_bytes; - bool freespace_inode; bool truncated = false; bool clear_reserved_extent = true; unsigned int clear_bits = 0; @@ -3238,9 +3237,7 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) if (!test_bit(BTRFS_ORDERED_NOCOW, &ordered_extent->flags)) clear_bits |= EXTENT_DEFRAG; - freespace_inode = btrfs_is_free_space_inode(inode); - if (!freespace_inode) - btrfs_lockdep_acquire(fs_info, btrfs_ordered_extent); + btrfs_lockdep_acquire(fs_info, btrfs_ordered_extent); if (unlikely(test_bit(BTRFS_ORDERED_IOERR, &ordered_extent->flags))) { ret = -EIO; @@ -3275,10 +3272,7 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) &cached_state); } - if (freespace_inode) - trans = btrfs_join_transaction_spacecache(root); - else - trans = btrfs_join_transaction(root); + trans = btrfs_join_transaction(root); if (IS_ERR(trans)) { ret = PTR_ERR(trans); trans = NULL; @@ -8135,7 +8129,6 @@ void btrfs_destroy_inode(struct inode *vfs_inode) struct btrfs_ordered_extent *ordered; struct btrfs_inode *inode = BTRFS_I(vfs_inode); struct btrfs_root *root = inode->root; - bool freespace_inode; WARN_ON(!hlist_empty(&vfs_inode->i_dentry)); WARN_ON(vfs_inode->i_data.nrpages); @@ -8158,12 +8151,6 @@ void btrfs_destroy_inode(struct inode *vfs_inode) if (!root) return; - /* - * If this is a free space inode do not take the ordered extents lockdep - * map. - */ - freespace_inode = btrfs_is_free_space_inode(inode); - while (1) { ordered = btrfs_lookup_first_ordered_extent(inode, (u64)-1); if (!ordered) @@ -8173,8 +8160,7 @@ void btrfs_destroy_inode(struct inode *vfs_inode) "found ordered extent %llu %llu on inode cleanup", ordered->file_offset, ordered->num_bytes); - if (!freespace_inode) - btrfs_lockdep_acquire(root->fs_info, btrfs_ordered_extent); + btrfs_lockdep_acquire(root->fs_info, btrfs_ordered_extent); btrfs_remove_ordered_extent(ordered); btrfs_put_ordered_extent(ordered); diff --git a/fs/btrfs/ordered-data.c b/fs/btrfs/ordered-data.c index e9f1cbeb555a4c..df74c75d6c2991 100644 --- a/fs/btrfs/ordered-data.c +++ b/fs/btrfs/ordered-data.c @@ -654,13 +654,6 @@ void btrfs_remove_ordered_extent(struct btrfs_ordered_extent *entry) struct btrfs_fs_info *fs_info = root->fs_info; struct rb_node *node; bool pending; - bool freespace_inode; - - /* - * If this is a free space inode the thread has not acquired the ordered - * extents lockdep map. - */ - freespace_inode = btrfs_is_free_space_inode(btrfs_inode); btrfs_lockdep_acquire(fs_info, btrfs_trans_pending_ordered); /* This is paired with alloc_ordered_extent(). */ @@ -735,8 +728,7 @@ void btrfs_remove_ordered_extent(struct btrfs_ordered_extent *entry) } spin_unlock(&root->ordered_extent_lock); wake_up(&entry->wait); - if (!freespace_inode) - btrfs_lockdep_release(fs_info, btrfs_ordered_extent); + btrfs_lockdep_release(fs_info, btrfs_ordered_extent); } static void btrfs_run_ordered_extent_work(struct btrfs_work *work) @@ -867,16 +859,9 @@ void btrfs_start_ordered_extent_nowriteback(struct btrfs_ordered_extent *entry, u64 start = entry->file_offset; u64 end = start + entry->num_bytes - 1; struct btrfs_inode *inode = entry->inode; - bool freespace_inode; trace_btrfs_ordered_extent_start(inode, entry); - /* - * If this is a free space inode do not take the ordered extents lockdep - * map. - */ - freespace_inode = btrfs_is_free_space_inode(inode); - /* * pages in the range can be dirty, clean or writeback. We * start IO on any dirty ones so the wait doesn't stall waiting @@ -896,8 +881,7 @@ void btrfs_start_ordered_extent_nowriteback(struct btrfs_ordered_extent *entry, } } - if (!freespace_inode) - btrfs_might_wait_for_event(inode->root->fs_info, btrfs_ordered_extent); + btrfs_might_wait_for_event(inode->root->fs_info, btrfs_ordered_extent); wait_event(entry->wait, test_bit(BTRFS_ORDERED_COMPLETE, &entry->flags)); } diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index b0e38039eca184..a875008feb8fce 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -854,12 +854,6 @@ struct btrfs_trans_handle *btrfs_join_transaction(struct btrfs_root *root) true); } -struct btrfs_trans_handle *btrfs_join_transaction_spacecache(struct btrfs_root *root) -{ - return start_transaction(root, 0, TRANS_JOIN_NOLOCK, - BTRFS_RESERVE_NO_FLUSH, true); -} - /* * Similar to regular join but it never starts a transaction when none is * running or when there's a running one at a state >= TRANS_STATE_UNBLOCKED. diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h index 17d136675d49db..70b6c95efe53af 100644 --- a/fs/btrfs/transaction.h +++ b/fs/btrfs/transaction.h @@ -294,7 +294,6 @@ struct btrfs_trans_handle *btrfs_start_transaction_fallback_global_rsv( struct btrfs_root *root, unsigned int num_items); struct btrfs_trans_handle *btrfs_join_transaction(struct btrfs_root *root); -struct btrfs_trans_handle *btrfs_join_transaction_spacecache(struct btrfs_root *root); struct btrfs_trans_handle *btrfs_join_transaction_nostart(struct btrfs_root *root); struct btrfs_trans_handle *btrfs_attach_transaction(struct btrfs_root *root); struct btrfs_trans_handle *btrfs_attach_transaction_barrier( From babc1b1d88126694a815be0cc62b4824dd85a263 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:10 -0400 Subject: [PATCH 0075/1012] btrfs: remove the free space inode special cases from the COW paths Free space inodes are never written anymore, so drop the special cases for them in cow_file_range(), fallback_to_cow(), can_nocow_file_extent() and btrfs_finish_one_ordered(). Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/inode.c | 58 +++++++++++++----------------------------------- 1 file changed, 15 insertions(+), 43 deletions(-) diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index c8c6f7bc1818f2..42557fe028820f 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -1369,11 +1369,6 @@ static noinline int cow_file_range(struct btrfs_inode *inode, goto out_unlock; } - if (btrfs_is_free_space_inode(inode)) { - ret = -EINVAL; - goto out_unlock; - } - num_bytes = ALIGN(end - start + 1, blocksize); num_bytes = max(blocksize, num_bytes); ASSERT(num_bytes <= btrfs_super_total_bytes(fs_info->super_copy)); @@ -1683,7 +1678,6 @@ static int fallback_to_cow(struct btrfs_inode *inode, struct folio *locked_folio, const u64 start, const u64 end) { - const bool is_space_ino = btrfs_is_free_space_inode(inode); const bool is_reloc_ino = btrfs_is_data_reloc_root(inode->root); const u64 range_bytes = end + 1 - start; struct extent_io_tree *io_tree = &inode->io_tree; @@ -1716,23 +1710,22 @@ static int fallback_to_cow(struct btrfs_inode *inode, * extent_clear_unlock_delalloc()) the bytes_may_use counter of the * data space info, which we incremented in the step above. * - * If we need to fallback to cow and the inode corresponds to a free - * space cache inode or an inode of the data relocation tree, we must - * also increment bytes_may_use of the data space_info for the same - * reason. Space caches and relocated data extents always get a prealloc - * extent for them, however scrub or balance may have set the block - * group that contains that extent to RO mode and therefore force COW - * when starting writeback. + * If we need to fallback to cow and the inode is in the data relocation + * tree, we must also increment bytes_may_use of the data space_info for + * the same reason. Relocated data extents always get a prealloc extent, + * however scrub or balance may have set the block group that contains + * that extent to RO mode and therefore force COW when starting + * writeback. */ btrfs_lock_extent(io_tree, start, end, &cached_state); count = btrfs_count_range_bits(io_tree, &range_start, end, range_bytes, EXTENT_NORESERVE, false, NULL); - if (count > 0 || is_space_ino || is_reloc_ino) { + if (count > 0 || is_reloc_ino) { u64 bytes = count; struct btrfs_fs_info *fs_info = inode->root->fs_info; struct btrfs_space_info *sinfo = fs_info->data_sinfo; - if (is_space_ino || is_reloc_ino) + if (is_reloc_ino) bytes = range_bytes; spin_lock(&sinfo->lock); @@ -1797,7 +1790,6 @@ static int can_nocow_file_extent(struct btrfs_path *path, struct btrfs_inode *inode, struct can_nocow_file_extent_args *args) { - const bool is_freespace_inode = btrfs_is_free_space_inode(inode); struct extent_buffer *leaf = path->nodes[0]; struct btrfs_root *root = inode->root; struct btrfs_file_extent_item *fi; @@ -1810,8 +1802,7 @@ static int can_nocow_file_extent(struct btrfs_path *path, bool nowait = path->nowait; /* If there are pending snapshots for this root, we must do COW. */ - if (args->writeback_path && !is_freespace_inode && - atomic_read(&root->snapshot_force_cow)) + if (args->writeback_path && atomic_read(&root->snapshot_force_cow)) goto out; fi = btrfs_item_ptr(leaf, path->slots[0], struct btrfs_file_extent_item); @@ -1860,7 +1851,6 @@ static int can_nocow_file_extent(struct btrfs_path *path, ret = btrfs_cross_ref_exist(inode, key->offset - args->file_extent.offset, args->file_extent.disk_bytenr, path); - WARN_ON_ONCE(ret > 0 && is_freespace_inode); if (ret != 0) goto out; @@ -1895,7 +1885,6 @@ static int can_nocow_file_extent(struct btrfs_path *path, ret = btrfs_lookup_csums_list(csum_root, io_start, io_start + args->file_extent.num_bytes - 1, NULL, nowait); - WARN_ON_ONCE(ret > 0 && is_freespace_inode); if (ret != 0) goto out; @@ -3221,6 +3210,7 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) int compress_type = 0; int ret = 0; u64 logical_len = ordered_extent->num_bytes; + u64 unwritten_start; bool truncated = false; bool clear_reserved_extent = true; unsigned int clear_bits = 0; @@ -3382,29 +3372,11 @@ int btrfs_finish_one_ordered(struct btrfs_ordered_extent *ordered_extent) if (ret) btrfs_mark_ordered_extent_error(ordered_extent); - /* - * Drop extent maps for the part of the extent we didn't write. - * - * We have an exception here for the free_space_inode, this is - * because when we do btrfs_get_extent() on the free space inode - * we will search the commit root. If this is a new block group - * we won't find anything, and we will trip over the assert in - * writepage where we do ASSERT(em->block_start != - * EXTENT_MAP_HOLE). - * - * Theoretically we could also skip this for any NOCOW extent as - * we don't mess with the extent map tree in the NOCOW case, but - * for now simply skip this if we are the free space inode. - */ - if (!btrfs_is_free_space_inode(inode)) { - u64 unwritten_start = start; - - if (truncated) - unwritten_start += logical_len; - - btrfs_drop_extent_map_range(inode, unwritten_start, - end, false); - } + /* Drop extent maps for the part of the extent we didn't write. */ + unwritten_start = start; + if (truncated) + unwritten_start += logical_len; + btrfs_drop_extent_map_range(inode, unwritten_start, end, false); /* * If the ordered extent had an IOERR or something else went From 10a475e4f857b066aa3adeb1f2e05629493000c7 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:11 -0400 Subject: [PATCH 0076/1012] btrfs: stop special-casing free space inodes in the delalloc accounting Free space inodes never have delalloc or outstanding extents any more, so they don't need to be kept off the root's delalloc inode list or out of the outstanding extents tracepoint. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/btrfs_inode.h | 2 -- fs/btrfs/inode.c | 5 ++--- 2 files changed, 2 insertions(+), 5 deletions(-) diff --git a/fs/btrfs/btrfs_inode.h b/fs/btrfs/btrfs_inode.h index 389eadc5c28f22..114a5c38afd374 100644 --- a/fs/btrfs/btrfs_inode.h +++ b/fs/btrfs/btrfs_inode.h @@ -403,8 +403,6 @@ static inline void btrfs_mod_outstanding_extents(struct btrfs_inode *inode, { lockdep_assert_held(&inode->lock); inode->outstanding_extents += mod; - if (btrfs_is_free_space_inode(inode)) - return; trace_btrfs_inode_mod_outstanding_extents(inode->root, btrfs_ino(inode), mod, inode->outstanding_extents); } diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 42557fe028820f..34940147393f73 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -2632,7 +2632,7 @@ void btrfs_set_delalloc_extent(struct btrfs_inode *inode, struct extent_state *s * and are therefore protected against concurrent calls of this * function and btrfs_clear_delalloc_extent(). */ - if (!btrfs_is_free_space_inode(inode) && prev_delalloc_bytes == 0) + if (prev_delalloc_bytes == 0) btrfs_add_delalloc_inode(inode); } @@ -2690,7 +2690,6 @@ void btrfs_clear_delalloc_extent(struct btrfs_inode *inode, return; if (!btrfs_is_data_reloc_root(root) && - !btrfs_is_free_space_inode(inode) && !(state->state & EXTENT_NORESERVE) && (bits & EXTENT_CLEAR_DATA_RESV)) btrfs_free_reserved_data_space_noquota(inode, len); @@ -2708,7 +2707,7 @@ void btrfs_clear_delalloc_extent(struct btrfs_inode *inode, * and are therefore protected against concurrent calls of this * function and btrfs_set_delalloc_extent(). */ - if (!btrfs_is_free_space_inode(inode) && new_delalloc_bytes == 0) { + if (new_delalloc_bytes == 0) { spin_lock(&root->delalloc_lock); btrfs_del_delalloc_inode(inode); spin_unlock(&root->delalloc_lock); From 62306ae8550aace11ac6aab34257b9c8fa9be2d3 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:12 -0400 Subject: [PATCH 0077/1012] btrfs: stop reading free space inodes from the commit root Free space inode data was only read when loading the v1 cache, which is gone, so btrfs_get_extent() and btrfs_lookup_bio_sums() no longer need to search the commit root for them. Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/file-item.c | 11 ----------- fs/btrfs/inode.c | 10 ---------- 2 files changed, 21 deletions(-) diff --git a/fs/btrfs/file-item.c b/fs/btrfs/file-item.c index ae1fd4da38d31b..ff8f8cad00fc0d 100644 --- a/fs/btrfs/file-item.c +++ b/fs/btrfs/file-item.c @@ -396,17 +396,6 @@ int btrfs_lookup_bio_sums(struct btrfs_bio *bbio) if (nblocks > fs_info->csums_per_leaf) path->reada = READA_FORWARD; - /* - * the free space stuff is only read when it hasn't been - * updated in the current transaction. So, we can safely - * read from the commit root and sidestep a nasty deadlock - * between reading the free space cache and updating the csum tree. - */ - if (btrfs_is_free_space_inode(inode)) { - path->search_commit_root = true; - path->skip_locking = true; - } - /* * If we are searching for a csum of an extent from a past * transaction, we can search in the commit root and reduce diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c index 34940147393f73..f4b68205f621e3 100644 --- a/fs/btrfs/inode.c +++ b/fs/btrfs/inode.c @@ -7234,16 +7234,6 @@ struct extent_map *btrfs_get_extent(struct btrfs_inode *inode, /* Chances are we'll be called again, so go ahead and do readahead */ path->reada = READA_FORWARD; - /* - * The same explanation in load_free_space_cache applies here as well, - * we only read when we're loading the free space cache, and at that - * point the commit_root has everything we need. - */ - if (btrfs_is_free_space_inode(inode)) { - path->search_commit_root = true; - path->skip_locking = true; - } - ret = btrfs_lookup_file_extent(NULL, root, path, objectid, start, 0); if (ret < 0) { goto out; From 1397021ba66f3c04ce85760af47d0cd7f52f7d4b Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Thu, 17 Sep 2026 00:00:13 -0400 Subject: [PATCH 0078/1012] btrfs: remove TRANS_JOIN_NOLOCK btrfs_join_transaction_spacecache() was the only user of TRANS_JOIN_NOLOCK and is gone, so remove the join type, its entries in the blocked types table, and the special cases in join_transaction() and start_transaction(). Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Tal Zussman Signed-off-by: David Sterba --- fs/btrfs/transaction.c | 17 +---------------- fs/btrfs/transaction.h | 2 -- 2 files changed, 1 insertion(+), 18 deletions(-) diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c index a875008feb8fce..84f012bfcffca6 100644 --- a/fs/btrfs/transaction.c +++ b/fs/btrfs/transaction.c @@ -125,17 +125,14 @@ static const unsigned int btrfs_blocked_trans_types[TRANS_STATE_MAX] = { [TRANS_STATE_UNBLOCKED] = (__TRANS_START | __TRANS_ATTACH | __TRANS_JOIN | - __TRANS_JOIN_NOLOCK | __TRANS_JOIN_NOSTART), [TRANS_STATE_SUPER_COMMITTED] = (__TRANS_START | __TRANS_ATTACH | __TRANS_JOIN | - __TRANS_JOIN_NOLOCK | __TRANS_JOIN_NOSTART), [TRANS_STATE_COMPLETED] = (__TRANS_START | __TRANS_ATTACH | __TRANS_JOIN | - __TRANS_JOIN_NOLOCK | __TRANS_JOIN_NOSTART), }; @@ -310,12 +307,6 @@ static noinline int join_transaction(struct btrfs_fs_info *fs_info, if (type == TRANS_ATTACH || type == TRANS_JOIN_NOSTART) return -ENOENT; - /* - * JOIN_NOLOCK only happens during the transaction commit, so - * it is impossible that ->running_transaction is NULL - */ - BUG_ON(type == TRANS_JOIN_NOLOCK); - cur_trans = kmalloc_obj(*cur_trans, GFP_NOFS); if (!cur_trans) return -ENOMEM; @@ -709,14 +700,8 @@ start_transaction(struct btrfs_root *root, unsigned int num_items, } /* - * If we are JOIN_NOLOCK we're already committing a transaction and - * waiting on this guy, so we don't need to do the sb_start_intwrite - * because we're already holding a ref. We need this because we could - * have raced in and did an fsync() on a file which can kick a commit - * and then we deadlock with somebody doing a freeze. - * * If we are ATTACH, it means we just want to catch the current - * transaction and commit it, so we needn't do sb_start_intwrite(). + * transaction and commit it, so we needn't do sb_start_intwrite(). */ if (type & __TRANS_FREEZABLE) sb_start_intwrite(fs_info->sb); diff --git a/fs/btrfs/transaction.h b/fs/btrfs/transaction.h index 70b6c95efe53af..68c724e708095a 100644 --- a/fs/btrfs/transaction.h +++ b/fs/btrfs/transaction.h @@ -106,7 +106,6 @@ enum { ENUM_BIT(__TRANS_START), ENUM_BIT(__TRANS_ATTACH), ENUM_BIT(__TRANS_JOIN), - ENUM_BIT(__TRANS_JOIN_NOLOCK), ENUM_BIT(__TRANS_DUMMY), ENUM_BIT(__TRANS_JOIN_NOSTART), }; @@ -114,7 +113,6 @@ enum { #define TRANS_START (__TRANS_START | __TRANS_FREEZABLE) #define TRANS_ATTACH (__TRANS_ATTACH) #define TRANS_JOIN (__TRANS_JOIN | __TRANS_FREEZABLE) -#define TRANS_JOIN_NOLOCK (__TRANS_JOIN_NOLOCK) #define TRANS_JOIN_NOSTART (__TRANS_JOIN_NOSTART) #define TRANS_EXTWRITERS (__TRANS_START | __TRANS_ATTACH) From d05355bdd84d11496826a49d465d71bef37d4055 Mon Sep 17 00:00:00 2001 From: Puranjay Mohan Date: Mon, 10 Aug 2026 05:27:50 -0700 Subject: [PATCH 0079/1012] rcu: Make call_rcu() safe to call from any context RCU's per-CPU callback list is only touched with interrupts disabled: the enqueue runs under local_irq_save() (and the nocb locks when offloaded), as do callback invocation and grace-period work. A call_rcu() that arrives with interrupts already disabled, whether from an NMI or from instrumentation that re-enters RCU, can interrupt one of those and corrupt the list or deadlock. Defer instead: stage the callback on a per-CPU llist and raise an irq_work that re-issues it once interrupts are on, straight to the enqueue so it cannot defer again. The gate is bare irqs_disabled(), so callers that merely hold interrupts off are deferred too and pay one irq_work hop. Skip it while the scheduler is down (RCU_SCHEDULER_INACTIVE): irq_work is not usable that early, rcu_init() already calls call_rcu(), and the per-CPU deferral state is not initialised until rcu_init_one() runs later in it. rcu_barrier() drains every CPU's ->defer_head before it scans the lists, and rcutree_migrate_callbacks() drains an outgoing CPU's. A drain re-issues onto the draining CPU, so a barrier moves other CPUs' staged callbacks onto its own ->cblist; call_rcu() promises no CPU affinity for invocation. ->defer_lock is held across llist_del_all() and the whole re-issue so the drainers serialize: one that finds the list empty can conclude that everything staged before it is already on a callback list. Interrupts stay off for the batch. Where the arch has an irq_work self-IPI that is what one interrupts-disabled region could stage, normally a single callback; where arch_irq_work_has_interrupt() is false the drain waits for the tick, so several regions can accumulate first. The drain clears ->next before re-issuing. A double call_rcu() on a head that is already debug-object-active self-links the staged node, and rcu_do_enqueue()'s duplicate path returns without clearing it, so the drain would spin. A re-add behind other staged callbacks makes a longer cycle, which that does not bound; a double call_rcu() stays undefined. llist_del_all() yields newest-first, so a batch is re-issued in reverse call order; nothing depends on call_rcu() ordering. The re-issue drops the lazy hint, since staging records only ->func, so a deferred callback loses its batching on CONFIG_RCU_LAZY. kasan_record_aux_stack() moves to __call_rcu_common() so a use-after-free report names the caller rather than the irq_work. The re-issue runs with interrupts disabled, so instrumentation on the enqueue path can re-enter call_rcu(), stage another callback and re-raise the irq_work, livelocking the drain. A per-CPU flag guards it: a deferral that arrives while this CPU is draining, and is not from an NMI, is dropped. The WARN_ONCE() is under CONFIG_PROVE_RCU, so a production kernel drops it silently. That leaks the callback and can strand state the caller tied to it, since a one-shot flag only the callback clears never resets, but the alternative is an unbounded loop. A callback deferred past the CPUHP_AP_SMPCFD_DYING irq_work flush leaves ->defer_work claimed with its self-IPI lost. rcutree_migrate_callbacks() still re-issues the callback, but the first deferral after that CPU comes back raises no IPI and waits for the next irq_work there, or for rcu_barrier(). Unqueueing an irq_work is not something the API offers. The irq_work is IRQ_WORK_INIT_HARD so the re-issue stays prompt on PREEMPT_RT, where a non-HARD irq_work runs in a kthread that can be delayed under load. A hidden CONFIG_RCU_DEFER gates the deferral code and its IRQ_WORK dependency, though the rcu_data members are unconditional; without it call_rcu() enqueues directly as before. Under CONFIG_PROVE_RCU, warn if the direct path is reached from an NMI. Suggested-by: Paul E. McKenney Signed-off-by: Puranjay Mohan Signed-off-by: Paul E. McKenney --- kernel/rcu/Kconfig | 6 +++ kernel/rcu/rcu.h | 11 ++++ kernel/rcu/tree.c | 131 +++++++++++++++++++++++++++++++++++++++++---- kernel/rcu/tree.h | 6 +++ 4 files changed, 143 insertions(+), 11 deletions(-) diff --git a/kernel/rcu/Kconfig b/kernel/rcu/Kconfig index 332df7a7a6347c..bc9483c0b5d84f 100644 --- a/kernel/rcu/Kconfig +++ b/kernel/rcu/Kconfig @@ -175,6 +175,12 @@ config RCU_STALL_COMMON config RCU_NEED_SEGCBLIST def_bool ( TREE_RCU || TREE_SRCU || TASKS_RCU_GENERIC ) +# The deferral (and the IRQ_WORK it uses) is only needed where call_rcu() / +# call_srcu() can be invoked while a callback-list operation is in flight. +config RCU_DEFER + def_bool HAVE_NMI || KPROBES || FUNCTION_TRACER || TRACEPOINTS + select IRQ_WORK + config RCU_FANOUT int "Tree-based hierarchical RCU fanout value" range 2 64 if 64BIT diff --git a/kernel/rcu/rcu.h b/kernel/rcu/rcu.h index 39a9f6fa9a7b29..91e33571a554d9 100644 --- a/kernel/rcu/rcu.h +++ b/kernel/rcu/rcu.h @@ -572,6 +572,17 @@ static inline void tasks_cblist_init_generic(void) { } #define RCU_SCHEDULER_INIT 1 #define RCU_SCHEDULER_RUNNING 2 +/* + * Defer whenever interrupts are disabled, since a callback-list operation may + * be in flight on this CPU. Not before the scheduler is up: irq_work is not + * usable that early, and rcu_init() itself calls call_rcu(). + */ +static inline bool should_rcu_defer(void) +{ + return IS_ENABLED(CONFIG_RCU_DEFER) && irqs_disabled() && + rcu_scheduler_active != RCU_SCHEDULER_INACTIVE; +} + enum rcutorture_type { RCU_FLAVOR, RCU_TASKS_FLAVOR, diff --git a/kernel/rcu/tree.c b/kernel/rcu/tree.c index 96848fc1f02b8f..ff9a2395c9e8e9 100644 --- a/kernel/rcu/tree.c +++ b/kernel/rcu/tree.c @@ -24,6 +24,7 @@ #include #include #include +#include #include #include #include @@ -3148,21 +3149,19 @@ static void check_cb_ovld(struct rcu_data *rdp) raw_spin_unlock_rcu_node(rnp); } -static void -__call_rcu_common(struct rcu_head *head, rcu_callback_t func, bool lazy_in) +/* + * Also called by __rcu_defer_drain() to re-issue a deferred callback, so it + * must not re-check the deferral condition. Either caller may have interrupts + * already disabled, and a drain of a remote CPU re-issues onto the draining + * CPU. + */ +static void rcu_do_enqueue(struct rcu_head *head, rcu_callback_t func, bool lazy_in) { static atomic_t doublefrees; unsigned long flags; bool lazy; struct rcu_data *rdp; - /* Misaligned rcu_head! */ - WARN_ON_ONCE((unsigned long)head & (sizeof(void *) - 1)); - - /* Avoid NULL dereference if callback is NULL. */ - if (WARN_ON_ONCE(!func)) - return; - if (debug_rcu_head_queue(head)) { /* * Probable double call_rcu(), so leak the callback. @@ -3178,7 +3177,6 @@ __call_rcu_common(struct rcu_head *head, rcu_callback_t func, bool lazy_in) } head->func = func; head->next = NULL; - kasan_record_aux_stack(head); local_irq_save(flags); rdp = this_cpu_ptr(&rcu_data); @@ -3206,6 +3204,103 @@ __call_rcu_common(struct rcu_head *head, rcu_callback_t func, bool lazy_in) local_irq_restore(flags); } +/* + * Re-issue deferred callbacks straight to the enqueue so they cannot defer + * again. ->defer_lock serializes the drainers: this CPU's irq_work, + * rcu_defer_flush() and rcutree_migrate_callbacks(). + */ +static void __rcu_defer_drain(struct rcu_data *rdp) +{ + struct llist_node *node, *next; + unsigned long flags; + + if (!IS_ENABLED(CONFIG_RCU_DEFER)) + return; + + raw_spin_lock_irqsave(&rdp->defer_lock, flags); + llist_for_each_safe(node, next, llist_del_all(&rdp->defer_head)) { + struct rcu_head *head = (struct rcu_head *)node; + + /* Bounds a node self-linked by a double call_rcu(). */ + head->next = NULL; + rcu_do_enqueue(head, head->func, false); + } + raw_spin_unlock_irqrestore(&rdp->defer_lock, flags); +} + +/* + * Only the irq_work drain can be re-fed by its own re-issue, so only it sets + * ->defer_draining. Anything staged during a direct drain is picked up by the + * staging CPU's own irq_work. Every caller of irq_work_run_list() has + * interrupts disabled, so the flag is never visible with them enabled. + */ +static void rcu_defer_drain(struct irq_work *iw) +{ + struct rcu_data *rdp = container_of(iw, struct rcu_data, defer_work); + + WRITE_ONCE(rdp->defer_draining, true); + __rcu_defer_drain(rdp); + WRITE_ONCE(rdp->defer_draining, false); +} + +/* + * Stage @head for this CPU's irq_work to re-issue once interrupts are on. Only + * the drain side takes a lock, so this stays safe from NMI. + */ +static void call_rcu_defer(struct rcu_head *head, rcu_callback_t func) +{ + struct rcu_data *rdp = this_cpu_ptr(&rcu_data); + + /* + * Instrumentation on the enqueue path can re-enter here from inside the + * drain. Re-queuing would livelock it, so drop the callback; an NMI + * cannot loop, so let it through. + */ + if (READ_ONCE(rdp->defer_draining) && !in_nmi()) { + WARN_ONCE(IS_ENABLED(CONFIG_PROVE_RCU), + "call_rcu() re-entered during callback drain; leaking callback\n"); + return; + } + head->func = func; + if (llist_add((struct llist_node *)head, &rdp->defer_head)) + irq_work_queue(&rdp->defer_work); +} + +static void rcu_defer_flush(void) +{ + int cpu; + + for_each_possible_cpu(cpu) + __rcu_defer_drain(per_cpu_ptr(&rcu_data, cpu)); +} + +static void +__call_rcu_common(struct rcu_head *head, rcu_callback_t func, bool lazy_in) +{ + /* Misaligned rcu_head! */ + WARN_ON_ONCE((unsigned long)head & (sizeof(void *) - 1)); + + /* Avoid NULL dereference if callback is NULL. */ + if (WARN_ON_ONCE(!func)) + return; + + /* Record the caller: the irq_work's stack says nothing about it. */ + kasan_record_aux_stack(head); + + if (should_rcu_defer()) { + call_rcu_defer(head, func); + return; + } + + /* + * Only reachable from an NMI when deferral is off: before the scheduler + * is up, or with CONFIG_RCU_DEFER=n. The enqueue can then race. + */ + WARN_ON_ONCE(IS_ENABLED(CONFIG_PROVE_RCU) && in_nmi()); + + rcu_do_enqueue(head, func, lazy_in); +} + #ifdef CONFIG_RCU_LAZY static bool enable_rcu_lazy __read_mostly = !IS_ENABLED(CONFIG_RCU_LAZY_DEFAULT_OFF); module_param(enable_rcu_lazy, bool, 0444); @@ -3896,8 +3991,12 @@ void rcu_barrier(void) unsigned long flags; unsigned long gseq; struct rcu_data *rdp; - unsigned long s = rcu_seq_snap(&rcu_state.barrier_sequence); + unsigned long s; + /* Register any deferred callbacks before snapshotting the sequence. */ + rcu_defer_flush(); + + s = rcu_seq_snap(&rcu_state.barrier_sequence); rcu_barrier_trace(TPS("Begin"), -1, s); /* Take mutex to serialize concurrent rcu_barrier() requests. */ @@ -4231,6 +4330,9 @@ rcu_boot_init_percpu_data(int cpu) rdp->rcu_onl_gp_state = RCU_GP_CLEANED; rdp->last_sched_clock = jiffies; rdp->cpu = cpu; + init_llist_head(&rdp->defer_head); + raw_spin_lock_init(&rdp->defer_lock); + rdp->defer_work = IRQ_WORK_INIT_HARD(rcu_defer_drain); rcu_boot_init_nocb_percpu_data(rdp); } @@ -4528,6 +4630,13 @@ void rcutree_migrate_callbacks(int cpu) struct rcu_data *rdp = per_cpu_ptr(&rcu_data, cpu); bool needwake; + /* + * Callbacks deferred past the point the outgoing CPU's irq_work can run + * sit on ->defer_head, which the ->cblist migration below does not + * cover. Drain them here, before the early returns. + */ + __rcu_defer_drain(rdp); + if (rcu_rdp_is_offloaded(rdp)) return; diff --git a/kernel/rcu/tree.h b/kernel/rcu/tree.h index eedfa43059e802..b7cac7a13b4f22 100644 --- a/kernel/rcu/tree.h +++ b/kernel/rcu/tree.h @@ -229,6 +229,12 @@ struct rcu_data { struct rcu_head barrier_head; int exp_watching_snap; /* Double-check need for IPI. */ + /* Deferral of an NMI/reentrant call_rcu(); see __call_rcu_common(). */ + struct llist_head defer_head; + struct irq_work defer_work; + raw_spinlock_t defer_lock; + bool defer_draining; + /* 5) Callback offloading. */ #ifdef CONFIG_RCU_NOCB_CPU struct swait_queue_head nocb_cb_wq; /* For nocb kthreads to sleep on. */ From 2402db26994492b616198eaf6dc846c4a4523866 Mon Sep 17 00:00:00 2001 From: Puranjay Mohan Date: Mon, 10 Aug 2026 05:27:51 -0700 Subject: [PATCH 0080/1012] rcu: Make Tiny call_rcu() safe to call from any context Give Tiny call_rcu() the same treatment as Tree RCU. When interrupts are disabled and the scheduler is up, stage the callback on a lockless list that an irq_work re-issues later. One global list and irq_work suffice since Tiny RCU is uniprocessor, and there is no CPU-offline drain. The re-issue runs with interrupts disabled and can be re-entered by instrumentation, so a draining flag drops a deferring call_rcu() seen mid-drain (unless from an NMI), as in Tree RCU. Gated by CONFIG_RCU_DEFER, though the deferral state is unconditional. Interrupts stay off for the whole batch, but the re-issue is a tail append with no locks. TINY_RCU implies !SMP, where arch_irq_work_has_interrupt() is false, so the drain always waits for the tick and a batch is whatever one tick's worth of interrupts-disabled call_rcu()s staged. As in Tree RCU the drain clears ->next before re-issuing, which bounds a node self-linked by a double call_rcu(): rcu_do_enqueue()'s duplicate path returns without clearing it. A longer cycle is not bounded; a double call_rcu() stays undefined. The idle-task reschedule moves out of the enqueue helper so that a drain does it once for the batch rather than once per callback, which would otherwise take the runqueue lock N times with interrupts disabled. Suggested-by: Paul E. McKenney Signed-off-by: Puranjay Mohan Signed-off-by: Paul E. McKenney --- kernel/rcu/tiny.c | 127 +++++++++++++++++++++++++++++++++++++--------- 1 file changed, 104 insertions(+), 23 deletions(-) diff --git a/kernel/rcu/tiny.c b/kernel/rcu/tiny.c index dccccd6be94111..656b6a682e31a0 100644 --- a/kernel/rcu/tiny.c +++ b/kernel/rcu/tiny.c @@ -11,6 +11,8 @@ */ #include #include +#include +#include #include #include #include @@ -42,8 +44,100 @@ static struct rcu_ctrlblk rcu_ctrlblk = { .gp_seq = 0 - 300UL, }; +/* + * The callback list is only accessed with interrupts disabled, so a call_rcu() + * that arrives with interrupts off stages the callback on a lockless list that + * an irq_work re-issues later. One global list and irq_work suffice, as Tiny + * RCU is uniprocessor. + */ +static void rcu_defer_drain(struct irq_work *iw); +static LLIST_HEAD(rcu_defer_list); +static struct irq_work rcu_defer_iw = IRQ_WORK_INIT_HARD(rcu_defer_drain); +static bool rcu_defer_draining; + +/* + * Also called by __rcu_defer_drain() to re-issue a deferred callback, so it + * must not re-check the deferral condition. + */ +static void rcu_do_enqueue(struct rcu_head *head, rcu_callback_t func) +{ + static atomic_t doublefrees; + unsigned long flags; + + if (debug_rcu_head_queue(head)) { + if (atomic_inc_return(&doublefrees) < 4) { + pr_err("%s(): Double-freed CB %p->%pS()!!! ", __func__, head, head->func); + mem_dump_obj(head); + } + return; + } + + head->func = func; + head->next = NULL; + + local_irq_save(flags); + *rcu_ctrlblk.curtail = head; + rcu_ctrlblk.curtail = &head->next; + local_irq_restore(flags); +} + +/* Force scheduling for rcu_qs() when enqueuing from the idle task. */ +static void rcu_resched_if_idle(void) +{ + if (unlikely(is_idle_task(current))) + resched_cpu(0); +} + +static void __rcu_defer_drain(void) +{ + struct llist_node *node, *next; + bool drained = false; + unsigned long flags; + + if (!IS_ENABLED(CONFIG_RCU_DEFER)) + return; + + /* Re-issued newest-first; nothing depends on call_rcu() ordering. */ + local_irq_save(flags); + llist_for_each_safe(node, next, llist_del_all(&rcu_defer_list)) { + struct rcu_head *head = (struct rcu_head *)node; + + /* Bounds a node self-linked by a double call_rcu(). */ + head->next = NULL; + rcu_do_enqueue(head, head->func); + drained = true; + } + local_irq_restore(flags); + + if (drained) + rcu_resched_if_idle(); +} + +/* Only the irq_work drain can be re-fed by its own re-issue; see Tree RCU. */ +static void rcu_defer_drain(struct irq_work *iw) +{ + WRITE_ONCE(rcu_defer_draining, true); + __rcu_defer_drain(); + WRITE_ONCE(rcu_defer_draining, false); +} + +static void call_rcu_defer(struct rcu_head *head, rcu_callback_t func) +{ + /* A re-entrant call_rcu() during the drain would livelock it; drop it. */ + if (READ_ONCE(rcu_defer_draining) && !in_nmi()) { + WARN_ONCE(IS_ENABLED(CONFIG_PROVE_RCU), + "call_rcu() re-entered during callback drain; leaking callback\n"); + return; + } + head->func = func; + if (llist_add((struct llist_node *)head, &rcu_defer_list)) + irq_work_queue(&rcu_defer_iw); +} + void rcu_barrier(void) { + /* Register any deferred callbacks so the wait below covers them. */ + __rcu_defer_drain(); wait_rcu_gp(call_rcu_hurry); } EXPORT_SYMBOL(rcu_barrier); @@ -157,29 +251,19 @@ EXPORT_SYMBOL_GPL(synchronize_rcu); */ void call_rcu(struct rcu_head *head, rcu_callback_t func) { - static atomic_t doublefrees; - unsigned long flags; - - if (debug_rcu_head_queue(head)) { - if (atomic_inc_return(&doublefrees) < 4) { - pr_err("%s(): Double-freed CB %p->%pS()!!! ", __func__, head, head->func); - mem_dump_obj(head); - } + if (should_rcu_defer()) { + call_rcu_defer(head, func); return; } - head->func = func; - head->next = NULL; + /* + * Only reachable from an NMI when deferral is off: before the scheduler + * is up, or with CONFIG_RCU_DEFER=n. The enqueue can then race. + */ + WARN_ON_ONCE(IS_ENABLED(CONFIG_PROVE_RCU) && in_nmi()); - local_irq_save(flags); - *rcu_ctrlblk.curtail = head; - rcu_ctrlblk.curtail = &head->next; - local_irq_restore(flags); - - if (unlikely(is_idle_task(current))) { - /* force scheduling for rcu_qs() */ - resched_cpu(0); - } + rcu_do_enqueue(head, func); + rcu_resched_if_idle(); } EXPORT_SYMBOL_GPL(call_rcu); @@ -211,10 +295,7 @@ unsigned long start_poll_synchronize_rcu(void) { unsigned long gp_seq = get_state_synchronize_rcu(); - if (unlikely(is_idle_task(current))) { - /* force scheduling for rcu_qs() */ - resched_cpu(0); - } + rcu_resched_if_idle(); return gp_seq; } EXPORT_SYMBOL_GPL(start_poll_synchronize_rcu); From 8e5d94ff42236b732cc2b5d0d5cae9e66e0cee34 Mon Sep 17 00:00:00 2001 From: Puranjay Mohan Date: Mon, 10 Aug 2026 05:27:52 -0700 Subject: [PATCH 0081/1012] srcu: Make call_srcu() safe to call from any context call_srcu() has the same constraint as call_rcu(): srcu_gp_start_if_needed() enqueues under raw_spin_lock_irqsave() and may walk the srcu_node tree, as do callback invocation and grace-period work, so a call_srcu() with interrupts already disabled can race an operation in flight on this CPU. call_rcu_tasks_trace() is call_srcu() under the hood, so a sleepable BPF program freeing an object can reach this. Defer as call_rcu() does: stage the callback on the srcu_data's ->defer_cbs, chain that srcu_data onto a per-CPU list, and raise a per-CPU irq_work that re-issues it straight to the enqueue helper, never back through __call_srcu(). The irqs-enabled path is unchanged; as for call_rcu() the gate is bare irqs_disabled(), so callers that merely hold interrupts off are deferred too and pay one irq_work hop, including call_rcu_tasks_trace() from the BPF memalloc irq_work. The irq_work is per-CPU rather than per-srcu_struct and statically initialized, so deferral never runs check_init_srcu_struct(); it is IRQ_WORK_INIT_HARD as for call_rcu(). srcu_barrier() flushes it first, and rcutree_migrate_callbacks() calls srcu_offline_drain() for an outgoing CPU. cleanup_srcu_struct() drains before its "just leak it" early returns, and srcu_module_going() before freeing any ->sda, since a staged srcu_data left chained on a per-CPU list would dangle. Staging is two steps, the callback onto ->defer_cbs and then the srcu_data onto the per-CPU list, so a flusher can find the per-CPU list empty while a callback whose call_srcu() has not returned sits on ->defer_cbs; the staging CPU's own irq_work takes that one. The per-CPU srcu_defer ->lock, not any srcu_data's, is held with interrupts off across the whole nested drain: the chain of srcu_datas staged on that CPU and, for each, its callbacks, with srcu_do_enqueue() taking that srcu_data's ->lock and possibly starting a grace period for every one. That is what serializes the drainers. The bound is as for call_rcu(): what one interrupts-disabled region could stage, normally a single callback, or whatever accumulates before the tick where arch_irq_work_has_interrupt() is false. Both lists are drained newest-first; nothing depends on call_srcu() ordering. The drain clears ->next before re-issuing, which bounds a node self-linked by a double call_srcu(); a longer cycle is not bounded, and a double call_srcu() stays undefined, as for call_rcu(). A callback deferred past the CPUHP_AP_SMPCFD_DYING irq_work flush leaves that CPU's srcu_defer ->iw claimed with its self-IPI lost, as for call_rcu(). srcu_offline_drain() still re-issues the callback, but the irq_work cannot be un-queued, and here the claim is shared by every srcu_struct on the CPU. As in call_rcu(), the re-issue runs with interrupts disabled and can be re-entered by instrumentation, so a per-CPU flag, set only while that CPU is inside its own irq_work drain, drops a deferring call_srcu() seen mid-drain unless it comes from an NMI. Such a drop can strand state the caller associated with the callback, not just the callback itself. Staging records only the callback, so a deferred expedited call_srcu() completes as a normal grace period. Only srcu_expedite_current() can hit that, and only when invoked with interrupts already disabled. Gated by CONFIG_RCU_DEFER, though the srcu_data members and the per-CPU srcu_defer are unconditional. Under CONFIG_PROVE_RCU, warn if the direct path is reached from an NMI. Suggested-by: Paul E. McKenney Signed-off-by: Puranjay Mohan Signed-off-by: Paul E. McKenney --- include/linux/srcutree.h | 4 + kernel/rcu/rcu.h | 3 + kernel/rcu/srcutree.c | 170 ++++++++++++++++++++++++++++++++++++++- kernel/rcu/tree.c | 2 + 4 files changed, 175 insertions(+), 4 deletions(-) diff --git a/include/linux/srcutree.h b/include/linux/srcutree.h index 75e54e4f963fac..1ce759fb70948b 100644 --- a/include/linux/srcutree.h +++ b/include/linux/srcutree.h @@ -13,6 +13,8 @@ #include #include +#include +#include struct srcu_node; struct srcu_struct; @@ -41,6 +43,8 @@ struct srcu_data { bool srcu_cblist_invoking; /* Invoking these CBs? */ struct timer_list delay_work; /* Delay for CB invoking */ struct work_struct work; /* Context for CB invoking. */ + struct llist_head defer_cbs; /* Callbacks deferred on re-entry. */ + struct llist_node defer_link; /* Links onto the per-CPU deferral drain list */ struct rcu_head srcu_barrier_head; /* For srcu_barrier() use. */ struct rcu_head srcu_ec_head; /* For srcu_expedite_current() use. */ int srcu_ec_state; /* State for srcu_expedite_current(). */ diff --git a/kernel/rcu/rcu.h b/kernel/rcu/rcu.h index 91e33571a554d9..d60444bf3a02dd 100644 --- a/kernel/rcu/rcu.h +++ b/kernel/rcu/rcu.h @@ -583,6 +583,9 @@ static inline bool should_rcu_defer(void) rcu_scheduler_active != RCU_SCHEDULER_INACTIVE; } +/* Drain an outgoing CPU's deferred SRCU callbacks; see rcutree_migrate_callbacks(). */ +void srcu_offline_drain(int cpu); + enum rcutorture_type { RCU_FLAVOR, RCU_TASKS_FLAVOR, diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index ed204b3f4b8440..d32c374ee72af8 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -20,6 +20,7 @@ #include #include #include +#include #include #include #include @@ -79,6 +80,38 @@ static void process_srcu(struct work_struct *work); static void srcu_irq_work(struct irq_work *work); static void srcu_delay_timer(struct timer_list *t); +struct srcu_defer; +static void srcu_defer_drain(struct irq_work *iw); +static void __srcu_defer_drain(struct srcu_defer *sndp); + +/* + * Per-CPU call_srcu() deferral state, shared by every srcu_struct. A deferred + * callback is staged on its srcu_data's ->defer_cbs; that srcu_data is chained + * via ->defer_link onto ->list, which the irq_work walks. + */ +struct srcu_defer { + struct llist_head list; + struct irq_work iw; + raw_spinlock_t lock; + bool draining; +}; + +static DEFINE_PER_CPU(struct srcu_defer, srcu_defer) = { + .lock = __RAW_SPIN_LOCK_UNLOCKED(srcu_defer.lock), + .iw = IRQ_WORK_INIT_HARD(srcu_defer_drain), +}; + +/* + * Flush pending deferred callbacks so a following srcu_barrier() waits for them. + */ +static void srcu_defer_flush(void) +{ + int cpu; + + for_each_possible_cpu(cpu) + __srcu_defer_drain(&per_cpu(srcu_defer, cpu)); +} + /* * Initialize SRCU per-CPU data. Note that statically allocated * srcu_struct structures might already have srcu_read_lock() and @@ -107,6 +140,11 @@ static void init_srcu_struct_data(struct srcu_struct *ssp) sdp->cpu = cpu; INIT_WORK(&sdp->work, srcu_invoke_callbacks); timer_setup(&sdp->delay_work, srcu_delay_timer, 0); + /* + * ->defer_cbs and ->defer_link are valid when zeroed and are not + * reinitialized here: that would clobber callbacks a reentrant + * call_srcu() already staged. See __call_srcu(). + */ sdp->ssp = ssp; } } @@ -688,6 +726,14 @@ void cleanup_srcu_struct(struct srcu_struct *ssp) unsigned long delay; struct srcu_usage *sup = ssp->srcu_sup; + /* + * Drain before the early returns below: they leak the srcu_struct, but + * srcu_module_going() frees ->sda regardless, and a staged srcu_data + * left chained on a per-CPU list would then dangle. Draining first also + * has to precede the ->irq_work sync, since re-issuing a callback can + * start a grace period and re-queue ->irq_work, which schedules ->work. + */ + srcu_defer_flush(); raw_spin_lock_irq_rcu_node(ssp->srcu_sup); delay = srcu_get_delay(ssp); raw_spin_unlock_irq_rcu_node(ssp->srcu_sup); @@ -695,7 +741,6 @@ void cleanup_srcu_struct(struct srcu_struct *ssp) return; /* Just leak it! */ if (WARN_ON(srcu_readers_active(ssp))) return; /* Just leak it! */ - /* Wait for irq_work to finish first as it may queue a new work. */ irq_work_sync(&sup->irq_work); flush_delayed_work(&sup->work); for_each_possible_cpu(cpu) { @@ -1411,8 +1456,8 @@ static unsigned long srcu_gp_start_if_needed(struct srcu_struct *ssp, * srcu_read_lock(), and srcu_read_unlock() that are all passed the same * srcu_struct structure. */ -static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp, - rcu_callback_t func, bool do_norm) +static void srcu_do_enqueue(struct srcu_struct *ssp, struct rcu_head *rhp, + rcu_callback_t func, bool do_norm) { if (debug_rcu_head_queue(rhp)) { /* Probable double call_srcu(), so leak the callback. */ @@ -1424,6 +1469,108 @@ static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp, (void)srcu_gp_start_if_needed(ssp, rhp, do_norm); } +/* + * The srcu_cblist and srcu_node tree are only accessed with interrupts + * disabled, so defer when interrupts are already off rather than enqueue into + * an operation that may be in flight on this CPU. + */ +static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp, + rcu_callback_t func, bool do_norm) +{ + if (should_rcu_defer()) { + struct srcu_defer *sndp = this_cpu_ptr(&srcu_defer); + struct srcu_data *sdp; + + /* + * Instrumentation on the enqueue path can re-enter here from + * inside the drain. Re-queuing would livelock it, so drop the + * callback; an NMI cannot loop, so let it through. + */ + if (READ_ONCE(sndp->draining) && !in_nmi()) { + WARN_ONCE(IS_ENABLED(CONFIG_PROVE_RCU), + "call_srcu() re-entered during callback drain; leaking callback\n"); + return; + } + sdp = this_cpu_ptr(ssp->sda); + rhp->func = func; + if (llist_add((struct llist_node *)rhp, &sdp->defer_cbs)) { + /* + * Chain this srcu_data for the drain. ->ssp must be + * published here: deferral skips + * check_init_srcu_struct(), so on a never-initialized + * static srcu_struct the srcu_data are still zeroed and + * the drain would read a NULL ->ssp. + */ + sdp->ssp = ssp; + if (llist_add(&sdp->defer_link, &sndp->list)) + irq_work_queue(&sndp->iw); + } + return; + } + + /* + * Only reachable from an NMI when deferral is off: before the scheduler + * is up, or with CONFIG_RCU_DEFER=n. The enqueue can then race. + */ + WARN_ON_ONCE(IS_ENABLED(CONFIG_PROVE_RCU) && in_nmi()); + + srcu_do_enqueue(ssp, rhp, func, do_norm); +} + +/* + * Re-issue deferred callbacks straight to srcu_do_enqueue() so they cannot defer + * again. ->lock serializes the drainers: the irq_work, srcu_defer_flush() and + * srcu_offline_drain(). + */ +static void __srcu_defer_drain(struct srcu_defer *sndp) +{ + struct llist_node *snode, *snext; + unsigned long flags; + + if (!IS_ENABLED(CONFIG_RCU_DEFER)) + return; + + raw_spin_lock_irqsave(&sndp->lock, flags); + llist_for_each_safe(snode, snext, llist_del_all(&sndp->list)) { + struct srcu_data *sdp = container_of(snode, struct srcu_data, defer_link); + struct srcu_struct *ssp = sdp->ssp; + struct llist_node *cnode, *cnext; + + cnode = llist_del_all(&sdp->defer_cbs); + llist_for_each_safe(cnode, cnext, cnode) { + struct rcu_head *rhp = (struct rcu_head *)cnode; + + /* Bounds a node self-linked by a double call_srcu(). */ + rhp->next = NULL; + srcu_do_enqueue(ssp, rhp, rhp->func, true); + } + } + raw_spin_unlock_irqrestore(&sndp->lock, flags); +} + +/* + * Only the irq_work drain can be re-fed by its own re-issue, so only it sets + * ->draining. A direct drain re-issues onto this CPU, and anything staged + * during it is picked up by that CPU's own irq_work. + */ +static void srcu_defer_drain(struct irq_work *iw) +{ + struct srcu_defer *sndp = container_of(iw, struct srcu_defer, iw); + + WRITE_ONCE(sndp->draining, true); + __srcu_defer_drain(sndp); + WRITE_ONCE(sndp->draining, false); +} + +/* + * Drain @cpu's deferred call_srcu() callbacks once @cpu is dead. One pass + * covers every srcu_struct; the re-issue lands on the current CPU. + */ +void srcu_offline_drain(int cpu) +{ + __srcu_defer_drain(&per_cpu(srcu_defer, cpu)); +} + /** * call_srcu() - Queue a callback for invocation after an SRCU grace period * @ssp: srcu_struct in queue the callback @@ -1678,9 +1825,18 @@ void srcu_barrier(struct srcu_struct *ssp) { int cpu; int idx; - unsigned long s = rcu_seq_snap(&ssp->srcu_sup->srcu_barrier_seq); + unsigned long s; check_init_srcu_struct(ssp); + + /* + * Register any deferred callbacks before snapshotting the sequence. The + * staging list is per-CPU, not per-srcu_struct, so this also drains + * other srcu_structs'. + */ + srcu_defer_flush(); + + s = rcu_seq_snap(&ssp->srcu_sup->srcu_barrier_seq); mutex_lock(&ssp->srcu_sup->srcu_barrier_mutex); if (rcu_seq_done(&ssp->srcu_sup->srcu_barrier_seq, s)) { smp_mb(); /* Force ordering following return. */ @@ -2135,6 +2291,12 @@ static void srcu_module_going(struct module *mod) struct srcu_struct *ssp; struct srcu_struct **sspp = mod->srcu_struct_ptrs; + /* + * Deferral skips check_init_srcu_struct(), so cleanup_srcu_struct() + * below can be skipped for an srcu_struct that has staged callbacks. + * Drain them before any ->sda is freed. + */ + srcu_defer_flush(); for (i = 0; i < mod->num_srcu_structs; i++) { ssp = *(sspp++); if (!rcu_seq_state(smp_load_acquire(&ssp->srcu_sup->srcu_gp_seq_needed)) && diff --git a/kernel/rcu/tree.c b/kernel/rcu/tree.c index ff9a2395c9e8e9..e363e1a6a33c9f 100644 --- a/kernel/rcu/tree.c +++ b/kernel/rcu/tree.c @@ -4636,6 +4636,8 @@ void rcutree_migrate_callbacks(int cpu) * cover. Drain them here, before the early returns. */ __rcu_defer_drain(rdp); + /* Likewise for the outgoing CPU's deferred call_srcu() callbacks. */ + srcu_offline_drain(cpu); if (rcu_rdp_is_offloaded(rdp)) return; From 0992dbc891b839d93e12c0b216b1619d85afed13 Mon Sep 17 00:00:00 2001 From: Puranjay Mohan Date: Mon, 10 Aug 2026 05:27:53 -0700 Subject: [PATCH 0082/1012] srcu: Make Tiny call_srcu() safe to call from any context Give Tiny call_srcu() the same treatment as Tree SRCU. When interrupts are disabled and the scheduler is up, stage the callback on the srcu_struct's lockless list for an irq_work to re-issue later. Tiny SRCU is uniprocessor, so there is no CPU-offline drain. A draining flag drops a deferring call_srcu() that re-enters mid-drain (unless from an NMI), as in Tree SRCU; such a drop can strand state the caller tied to the callback, not just the callback itself. Interrupts stay off for the whole batch. TINY_SRCU implies !SMP, where arch_irq_work_has_interrupt() is false, so the drain always waits for the tick and a batch is whatever one tick's worth of interrupts-disabled call_srcu()s staged. Unlike the other three flavors srcu_do_enqueue() here has no debug_rcu_head_queue(), so nothing reports a double call_srcu(); termination of the drain rests on srcu_do_enqueue() clearing ->next, and the callback list self-links at the tail exactly as a double call_srcu() made it before. srcu_barrier() (now out of line) and cleanup_srcu_struct() drain the deferred list first, so a deferred callback is re-issued onto the callback list and invoked by the grace-period work that cleanup_srcu_struct() flushes, rather than stranded on a soon-to-be-freed srcu_struct. cleanup_srcu_struct() also syncs ->defer_iw, since that irq_work is embedded in the srcu_struct the caller is about to free. The draining flag is global rather than per-srcu_struct: a re-entrant call_srcu(B) inside a drain of A raises B's own ->defer_iw, whose drain can stage back onto A, so a per-srcu_struct flag would not break the chain. The cost is that a drain of A also drops a non-NMI call_srcu() to any other srcu_struct for its duration. Gated by CONFIG_RCU_DEFER like Tree SRCU, though the srcu_struct members are unconditional. Suggested-by: Paul E. McKenney Signed-off-by: Puranjay Mohan Signed-off-by: Paul E. McKenney --- include/linux/srcutiny.h | 12 ++++-- kernel/rcu/srcutiny.c | 93 ++++++++++++++++++++++++++++++++++++++-- 2 files changed, 97 insertions(+), 8 deletions(-) diff --git a/include/linux/srcutiny.h b/include/linux/srcutiny.h index fbcf13bc12d15e..85b5de438450b3 100644 --- a/include/linux/srcutiny.h +++ b/include/linux/srcutiny.h @@ -12,6 +12,7 @@ #define _LINUX_SRCU_TINY_H #include +#include #include struct srcu_struct { @@ -26,6 +27,8 @@ struct srcu_struct { struct rcu_head **srcu_cb_tail; /* Pending callbacks: Tail. */ struct work_struct srcu_work; /* For driving grace periods. */ struct irq_work srcu_irq_work; /* Defer schedule_work() to irq work. */ + struct llist_head defer_cbs; /* Callbacks deferred on re-entry. */ + struct irq_work defer_iw; /* Re-issues defer_cbs later. */ #ifdef CONFIG_DEBUG_LOCK_ALLOC struct lockdep_map dep_map; #endif /* #ifdef CONFIG_DEBUG_LOCK_ALLOC */ @@ -33,6 +36,7 @@ struct srcu_struct { void srcu_drive_gp(struct work_struct *wp); void srcu_tiny_irq_work(struct irq_work *irq_work); +void srcu_defer_drain(struct irq_work *irq_work); #define __SRCU_STRUCT_INIT(name, __ignored, ___ignored, ____ignored) \ { \ @@ -40,6 +44,9 @@ void srcu_tiny_irq_work(struct irq_work *irq_work); .srcu_cb_tail = &name.srcu_cb_head, \ .srcu_work = __WORK_INITIALIZER(name.srcu_work, srcu_drive_gp), \ .srcu_irq_work = { .func = srcu_tiny_irq_work }, \ + .defer_cbs = LLIST_HEAD_INIT(name.defer_cbs), \ + .defer_iw = { .node = { .u_flags = IRQ_WORK_HARD_IRQ }, \ + .func = srcu_defer_drain }, \ __SRCU_DEP_MAP_INIT(name) \ } @@ -131,10 +138,7 @@ static inline void synchronize_srcu_expedited(struct srcu_struct *ssp) synchronize_srcu(ssp); } -static inline void srcu_barrier(struct srcu_struct *ssp) -{ - synchronize_srcu(ssp); -} +void srcu_barrier(struct srcu_struct *ssp); static inline void srcu_expedite_current(struct srcu_struct *ssp) { } #define srcu_check_read_flavor(ssp, read_flavor) do { } while (0) diff --git a/kernel/rcu/srcutiny.c b/kernel/rcu/srcutiny.c index 558ba8d316db6c..5de9a690583835 100644 --- a/kernel/rcu/srcutiny.c +++ b/kernel/rcu/srcutiny.c @@ -10,6 +10,7 @@ #include #include +#include #include #include #include @@ -29,6 +30,8 @@ extern int rcu_scheduler_active; static LIST_HEAD(srcu_boot_list); static bool srcu_init_done; +static void __srcu_defer_drain(struct srcu_struct *ssp); + static int init_srcu_struct_fields(struct srcu_struct *ssp) { ssp->srcu_lock_nesting[0] = 0; @@ -43,6 +46,8 @@ static int init_srcu_struct_fields(struct srcu_struct *ssp) INIT_WORK(&ssp->srcu_work, srcu_drive_gp); INIT_LIST_HEAD(&ssp->srcu_work.entry); init_irq_work(&ssp->srcu_irq_work, srcu_tiny_irq_work); + init_llist_head(&ssp->defer_cbs); + ssp->defer_iw = IRQ_WORK_INIT_HARD(srcu_defer_drain); return 0; } @@ -86,6 +91,16 @@ EXPORT_SYMBOL_GPL(init_srcu_struct_generic); void cleanup_srcu_struct(struct srcu_struct *ssp) { WARN_ON(srcu_readers_active(ssp)); + /* + * Re-issue any deferred callbacks, then wait out ->defer_iw before it is + * freed. Skipped entirely with CONFIG_RCU_DEFER=n: irq_work_sync() ends + * in an unconditional synchronize_rcu() wherever + * arch_irq_work_has_interrupt() is false, which is every !SMP target. + */ + if (IS_ENABLED(CONFIG_RCU_DEFER)) { + __srcu_defer_drain(ssp); + irq_work_sync(&ssp->defer_iw); + } irq_work_sync(&ssp->srcu_irq_work); flush_work(&ssp->srcu_work); WARN_ON(ssp->srcu_gp_running); @@ -213,11 +228,11 @@ static void srcu_gp_start_if_needed(struct srcu_struct *ssp) } /* - * Enqueue an SRCU callback on the specified srcu_struct structure, - * initiating grace-period processing if it is not already running. + * Also called by __srcu_defer_drain() to re-issue a deferred callback, so it + * must not re-check the deferral condition. */ -void call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp, - rcu_callback_t func) +static void srcu_do_enqueue(struct srcu_struct *ssp, struct rcu_head *rhp, + rcu_callback_t func) { unsigned long flags; @@ -231,6 +246,68 @@ void call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp, srcu_gp_start_if_needed(ssp); preempt_enable(); } + +/* + * Set only by the irq_work drain, the one drain its own re-issue can re-feed; + * a callback staged during a direct drain is taken by ->defer_iw afterwards. + * Global rather than per-srcu_struct: a re-entrant call_srcu(B) inside a drain + * of A raises B's own ->defer_iw, whose drain can stage back onto A. + */ +static bool srcu_defer_draining; + +static void __srcu_defer_drain(struct srcu_struct *ssp) +{ + struct llist_node *node, *next; + unsigned long flags; + + if (!IS_ENABLED(CONFIG_RCU_DEFER)) + return; + + /* Re-issued newest-first; nothing depends on call_srcu() ordering. */ + local_irq_save(flags); + llist_for_each_safe(node, next, llist_del_all(&ssp->defer_cbs)) { + struct rcu_head *rhp = (struct rcu_head *)node; + + srcu_do_enqueue(ssp, rhp, rhp->func); + } + local_irq_restore(flags); +} + +/* Only the irq_work drain can be re-fed by its own re-issue; see Tree SRCU. */ +void srcu_defer_drain(struct irq_work *iw) +{ + struct srcu_struct *ssp = container_of(iw, struct srcu_struct, defer_iw); + + WRITE_ONCE(srcu_defer_draining, true); + __srcu_defer_drain(ssp); + WRITE_ONCE(srcu_defer_draining, false); +} +EXPORT_SYMBOL_GPL(srcu_defer_drain); + +void call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp, + rcu_callback_t func) +{ + if (should_rcu_defer()) { + /* A re-entrant call_srcu() during the drain would livelock it. */ + if (READ_ONCE(srcu_defer_draining) && !in_nmi()) { + WARN_ONCE(IS_ENABLED(CONFIG_PROVE_RCU), + "call_srcu() re-entered during callback drain; leaking callback\n"); + return; + } + rhp->func = func; + if (llist_add((struct llist_node *)rhp, &ssp->defer_cbs)) + irq_work_queue(&ssp->defer_iw); + return; + } + + /* + * Only reachable from an NMI when deferral is off: before the scheduler + * is up, or with CONFIG_RCU_DEFER=n. The enqueue can then race. + */ + WARN_ON_ONCE(IS_ENABLED(CONFIG_PROVE_RCU) && in_nmi()); + + srcu_do_enqueue(ssp, rhp, func); +} EXPORT_SYMBOL_GPL(call_srcu); /* @@ -260,6 +337,14 @@ void synchronize_srcu(struct srcu_struct *ssp) } EXPORT_SYMBOL_GPL(synchronize_srcu); +/* Register any deferred callbacks, then wait for all in-flight ones. */ +void srcu_barrier(struct srcu_struct *ssp) +{ + __srcu_defer_drain(ssp); + synchronize_srcu(ssp); +} +EXPORT_SYMBOL_GPL(srcu_barrier); + /* * get_state_synchronize_srcu - Provide an end-of-grace-period cookie */ From ebaffa99c89ca826a9082678b4759e205645b3de Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Wed, 5 Aug 2026 13:18:58 -0700 Subject: [PATCH 0083/1012] rcutorture: Disable fragile readers during overload testing Vanilla RCU readers, when either non-preemptible or preemptible with priority-boosting enabled, can handle heavy overload conditions. Other RCU implementations are more fragile. This commit adds an rcu_torture_ops structure field named ->rdrs_handle_load that is set for non-fragile configurations of vanilla RCU, but cleared otherwise. It also refrains from running fragile readers concurrently with overload testing, thus avoiding false-positive overload-testing failures. Signed-off-by: Paul E. McKenney --- kernel/rcu/rcutorture.c | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/kernel/rcu/rcutorture.c b/kernel/rcu/rcutorture.c index 794937e13e7c32..7e08be857f01fa 100644 --- a/kernel/rcu/rcutorture.c +++ b/kernel/rcu/rcutorture.c @@ -440,6 +440,7 @@ struct rcu_torture_ops { int debug_objects; int start_poll_irqsoff; int have_up_down; + int rdrs_handle_load; const char *name; }; @@ -648,6 +649,7 @@ static struct rcu_torture_ops rcu_ops = { .extendables = RCUTORTURE_MAX_EXTEND, .debug_objects = 1, .start_poll_irqsoff = 1, + .rdrs_handle_load = !IS_ENABLED(CONFIG_PREEMPT_RCU) || IS_ENABLED(CONFIG_RCU_BOOST), .name = "rcu" }; @@ -2534,8 +2536,10 @@ static bool rcu_torture_one_read_start(struct rcu_torture_one_read_state *rtorsp rtorsp->p = rcu_dereference_check(rcu_torture_current, !cur_ops->readlock_held || cur_ops->readlock_held() || (rtorsp->readstate & RCUTORTURE_RDR_UPDOWN)); - if (rtorsp->p == NULL) { - /* Wait for rcu_torture_writer to get underway */ + if ((!cur_ops->rdrs_handle_load && atomic_read(&rcu_fwd_cb_nodelay)) || rtorsp->p == NULL) { + // Wait for rcu_torture_writer to get underway and + // (if readers cannot handle heavy loads) for any + // forward-progress testing to complete. rcutorture_one_extend(&rtorsp->readstate, 0, trsp, rtorsp->rtrsp); return false; } From 619f531061a0db1341d3079e4a241d73892d9a75 Mon Sep 17 00:00:00 2001 From: Puranjay Mohan Date: Mon, 10 Aug 2026 05:27:54 -0700 Subject: [PATCH 0084/1012] rcutorture: Exercise ->call() from NMI context call_rcu() and call_srcu() are now safe to invoke from NMI, but rcutorture never does, leaving the deferral path untested. Add an ->nmi_capable flag to rcu_torture_ops. For flavors that set it, arm a per-CPU hardware perf counter whose overflow handler submits a callback via ->call(). The handler acts only when in_nmi(), so only a genuine NMI exercises the deferral path. One preallocated callback per CPU is kept in flight, guarded by an atomic, to avoid allocating in NMI. The counter uses a fixed sample period rather than a frequency: a frequency-based event sets TICK_DEP_BIT_PERF_EVENTS and would pin the tick for the whole run on NO_HZ_FULL kernels. Report the count issued from NMI ("nmi-calls:") and the count invoked ("nmi-cbs:"). rcu_torture_cleanup() disables the counters and then calls cb_barrier(), which drains every deferred callback, so the two counts must then match; a mismatch fails the test. This relies on rcu_barrier()/srcu_barrier() flushing deferred callbacks, as added earlier in the series. Set ->nmi_capable on the NMI-safe flavors: rcu, srcu, srcud, and tasks-tracing (call_srcu() under the hood). Tasks and Tasks Rude are left alone, as call_rcu_tasks_generic() is not yet NMI-safe. Enabled by default; the nmi_calls parameter disables it, which helps rule NMI handling in or out when triaging a failure. Requires CONFIG_PERF_EVENTS and a hardware PMU: without one nothing is issued from NMI and the end-of-test check compares zero against zero, so a pass does not by itself mean the path ran. That is the case under kvm.sh, which boots qemu with -cpu kvm64 and no vPMU; rcu_torture_nmi_cleanup() says so on the console. Signed-off-by: Puranjay Mohan Signed-off-by: Paul E. McKenney --- .../admin-guide/kernel-parameters.txt | 7 + kernel/rcu/rcutorture.c | 150 +++++++++++++++++- 2 files changed, 155 insertions(+), 2 deletions(-) diff --git a/Documentation/admin-guide/kernel-parameters.txt b/Documentation/admin-guide/kernel-parameters.txt index 68647ff4bdd24b..fd9acc3fd9b7a4 100644 --- a/Documentation/admin-guide/kernel-parameters.txt +++ b/Documentation/admin-guide/kernel-parameters.txt @@ -6173,6 +6173,13 @@ Kernel parameters stress RCU, they don't participate in the actual test, hence the "fake". + rcutorture.nmi_calls= [KNL] + Enable issuing RCU callbacks from an NMI, on the + RCU flavors that support it, to exercise the + any-context callback path. Requires + CONFIG_PERF_EVENTS and a hardware PMU; without + both, nothing is issued. Defaults to enabled. + rcutorture.nocbs_nthreads= [KNL] Set number of RCU callback-offload togglers. Zero (the default) disables toggling. diff --git a/kernel/rcu/rcutorture.c b/kernel/rcu/rcutorture.c index 7e08be857f01fa..4d1be2a49f011b 100644 --- a/kernel/rcu/rcutorture.c +++ b/kernel/rcu/rcutorture.c @@ -48,6 +48,7 @@ #include #include #include +#include #include "rcu.h" @@ -115,6 +116,7 @@ torture_param(int, leakpointer, 0, "Leak pointer dereferences from readers"); torture_param(int, n_barrier_cbs, 0, "# of callbacks/kthreads for barrier testing"); torture_param(int, n_up_down, 32, "# of concurrent up/down hrtimer-based RCU readers"); torture_param(int, nfakewriters, 4, "Number of RCU fake writer threads"); +torture_param(bool, nmi_calls, true, "Exercise ->call() from NMI on nmi_capable flavors"); torture_param(int, nreaders, -1, "Number of RCU reader threads"); torture_param(bool, nwriters, 1, "Number of RCU writer threads (0 or 1)"); torture_param(int, object_debug, 0, "Enable debug-object double call_rcu() testing"); @@ -216,6 +218,8 @@ static long n_rcu_torture_boost_failure; static long n_rcu_torture_boosts; static atomic_long_t n_rcu_torture_timers; static atomic_long_t n_rcu_torture_irqs; +static atomic_long_t n_rcu_torture_nmi_call; +static atomic_long_t n_rcu_torture_nmi_cb; static long n_barrier_attempts; static long n_barrier_successes; /* did rcu_barrier test succeed? */ static unsigned long n_read_exits; @@ -433,6 +437,7 @@ struct rcu_torture_ops { bool (*is_task_rcu_boosted)(void); long cbflood_max; int irq_capable; + int nmi_capable; int can_boost; int extendables; int slow_gps; @@ -650,6 +655,7 @@ static struct rcu_torture_ops rcu_ops = { .debug_objects = 1, .start_poll_irqsoff = 1, .rdrs_handle_load = !IS_ENABLED(CONFIG_PREEMPT_RCU) || IS_ENABLED(CONFIG_RCU_BOOST), + .nmi_capable = 1, .name = "rcu" }; @@ -944,6 +950,7 @@ static struct rcu_torture_ops srcu_ops = { .debug_objects = 1, .have_up_down = IS_ENABLED(CONFIG_TINY_SRCU) ? 0 : SRCU_READ_FLAVOR_NORMAL | SRCU_READ_FLAVOR_FAST_UPDOWN, + .nmi_capable = 1, .name = "srcu" }; @@ -1007,6 +1014,7 @@ static struct rcu_torture_ops srcud_ops = { .debug_objects = 1, .have_up_down = IS_ENABLED(CONFIG_TINY_SRCU) ? 0 : SRCU_READ_FLAVOR_NORMAL | SRCU_READ_FLAVOR_FAST_UPDOWN, + .nmi_capable = 1, .name = "srcud" }; @@ -1271,6 +1279,7 @@ static struct rcu_torture_ops tasks_tracing_ops = { .cbflood_max = 50000, .irq_capable = 1, .slow_gps = 1, + .nmi_capable = 1, .name = "tasks-tracing" }; @@ -2663,6 +2672,124 @@ static bool rcu_torture_one_read(struct torture_random_state *trsp, long myid) static DEFINE_TORTURE_RANDOM_PERCPU(rcu_torture_timer_rand); +/* + * Exercise ->call() from NMI context for flavors that set ->nmi_capable. A + * per-CPU hardware perf counter overflows into an NMI, and its handler submits + * a preallocated callback via ->call(). One callback per CPU is in flight at a + * time (guarded by an atomic) to avoid allocating in NMI. + */ +#ifdef CONFIG_PERF_EVENTS +static struct perf_event_attr rcu_torture_nmi_attr = { + .type = PERF_TYPE_HARDWARE, + .config = PERF_COUNT_HW_CPU_CYCLES, + .size = sizeof(struct perf_event_attr), + .pinned = 1, + .disabled = 1, + /* + * A fixed period rather than .freq: a frequency-based event bumps + * nr_freq_events, which sets TICK_DEP_BIT_PERF_EVENTS and would pin the + * tick for the whole run on NO_HZ_FULL kernels. + */ + .sample_period = 20 * 1000 * 1000, +}; + +/* One in-flight callback per CPU; ->inuse is released by the callback. */ +struct rcu_torture_nmi_cb { + struct rcu_head rh; + atomic_t inuse; +}; + +static struct perf_event **rcu_torture_nmi_events; +static int rcu_torture_nmi_hp_state; +static DEFINE_PER_CPU(struct rcu_torture_nmi_cb, rcu_torture_nmi_cb); + +static void rcu_torture_nmi_invoked(struct rcu_head *rhp) +{ + struct rcu_torture_nmi_cb *rtncp = container_of(rhp, struct rcu_torture_nmi_cb, rh); + + atomic_long_inc(&n_rcu_torture_nmi_cb); + atomic_set(&rtncp->inuse, 0); +} + +static void rcu_torture_nmi_overflow(struct perf_event *event, + struct perf_sample_data *data, + struct pt_regs *regs) +{ + struct rcu_torture_nmi_cb *rtncp = this_cpu_ptr(&rcu_torture_nmi_cb); + + if (!in_nmi()) + return; + if (cur_ops->call && !atomic_xchg(&rtncp->inuse, 1)) { + atomic_long_inc(&n_rcu_torture_nmi_call); + cur_ops->call(&rtncp->rh, rcu_torture_nmi_invoked); + } +} + +static int rcu_torture_nmi_online(unsigned int cpu) +{ + struct perf_event *event; + + event = perf_event_create_kernel_counter(&rcu_torture_nmi_attr, cpu, NULL, + rcu_torture_nmi_overflow, NULL); + if (IS_ERR(event)) + return 0; + rcu_torture_nmi_events[cpu] = event; + perf_event_enable(event); + return 0; +} + +static int rcu_torture_nmi_offline(unsigned int cpu) +{ + struct perf_event *event = rcu_torture_nmi_events[cpu]; + + if (event) { + rcu_torture_nmi_events[cpu] = NULL; + perf_event_disable(event); + perf_event_release_kernel(event); + } + return 0; +} + +/* Drive the counters from hotplug callbacks so coverage survives onoff. */ +static void rcu_torture_nmi_init(void) +{ + int ret; + + if (!nmi_calls || !cur_ops->nmi_capable || !cur_ops->call) + return; + rcu_torture_nmi_events = kcalloc(nr_cpu_ids, sizeof(*rcu_torture_nmi_events), + GFP_KERNEL); + if (!rcu_torture_nmi_events) + return; + ret = cpuhp_setup_state(CPUHP_AP_ONLINE_DYN, "rcutorture/nmi:online", + rcu_torture_nmi_online, rcu_torture_nmi_offline); + if (ret < 0) { + kfree(rcu_torture_nmi_events); + rcu_torture_nmi_events = NULL; + return; + } + rcu_torture_nmi_hp_state = ret; +} + +static void rcu_torture_nmi_cleanup(void) +{ + if (!rcu_torture_nmi_events) + return; + if (rcu_torture_nmi_hp_state > 0) { + cpuhp_remove_state(rcu_torture_nmi_hp_state); + rcu_torture_nmi_hp_state = 0; + } + kfree(rcu_torture_nmi_events); + rcu_torture_nmi_events = NULL; + if (!atomic_long_read(&n_rcu_torture_nmi_call)) + pr_alert("%s: nmi_calls set but no ->call() ever issued from NMI, so NMI ->call() went untested (no PMU, or NMIs unavailable here).\n", + __func__); +} +#else /* #ifdef CONFIG_PERF_EVENTS */ +static void rcu_torture_nmi_init(void) { } +static void rcu_torture_nmi_cleanup(void) { } +#endif /* #else #ifdef CONFIG_PERF_EVENTS */ + /* * RCU torture reader from timer handler. Dereferences rcu_torture_current, * incrementing the corresponding element of the pipeline array. The @@ -3051,6 +3178,9 @@ rcu_torture_stats_print(void) data_race(n_barrier_attempts), data_race(n_rcu_torture_barrier_error)); pr_cont("read-exits: %ld ", data_race(n_read_exits)); // Statistic. + pr_cont("nmi-calls: %ld nmi-cbs: %ld ", + atomic_long_read(&n_rcu_torture_nmi_call), + atomic_long_read(&n_rcu_torture_nmi_cb)); pr_cont("nocb-toggles: %ld:%ld ", atomic_long_read(&n_nocb_offload), atomic_long_read(&n_nocb_deoffload)); pr_cont("gpwraps: %ld\n", n_gpwraps); @@ -3199,7 +3329,7 @@ rcu_torture_print_module_parms(struct rcu_torture_ops *cur_ops, const char *tag) "read_exit_delay=%d read_exit_burst=%d " "reader_flavor=%x " "nocbs_nthreads=%d nocbs_toggle=%d " - "test_nmis=%d " + "test_nmis=%d nmi_calls=%d " "preempt_duration=%d preempt_interval=%d n_up_down=%d\n", torture_type, tag, nrealreaders, nwriters, nrealfakewriters, stat_interval, verbose, test_no_idle_hz, shuffle_interval, @@ -3213,7 +3343,7 @@ rcu_torture_print_module_parms(struct rcu_torture_ops *cur_ops, const char *tag) read_exit_delay, read_exit_burst, reader_flavor, nocbs_nthreads, nocbs_toggle, - test_nmis, + test_nmis, nmi_calls, preempt_duration, preempt_interval, n_up_down); } @@ -4287,6 +4417,7 @@ rcu_torture_cleanup(void) int i; if (torture_cleanup_begin()) { + rcu_torture_nmi_cleanup(); if (cur_ops->cb_barrier != NULL) { pr_info("%s: Invoking %pS().\n", __func__, cur_ops->cb_barrier); cur_ops->cb_barrier(); @@ -4329,6 +4460,7 @@ rcu_torture_cleanup(void) kfree(reader_tasks); reader_tasks = NULL; } + rcu_torture_nmi_cleanup(); kfree(rcu_torture_reader_mbchk); rcu_torture_reader_mbchk = NULL; @@ -4358,6 +4490,19 @@ rcu_torture_cleanup(void) pr_info("%s: Invoking %pS().\n", __func__, cur_ops->cb_barrier); cur_ops->cb_barrier(); } + + /* + * cb_barrier() above drained every deferred callback, so the count + * issued from NMI must equal the count invoked. + */ + if (atomic_long_read(&n_rcu_torture_nmi_call) != + atomic_long_read(&n_rcu_torture_nmi_cb)) { + pr_alert("%s: NMI ->call() lost a callback: issued %ld invoked %ld\n", + __func__, atomic_long_read(&n_rcu_torture_nmi_call), + atomic_long_read(&n_rcu_torture_nmi_cb)); + atomic_inc(&n_rcu_torture_error); + } + if (cur_ops->cleanup != NULL) cur_ops->cleanup(); @@ -4790,6 +4935,7 @@ rcu_torture_init(void) firsterr = -ENOMEM; goto unwind; } + rcu_torture_nmi_init(); for (i = 0; i < nrealreaders; i++) { rcu_torture_reader_mbchk[i].rtc_chkrdr = -1; firsterr = torture_create_kthread(rcu_torture_reader, (void *)i, From a964c4bd9e7e5f65fdb2ea5a1725b44f6db18141 Mon Sep 17 00:00:00 2001 From: Puranjay Mohan Date: Mon, 10 Aug 2026 05:27:55 -0700 Subject: [PATCH 0085/1012] selftests/bpf: Add a call_srcu() re-entry reproducer Re-enter call_srcu() from a BPF program to exercise its any-context safety, via call_rcu_tasks_trace(), which is call_srcu() on rcu_tasks_trace_srcu_struct. An fentry program on rcu_segcblist_enqueue() fires mid-enqueue: that function is reached from srcu_gp_start_if_needed() with the srcu_data ->lock held. The program does a task-storage delete, whose only deferred work is call_rcu_tasks_trace(), re-entering the enqueue on the same CPU. The handler matches on TID and fires once; pinning the thread removes the migration window between picking the srcu_data and taking its lock. Without the fix the nested call re-takes the same sdp lock and self-deadlocks; with it the nested __call_srcu() sees interrupts disabled and defers via irq_work, so the delete returns and the test passes. The test skips where it does not apply: Tiny RCU has no rcu_segcblist_enqueue() to attach to, and a UP+PREEMPT kernel pairs Tree RCU with Tiny SRCU, so the attach succeeds but call_srcu() never reaches the enqueue. Tiny SRCU is told apart by srcu_expedite_current(), which it stubs out, so on Tree SRCU a zero hit count fails rather than skips and the reproducer cannot quietly stop reproducing. Signed-off-by: Puranjay Mohan Acked-by: Kumar Kartikeya Dwivedi Signed-off-by: Paul E. McKenney --- .../selftests/bpf/prog_tests/rcu_reentry.c | 93 +++++++++++++++++++ .../testing/selftests/bpf/progs/rcu_reentry.c | 51 ++++++++++ 2 files changed, 144 insertions(+) create mode 100644 tools/testing/selftests/bpf/prog_tests/rcu_reentry.c create mode 100644 tools/testing/selftests/bpf/progs/rcu_reentry.c diff --git a/tools/testing/selftests/bpf/prog_tests/rcu_reentry.c b/tools/testing/selftests/bpf/prog_tests/rcu_reentry.c new file mode 100644 index 00000000000000..de23a14b3d408f --- /dev/null +++ b/tools/testing/selftests/bpf/prog_tests/rcu_reentry.c @@ -0,0 +1,93 @@ +// SPDX-License-Identifier: GPL-2.0 +/* Exercise re-entry into call_srcu() from BPF; see progs/rcu_reentry.c. */ +#define _GNU_SOURCE +#include +#include +#include "task_local_storage_helpers.h" +#include "trace_helpers.h" +#include "rcu_reentry.skel.h" + +/* Tiny RCU has no rcu_segcblist_enqueue() to attach to. */ +static bool have_attach_target(void) +{ + unsigned long long addr; + + return kallsyms_find("rcu_segcblist_enqueue", &addr) == 0; +} + +/* Tiny SRCU stubs out srcu_expedite_current(); Tree SRCU exports it. */ +static bool have_tree_srcu(void) +{ + unsigned long long addr; + + return kallsyms_find("srcu_expedite_current", &addr) == 0; +} + +void test_rcu_reentry(void) +{ + struct rcu_reentry *skel; + int err, pidfd = -1, map_fd; + cpu_set_t set, old_set; + bool affinity_saved; + __u64 val = 1; + int cpu; + + if (!have_attach_target()) { + test__skip(); + return; + } + + skel = rcu_reentry__open_and_load(); + if (!ASSERT_OK_PTR(skel, "skel_open_and_load")) + return; + + err = rcu_reentry__attach(skel); + if (!ASSERT_OK(err, "skel_attach")) + goto out; + + /* Keep the re-entry on a single CPU; a cpuset may exclude CPU 0. */ + affinity_saved = !sched_getaffinity(0, sizeof(old_set), &old_set); + cpu = sched_getcpu(); + if (!ASSERT_GE(cpu, 0, "getcpu")) + goto out; + CPU_ZERO(&set); + CPU_SET(cpu, &set); + if (!ASSERT_OK(sched_setaffinity(0, sizeof(set), &set), "setaffinity")) + goto out; + + pidfd = sys_pidfd_open(getpid(), 0); + if (!ASSERT_GE(pidfd, 0, "pidfd_open")) + goto restore; + map_fd = bpf_map__fd(skel->maps.task_stg); + err = bpf_map_update_elem(map_fd, &pidfd, &val, BPF_NOEXIST); + if (!ASSERT_OK(err, "boot_create")) + goto restore; + + /* Arm the handler for this thread, then trigger call_rcu_tasks_trace(). */ + skel->bss->target_pid = syscall(__NR_gettid); + err = bpf_map_delete_elem(map_fd, &pidfd); + if (!ASSERT_OK(err, "boot_delete")) + goto restore; + + /* + * Only Tree SRCU reaches rcu_segcblist_enqueue() from call_srcu(); a + * UP+PREEMPT kernel pairs Tree RCU with Tiny SRCU, so the attach + * succeeds but nothing fires. On Tree SRCU it must fire. + */ + if (!skel->bss->hits) { + if (have_tree_srcu()) + ASSERT_GT(skel->bss->hits, 0, "prog_fired"); + else + test__skip(); + goto restore; + } + ASSERT_EQ(skel->bss->get_errs, 0, "nested_storage_get"); + ASSERT_EQ(skel->bss->del_errs, 0, "nested_storage_delete"); +restore: + if (affinity_saved) + sched_setaffinity(0, sizeof(old_set), &old_set); +out: + if (pidfd >= 0) + close(pidfd); + rcu_reentry__destroy(skel); +} diff --git a/tools/testing/selftests/bpf/progs/rcu_reentry.c b/tools/testing/selftests/bpf/progs/rcu_reentry.c new file mode 100644 index 00000000000000..47a36f704cf3e1 --- /dev/null +++ b/tools/testing/selftests/bpf/progs/rcu_reentry.c @@ -0,0 +1,51 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Re-enter call_srcu() from a BPF program. fentry on rcu_segcblist_enqueue() + * fires inside call_srcu()'s enqueue (reached from srcu_gp_start_if_needed() + * with the srcu_data ->lock held); the handler then calls call_rcu_tasks_trace() + * -- itself call_srcu() on rcu_tasks_trace_srcu_struct -- re-entering the same + * srcu_data on the same CPU. + */ +#include "vmlinux.h" +#include +#include + +char _license[] SEC("license") = "GPL"; + +struct { + __uint(type, BPF_MAP_TYPE_TASK_STORAGE); + __uint(map_flags, BPF_F_NO_PREALLOC); + __type(key, int); + __type(value, __u64); +} task_stg SEC(".maps"); + +int target_pid; +int hits; +int get_errs; +int del_errs; +int done; + +SEC("fentry/rcu_segcblist_enqueue") +int BPF_PROG(reenter) +{ + struct task_struct *cur; + + if (done || !target_pid) + return 0; + + cur = bpf_get_current_task_btf(); + if (cur->pid != target_pid) + return 0; + + /* Issue the nested call exactly once, so the test is deterministic. */ + done = 1; + __sync_fetch_and_add(&hits, 1); + + /* Re-enter via a task-storage delete, which calls call_rcu_tasks_trace(). */ + if (!bpf_task_storage_get(&task_stg, cur, 0, BPF_LOCAL_STORAGE_GET_F_CREATE)) + __sync_fetch_and_add(&get_errs, 1); + else if (bpf_task_storage_delete(&task_stg, cur)) + __sync_fetch_and_add(&del_errs, 1); + + return 0; +} From c19a50f60e61a12a7a8521609dc30a2698cd920b Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Thu, 9 Jul 2026 09:36:34 -0700 Subject: [PATCH 0086/1012] srcutiny: Make a Tiny SRCU grace period imply an RCU grace period In non-preemptible kernels, a Tiny SRCU grace period implies an RCU grace period because any context switch suffices. But in preemptible kernels, it is possible for a Tiny SRCU grace period to elapse without a corresponding RCU grace period. Which was OK until RCU Tasks Trace was re-implemented in terms of SRCU-fast, which in Tiny SRCU is implemented as SRCU, which is already fast. And RCU Tasks Trace grace periods are required to imply RCU grace periods. This commit therefore adds a synchronize_rcu(), but only in preemptible kernels. Because preemptible Tiny SRCU is not in mainline, this added call to synchronize_rcu() will not slow anything down: The comparison would instead be with TREE SRCU. But if this added call ever becomes a problem, the Tiny SRCU srcu_struct structure could track whether or not this is for SRCU-fast, and to add the synchronize_rcu() only in the SRCU-fast case. However, at the moment, this is seen as unnecessary complexity. Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutiny.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/kernel/rcu/srcutiny.c b/kernel/rcu/srcutiny.c index 5de9a690583835..32b37d63d58aa4 100644 --- a/kernel/rcu/srcutiny.c +++ b/kernel/rcu/srcutiny.c @@ -159,6 +159,8 @@ void srcu_drive_gp(struct work_struct *wp) WRITE_ONCE(ssp->srcu_idx, ssp->srcu_idx + 1); WRITE_ONCE(ssp->srcu_gp_waiting, true); /* srcu_read_unlock() wakes! */ preempt_enable(); + if (IS_ENABLED(CONFIG_PREEMPTION)) + synchronize_rcu(); // Needed for RCU Tasks Trace to imply RCU grace period do { // Deadlock issues prevent __srcu_read_unlock() from // doing an unconditional wakeup, so polling is required. From b3b1ea477db5a907c0c1a6231dc0f993a161dcac Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Tue, 18 Aug 2026 10:51:42 -0700 Subject: [PATCH 0087/1012] srcutree: Suppress srcu_advance_state() mutex_lock in atomic The new SRCU_READ_FLAVOR_ATOMIC prohibits sleeping in both readers and grace-period waits in order to permit waiting on an SRCU grace period within an OOM notifier. This means that srcu_advance_state() cannot acquire ->srcu_gp_mutex in this case, because mutex_lock() can sleep. This commit therefore adds an is_atomic parameter to srcu_advance_state(). When this parameter is false, current behavior is maintained, in other words, ->srcu_gp_mutex is acquired and released as before. But when this is_atomic parameter is false, the caller is responsible for excluding other callers. Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutree.c | 23 +++++++++++++++-------- 1 file changed, 15 insertions(+), 8 deletions(-) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index d32c374ee72af8..dc063eb49b0dc7 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -1941,13 +1941,16 @@ EXPORT_SYMBOL_GPL(srcu_batches_completed); /* * Core SRCU state machine. Push state bits of ->srcu_gp_seq * to SRCU_STATE_SCAN2, and invoke srcu_gp_end() when scan has - * completed in that state. + * completed in that state. Set is_atomic to indicate that + * the caller is excluding other calls so that ->srcu_gp_mutex + * is not needed, and to indicate that blocking is forbidden. */ -static void srcu_advance_state(struct srcu_struct *ssp) +static void srcu_advance_state(struct srcu_struct *ssp, bool is_atomic) { int idx; - mutex_lock(&ssp->srcu_sup->srcu_gp_mutex); + if (!is_atomic) + mutex_lock(&ssp->srcu_sup->srcu_gp_mutex); /* * Because readers might be delayed for an extended period after @@ -1965,7 +1968,8 @@ static void srcu_advance_state(struct srcu_struct *ssp) if (ULONG_CMP_GE(ssp->srcu_sup->srcu_gp_seq, ssp->srcu_sup->srcu_gp_seq_needed)) { WARN_ON_ONCE(rcu_seq_state(ssp->srcu_sup->srcu_gp_seq)); raw_spin_unlock_irq_rcu_node(ssp->srcu_sup); - mutex_unlock(&ssp->srcu_sup->srcu_gp_mutex); + if (!is_atomic) + mutex_unlock(&ssp->srcu_sup->srcu_gp_mutex); return; } idx = rcu_seq_state(READ_ONCE(ssp->srcu_sup->srcu_gp_seq)); @@ -1973,7 +1977,8 @@ static void srcu_advance_state(struct srcu_struct *ssp) srcu_gp_start(ssp); raw_spin_unlock_irq_rcu_node(ssp->srcu_sup); if (idx != SRCU_STATE_IDLE) { - mutex_unlock(&ssp->srcu_sup->srcu_gp_mutex); + if (!is_atomic) + mutex_unlock(&ssp->srcu_sup->srcu_gp_mutex); return; /* Someone else started the grace period. */ } } @@ -1981,7 +1986,8 @@ static void srcu_advance_state(struct srcu_struct *ssp) if (rcu_seq_state(READ_ONCE(ssp->srcu_sup->srcu_gp_seq)) == SRCU_STATE_SCAN1) { idx = !(ssp->srcu_ctrp - &ssp->sda->srcu_ctrs[0]); if (!try_check_zero(ssp, idx, 1)) { - mutex_unlock(&ssp->srcu_sup->srcu_gp_mutex); + if (!is_atomic) + mutex_unlock(&ssp->srcu_sup->srcu_gp_mutex); return; /* readers present, retry later. */ } srcu_flip(ssp); @@ -1999,7 +2005,8 @@ static void srcu_advance_state(struct srcu_struct *ssp) */ idx = !(ssp->srcu_ctrp - &ssp->sda->srcu_ctrs[0]); if (!try_check_zero(ssp, idx, 2)) { - mutex_unlock(&ssp->srcu_sup->srcu_gp_mutex); + if (!is_atomic) + mutex_unlock(&ssp->srcu_sup->srcu_gp_mutex); return; /* readers present, retry later. */ } ssp->srcu_sup->srcu_n_exp_nodelay = 0; @@ -2107,7 +2114,7 @@ static void process_srcu(struct work_struct *work) sup = container_of(work, struct srcu_usage, work.work); ssp = sup->srcu_ssp; - srcu_advance_state(ssp); + srcu_advance_state(ssp, false); raw_spin_lock_irq_rcu_node(ssp->srcu_sup); curdelay = srcu_get_delay(ssp); raw_spin_unlock_irq_rcu_node(ssp->srcu_sup); From ffeaead6183bcbc3496ad2a16eb4e1890d24d7fc Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Tue, 18 Aug 2026 11:41:01 -0700 Subject: [PATCH 0088/1012] srcutree: Suppress to-big transition for atomic SRCU Initially (and perhaps forever), call_srcu() will not be available for atomic srcu_struct structures. There is therefore no reason to transition such a structure to big, because the main purpose of such a transition is to reduce lock contention for concurrent SRCU callback queueing. This commit therefore adds an is_atomic parameter to both the check_init_srcu_struct() and init_srcu_struct_fields() functions, which suppresses the initialization-time transition to big that is enabled by default on large systems. It will still be possible to force a transition using rcutorture as a destructive test. This might (or might not) be adjusted later. [ paulmck: Apply feedback from Kunwu Chan. ] Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutree.c | 31 ++++++++++++++++++------------- 1 file changed, 18 insertions(+), 13 deletions(-) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index dc063eb49b0dc7..c611a7168c7009 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -238,8 +238,10 @@ static bool init_srcu_struct_nodes(struct srcu_struct *ssp, gfp_t gfp_flags) * Initialize non-compile-time initialized fields, including the * associated srcu_node and srcu_data structures. The is_static parameter * tells us that ->sda has already been wired up to srcu_data. + * The is_atomic parameter tells us that there is no reason to + * ever transition to big. */ -static int init_srcu_struct_fields(struct srcu_struct *ssp, bool is_static) +static int init_srcu_struct_fields(struct srcu_struct *ssp, bool is_static, bool is_atomic) { if (!is_static) ssp->srcu_sup = kzalloc_obj(*ssp->srcu_sup); @@ -267,7 +269,8 @@ static int init_srcu_struct_fields(struct srcu_struct *ssp, bool is_static) init_srcu_struct_data(ssp); ssp->srcu_sup->srcu_gp_seq_needed_exp = SRCU_GP_SEQ_INITIAL_VAL; ssp->srcu_sup->srcu_last_gp_end = ktime_get_mono_fast_ns(); - if (READ_ONCE(ssp->srcu_sup->srcu_size_state) == SRCU_SIZE_SMALL && SRCU_SIZING_IS_INIT()) { + if (!is_atomic && + READ_ONCE(ssp->srcu_sup->srcu_size_state) == SRCU_SIZE_SMALL && SRCU_SIZING_IS_INIT()) { if (!preemptible()) WRITE_ONCE(ssp->srcu_sup->srcu_size_state, SRCU_SIZE_ALLOC); else if (init_srcu_struct_nodes(ssp, GFP_KERNEL)) @@ -301,7 +304,7 @@ __init_srcu_struct_common(struct srcu_struct *ssp, const char *name, struct lock /* Don't re-initialize a lock while it is held. */ debug_check_no_locks_freed((void *)ssp, sizeof(*ssp)); lockdep_init_map(&ssp->dep_map, name, key, 0); - return init_srcu_struct_fields(ssp, false); + return init_srcu_struct_fields(ssp, false, false); } int init_srcu_struct_lockdep(struct srcu_struct *ssp, const char *name, @@ -343,7 +346,7 @@ EXPORT_SYMBOL_GPL(__init_srcu_struct_fast_updown); int init_srcu_struct_generic(struct srcu_struct *ssp) { ssp->srcu_reader_flavor = 0; - return init_srcu_struct_fields(ssp, false); + return init_srcu_struct_fields(ssp, false, false); } EXPORT_SYMBOL_GPL(init_srcu_struct_generic); @@ -360,7 +363,7 @@ EXPORT_SYMBOL_GPL(init_srcu_struct_generic); int init_srcu_struct_fast(struct srcu_struct *ssp) { ssp->srcu_reader_flavor = SRCU_READ_FLAVOR_FAST; - return init_srcu_struct_fields(ssp, false); + return init_srcu_struct_fields(ssp, false, false); } EXPORT_SYMBOL_GPL(init_srcu_struct_fast); @@ -378,7 +381,7 @@ EXPORT_SYMBOL_GPL(init_srcu_struct_fast); int init_srcu_struct_fast_updown(struct srcu_struct *ssp) { ssp->srcu_reader_flavor = SRCU_READ_FLAVOR_FAST_UPDOWN; - return init_srcu_struct_fields(ssp, false); + return init_srcu_struct_fields(ssp, false, false); } EXPORT_SYMBOL_GPL(init_srcu_struct_fast_updown); @@ -470,9 +473,11 @@ static void raw_spin_lock_irqsave_ssp_contention(struct srcu_struct *ssp, unsign * done with compile-time initialization, so this check is added * to each update-side SRCU primitive. Use ssp->lock, which -is- * compile-time initialized, to resolve races involving multiple - * CPUs trying to garner first-use privileges. + * CPUs trying to garner first-use privileges. The is_atomic + * parameter tells us that there will never be a reason to + * transition to big. */ -static void check_init_srcu_struct(struct srcu_struct *ssp) +static void check_init_srcu_struct(struct srcu_struct *ssp, bool is_atomic) { unsigned long flags; @@ -484,7 +489,7 @@ static void check_init_srcu_struct(struct srcu_struct *ssp) raw_spin_unlock_irqrestore_rcu_node(ssp->srcu_sup, flags); return; } - init_srcu_struct_fields(ssp, true); + init_srcu_struct_fields(ssp, true, is_atomic); raw_spin_unlock_irqrestore_rcu_node(ssp->srcu_sup, flags); } @@ -1279,7 +1284,7 @@ static bool srcu_should_expedite(struct srcu_struct *ssp) unsigned long t; unsigned long tlast; - check_init_srcu_struct(ssp); + check_init_srcu_struct(ssp, false); /* If _lite() readers, don't do unsolicited expediting. */ if (this_cpu_read(ssp->sda->srcu_reader_flavor) & SRCU_READ_FLAVOR_SLOWGP) return false; @@ -1338,7 +1343,7 @@ static unsigned long srcu_gp_start_if_needed(struct srcu_struct *ssp, struct srcu_node *sdp_mynode; int ss_state; - check_init_srcu_struct(ssp); + check_init_srcu_struct(ssp, false); /* * While starting a new grace period, make sure we are in an * SRCU read-side critical section so that the grace-period @@ -1617,7 +1622,7 @@ static void __synchronize_srcu(struct srcu_struct *ssp, bool do_norm) if (rcu_scheduler_active == RCU_SCHEDULER_INACTIVE) return; might_sleep(); - check_init_srcu_struct(ssp); + check_init_srcu_struct(ssp, false); init_completion(&rcu.completion); init_rcu_head_on_stack(&rcu.head); __call_srcu(ssp, &rcu.head, wakeme_after_rcu, do_norm); @@ -1827,7 +1832,7 @@ void srcu_barrier(struct srcu_struct *ssp) int idx; unsigned long s; - check_init_srcu_struct(ssp); + check_init_srcu_struct(ssp, false); /* * Register any deferred callbacks before snapshotting the sequence. The From dab06d6f93798fb2b8aa9475c59ba91ab8097e26 Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Tue, 18 Aug 2026 17:10:15 -0700 Subject: [PATCH 0089/1012] srcutree: Add an atomic Tree SRCU MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Some dedicated srcu_struct users have read-side critical sections which are short, never sleep, and never block on anything which may itself depend on memory allocation — because they were, until recently, spinlock or rwlock critical sections. For such a domain the update side can safely wait for readers by spinning, from contexts where sleeping is undesirable or the grace-period machinery's latency (workqueue scheduling, jiffy-paced retries) dominates the actual reader drain time. But that is only safe if every reader keeps the promise. Make the promise explicit and machine-checkable: - srcu_read_lock_atomic() / srcu_read_unlock_atomic() enter the usual (smp_mb-based) read-side critical section with preemption disabled and (except in hardirq, where it is redundant) non_block_start() armed, recording SRCU_READ_FLAVOR_ATOMIC in the per-CPU reader flavor. Disabling preemption enforces the no-sleeping promise on every configuration and bounds the section, so it is always running on some CPU; non_block_start() extends the enforcement to even potentially-sleeping calls on paths which happen not to block. The existing reader-flavor consistency checks complain about any mixing with other flavors. - synchronize_srcu_atomic() waits for all pre-existing readers by repeating the try_synchronize_srcu() both-epoch counter proof with cpu_relax() until it succeeds: no sleeping, no index flip, no grace-period sequence update, and therefore no interaction with concurrent call_srcu(), synchronize_srcu() or srcu_barrier(). It always provides the full grace-period guarantee: if the domain turns out to have had readers of any other flavor — a caller bug, since such a reader may be asleep and spinning on it would be unbounded — it complains and falls back to a real (sleeping) grace period internally, that being the only correct wait for a possibly-sleeping reader. The flavor mask is rechecked on every iteration so a first non-atomic reader appearing mid-spin takes the same path. The immediate motivation is the proposed conversion of KVM's gfn_to_pfn_cache to SRCU¹, whose mmu_notifier invalidation path drains readers before the primary MMU zaps a page. With the readers declared atomic, that drain becomes spin-only: no sleeping at all in the notifier, bounded by the longest reader section, satisfying even the strictest reading of the OOM-reaper non-blocking requirement without needing to touch the non_block_start() annotation². This commit also adds atomic-SRCU-specific initializers: init_srcu_struct_atomic(), DEFINE_SRCU_ATOMIC(), and DEFINE_STATIC_SRCU_ATOMIC(). This commit implements only Tree SRCU. Tiny SRCU will follow. ¹ https://lore.kernel.org/all/20260811132237.102400-1-dwmw2@infradead.org/ ² https://lore.kernel.org/all/20260812134934.GC662699@ziepe.ca/ [ paulmck: Apply Kunwu Chan feedback. ] Co-developed-by: David Woodhouse Signed-off-by: David Woodhouse Assisted-by: Claude:claude-mythos-5 Signed-off-by: Paul E. McKenney --- include/linux/srcu.h | 81 ++++++++++++++++++-- include/linux/srcutree.h | 9 ++- kernel/rcu/srcutree.c | 154 +++++++++++++++++++++++++++++++++++++-- 3 files changed, 227 insertions(+), 17 deletions(-) diff --git a/include/linux/srcu.h b/include/linux/srcu.h index 7d9bc06df98d0b..3f232f2e052446 100644 --- a/include/linux/srcu.h +++ b/include/linux/srcu.h @@ -36,6 +36,8 @@ static inline int __init_srcu_struct(struct srcu_struct *ssp, const char *name, int __init_srcu_struct_fast(struct srcu_struct *ssp, const char *name, struct lock_class_key *key); int __init_srcu_struct_fast_updown(struct srcu_struct *ssp, const char *name, struct lock_class_key *key); +int __init_srcu_struct_atomic(struct srcu_struct *ssp, const char *name, + struct lock_class_key *key); #endif // #ifndef CONFIG_TINY_SRCU #define init_srcu_struct_fast(ssp) \ @@ -52,6 +54,13 @@ int __init_srcu_struct_fast_updown(struct srcu_struct *ssp, const char *name, __init_srcu_struct_fast_updown((ssp), #ssp, &__srcu_key); \ }) +#define init_srcu_struct_atomic(ssp) \ +({ \ + static struct lock_class_key __srcu_key; \ + \ + __init_srcu_struct_atomic((ssp), #ssp, &__srcu_key); \ +}) + #define __SRCU_DEP_MAP_INIT(srcu_name) .dep_map = { .name = #srcu_name }, #else /* #ifdef CONFIG_DEBUG_LOCK_ALLOC */ @@ -64,6 +73,7 @@ static inline int __init_srcu_struct(struct srcu_struct *ssp, const char *name, #ifndef CONFIG_TINY_SRCU int init_srcu_struct_fast(struct srcu_struct *ssp); int init_srcu_struct_fast_updown(struct srcu_struct *ssp); +int init_srcu_struct_atomic(struct srcu_struct *ssp); #endif // #ifndef CONFIG_TINY_SRCU #define __SRCU_DEP_MAP_INIT(srcu_name) @@ -77,14 +87,18 @@ int init_srcu_struct_fast_updown(struct srcu_struct *ssp); }) /* Values for SRCU Tree srcu_data ->srcu_reader_flavor, but also used by rcutorture. */ -#define SRCU_READ_FLAVOR_NORMAL 0x1 // srcu_read_lock(). -#define SRCU_READ_FLAVOR_NMI 0x2 // srcu_read_lock_nmisafe(). -// 0x4 // SRCU-lite is no longer with us. -#define SRCU_READ_FLAVOR_FAST 0x4 // srcu_read_lock_fast(), also NMI-safe. -#define SRCU_READ_FLAVOR_FAST_UPDOWN 0x8 // srcu_read_lock_fast_updown(). +#define SRCU_READ_FLAVOR_NORMAL 0x01 // srcu_read_lock(). +#define SRCU_READ_FLAVOR_NMI 0x02 // srcu_read_lock_nmisafe(). +// 0x04 // SRCU-lite is no longer with us. +#define SRCU_READ_FLAVOR_FAST 0x04 // srcu_read_lock_fast(), also NMI-safe. +#define SRCU_READ_FLAVOR_FAST_UPDOWN 0x08 // srcu_read_lock_fast_updown(). +#define SRCU_READ_FLAVOR_ATOMIC 0x10 // srcu_read_lock_atomic(). #define SRCU_READ_FLAVOR_ALL (SRCU_READ_FLAVOR_NORMAL | SRCU_READ_FLAVOR_NMI | \ - SRCU_READ_FLAVOR_FAST | SRCU_READ_FLAVOR_FAST_UPDOWN) + SRCU_READ_FLAVOR_FAST | SRCU_READ_FLAVOR_FAST_UPDOWN | \ + SRCU_READ_FLAVOR_ATOMIC) // All of the above. +#define SRCU_READ_FLAVOR_PREDEF (SRCU_READ_FLAVOR_FAST | SRCU_READ_FLAVOR_ATOMIC) + // Flavors special DEFINE_SRCU() flavors. #define SRCU_READ_FLAVOR_SLOWGP (SRCU_READ_FLAVOR_FAST | SRCU_READ_FLAVOR_FAST_UPDOWN) // Flavors requiring synchronize_rcu() // instead of smp_mb(). @@ -102,6 +116,7 @@ void call_srcu(struct srcu_struct *ssp, struct rcu_head *head, void (*func)(struct rcu_head *head)); void cleanup_srcu_struct(struct srcu_struct *ssp); void synchronize_srcu(struct srcu_struct *ssp); +void synchronize_srcu_atomic(struct srcu_struct *ssp); #define SRCU_GET_STATE_COMPLETED 0x1 @@ -306,6 +321,43 @@ static inline int srcu_read_lock(struct srcu_struct *ssp) return retval; } +/** + * srcu_read_lock_atomic - register a new reader promising an atomic section + * @ssp: srcu_struct in which to register the new reader. + * + * As srcu_read_lock(), but the caller promises that the read-side + * critical section never sleeps and never blocks on anything which + * may itself depend on memory allocation to make progress. Preemption + * is disabled for the duration, which both enforces that promise (any + * sleepable call in the section will splat on every configuration) + * and bounds the section so that the update side may spin rather + * than sleep when waiting for readers: see synchronize_srcu_atomic(). + * + * The lock and matching srcu_read_unlock_atomic() must be invoked on + * the same CPU, from the same context; passing the return value to + * another task is not permitted for this flavor. + */ +static inline int srcu_read_lock_atomic(struct srcu_struct *ssp) + __acquires_shared(ssp) +{ + int retval; + + preempt_disable(); + /* + * Arm might_sleep() to catch even a *potentially* sleeping call + * in the section, not just an actual schedule: the atomic-domain + * promise must hold on every path, contended or not. In hardirq + * the annotation would land on the interrupted task; it is also + * redundant there, so skip it. + */ + if (!in_hardirq()) + non_block_start(); + srcu_check_read_flavor(ssp, SRCU_READ_FLAVOR_ATOMIC); + retval = __srcu_read_lock(ssp); + srcu_lock_acquire(&ssp->dep_map); + return retval; +} + /** * srcu_read_lock_fast - register a new reader for an SRCU-protected structure. * @ssp: srcu_struct in which to register the new reader. @@ -498,6 +550,23 @@ static inline void srcu_read_unlock(struct srcu_struct *ssp, int idx) __srcu_read_unlock(ssp, idx); } +/** + * srcu_read_unlock_atomic - unregister an atomic-section reader + * @ssp: srcu_struct from which to unregister the old reader. + * @idx: return value from corresponding srcu_read_lock_atomic(). + */ +static inline void srcu_read_unlock_atomic(struct srcu_struct *ssp, int idx) + __releases_shared(ssp) +{ + WARN_ON_ONCE(idx & ~0x1); + srcu_check_read_flavor(ssp, SRCU_READ_FLAVOR_ATOMIC); + srcu_lock_release(&ssp->dep_map); + __srcu_read_unlock(ssp, idx); + if (!in_hardirq()) + non_block_end(); + preempt_enable(); +} + /** * srcu_read_unlock_fast - unregister a old reader from an SRCU-protected structure. * @ssp: srcu_struct in which to unregister the old reader. diff --git a/include/linux/srcutree.h b/include/linux/srcutree.h index 1ce759fb70948b..ad9d9658b0a2f5 100644 --- a/include/linux/srcutree.h +++ b/include/linux/srcutree.h @@ -80,6 +80,7 @@ struct srcu_usage { struct mutex srcu_cb_mutex; /* Serialize CB preparation. */ raw_spinlock_t __private lock; /* Protect counters and size state. */ struct mutex srcu_gp_mutex; /* Serialize GP work. */ + atomic_t srcu_atomic_gp_flag; /* Serialize atomic GP work. */ unsigned long srcu_gp_seq; /* Grace-period seq #. */ unsigned long srcu_gp_seq_needed; /* Latest gp_seq needed. */ unsigned long srcu_gp_seq_needed_exp; /* Furthest future exp GP. */ @@ -229,14 +230,16 @@ struct srcu_struct { is_static struct srcu_struct name = \ __SRCU_STRUCT_INIT(name, name##_srcu_usage, name##_srcu_data, fast) #endif -#define DEFINE_SRCU(name) __DEFINE_SRCU(name, 0, /* not static */) +#define DEFINE_SRCU(name) __DEFINE_SRCU(name, 0, /* !static */) #define DEFINE_STATIC_SRCU(name) __DEFINE_SRCU(name, 0, static) -#define DEFINE_SRCU_FAST(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST, /* not static */) +#define DEFINE_SRCU_FAST(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST, /* !static */) #define DEFINE_STATIC_SRCU_FAST(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST, static) #define DEFINE_SRCU_FAST_UPDOWN(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST_UPDOWN, \ - /* not static */) + /* !static */) #define DEFINE_STATIC_SRCU_FAST_UPDOWN(name) \ __DEFINE_SRCU(name, SRCU_READ_FLAVOR_FAST_UPDOWN, static) +#define DEFINE_SRCU_ATOMIC(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_ATOMIC, /* !static */) +#define DEFINE_STATIC_SRCU_ATOMIC(name) __DEFINE_SRCU(name, SRCU_READ_FLAVOR_ATOMIC, static) int __srcu_read_lock(struct srcu_struct *ssp) __acquires_shared(ssp); void synchronize_srcu_expedited(struct srcu_struct *ssp); diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index c611a7168c7009..570d068d184029 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -253,6 +253,7 @@ static int init_srcu_struct_fields(struct srcu_struct *ssp, bool is_static, bool ssp->srcu_sup->node = NULL; mutex_init(&ssp->srcu_sup->srcu_cb_mutex); mutex_init(&ssp->srcu_sup->srcu_gp_mutex); + atomic_set(&ssp->srcu_sup->srcu_atomic_gp_flag, 0); ssp->srcu_sup->srcu_gp_seq = SRCU_GP_SEQ_INITIAL_VAL; ssp->srcu_sup->srcu_barrier_seq = 0; mutex_init(&ssp->srcu_sup->srcu_barrier_mutex); @@ -330,6 +331,13 @@ int __init_srcu_struct_fast_updown(struct srcu_struct *ssp, const char *name, } EXPORT_SYMBOL_GPL(__init_srcu_struct_fast_updown); +int __init_srcu_struct_atomic(struct srcu_struct *ssp, const char *name, struct lock_class_key *key) +{ + ssp->srcu_reader_flavor = SRCU_READ_FLAVOR_ATOMIC; + return __init_srcu_struct_common(ssp, name, key); +} +EXPORT_SYMBOL_GPL(__init_srcu_struct_atomic); + #else /* #ifdef CONFIG_DEBUG_LOCK_ALLOC */ /** @@ -385,6 +393,26 @@ int init_srcu_struct_fast_updown(struct srcu_struct *ssp) } EXPORT_SYMBOL_GPL(init_srcu_struct_fast_updown); +/** + * init_srcu_struct_atomic - initialize an atomic sleep-RCU structure + * @ssp: structure to initialize. + * + * Use this in place of DEFINE_SRCU_ATOMIC() and DEFINE_STATIC_SRCU_ATOMIC() + * for non-static srcu_struct structures that are to be passed to + * srcu_read_lock_atomic() and friends. It is necessary to invoke this on a + * given srcu_struct before passing that srcu_struct to any other function. + * Each srcu_struct represents a separate domain of SRCU protection. + * + * And yes, we really are defining a sleepable RCU implementation that + * cannot sleep. Strange universe we live in, isn't it? + */ +int init_srcu_struct_atomic(struct srcu_struct *ssp) +{ + ssp->srcu_reader_flavor = SRCU_READ_FLAVOR_ATOMIC; + return init_srcu_struct_fields(ssp, false, false); +} +EXPORT_SYMBOL_GPL(init_srcu_struct_atomic); + #endif /* #else #ifdef CONFIG_DEBUG_LOCK_ALLOC */ /* @@ -802,7 +830,10 @@ void __srcu_check_read_flavor(struct srcu_struct *ssp, int read_flavor) WARN_ON_ONCE(ssp->srcu_reader_flavor && read_flavor != ssp->srcu_reader_flavor); WARN_ON_ONCE(old_read_flavor && ssp->srcu_reader_flavor && old_read_flavor != ssp->srcu_reader_flavor); - WARN_ON_ONCE(read_flavor == SRCU_READ_FLAVOR_FAST && !ssp->srcu_reader_flavor); + WARN_ON_ONCE(!!(read_flavor & SRCU_READ_FLAVOR_PREDEF) && + read_flavor != ssp->srcu_reader_flavor); + WARN_ON_ONCE(!!(ssp->srcu_reader_flavor & SRCU_READ_FLAVOR_PREDEF) && + read_flavor != ssp->srcu_reader_flavor); if (!old_read_flavor) { old_read_flavor = cmpxchg(&sdp->srcu_reader_flavor, 0, read_flavor); if (!old_read_flavor) @@ -942,7 +973,7 @@ static void srcu_schedule_cbs_snp(struct srcu_struct *ssp, struct srcu_node *snp * are initiating callback invocation. This allows the ->srcu_have_cbs[] * array to have a finite number of elements. */ -static void srcu_gp_end(struct srcu_struct *ssp) +static void srcu_gp_end(struct srcu_struct *ssp, bool is_atomic) { unsigned long cbdelay = 1; bool cbs; @@ -958,7 +989,8 @@ static void srcu_gp_end(struct srcu_struct *ssp) struct srcu_usage *sup = ssp->srcu_sup; /* Prevent more than one additional grace period. */ - mutex_lock(&sup->srcu_cb_mutex); + if (!is_atomic) + mutex_lock(&sup->srcu_cb_mutex); /* End the current grace period. */ raw_spin_lock_irq_rcu_node(sup); @@ -973,7 +1005,8 @@ static void srcu_gp_end(struct srcu_struct *ssp) if (ULONG_CMP_LT(sup->srcu_gp_seq_needed_exp, gpseq)) WRITE_ONCE(sup->srcu_gp_seq_needed_exp, gpseq); raw_spin_unlock_irq_rcu_node(sup); - mutex_unlock(&sup->srcu_gp_mutex); + if (!is_atomic) + mutex_unlock(&sup->srcu_gp_mutex); /* A new grace period can start at this point. But only one. */ /* Initiate callback invocation as needed. */ @@ -1018,13 +1051,15 @@ static void srcu_gp_end(struct srcu_struct *ssp) } /* Callback initiation done, allow grace periods after next. */ - mutex_unlock(&sup->srcu_cb_mutex); + if (!is_atomic) + mutex_unlock(&sup->srcu_cb_mutex); /* Start a new grace period if needed. */ raw_spin_lock_irq_rcu_node(sup); gpseq = rcu_seq_current(&sup->srcu_gp_seq); if (!rcu_seq_state(gpseq) && ULONG_CMP_LT(gpseq, sup->srcu_gp_seq_needed)) { + WARN_ON_ONCE(ssp->srcu_reader_flavor & SRCU_READ_FLAVOR_ATOMIC); srcu_gp_start(ssp); raw_spin_unlock_irq_rcu_node(sup); srcu_reschedule(ssp, 0); @@ -1148,6 +1183,7 @@ static void srcu_funnel_gp_start(struct srcu_struct *ssp, struct srcu_data *sdp, /* If grace period not already in progress, start it. */ if (!WARN_ON_ONCE(rcu_seq_done(&sup->srcu_gp_seq, s)) && rcu_seq_state(sup->srcu_gp_seq) == SRCU_STATE_IDLE) { + WARN_ON_ONCE(ssp->srcu_reader_flavor & SRCU_READ_FLAVOR_ATOMIC); srcu_gp_start(ssp); // And how can that list_add() in the "else" clause @@ -1482,6 +1518,8 @@ static void srcu_do_enqueue(struct srcu_struct *ssp, struct rcu_head *rhp, static void __call_srcu(struct srcu_struct *ssp, struct rcu_head *rhp, rcu_callback_t func, bool do_norm) { + if (WARN_ON_ONCE(ssp->srcu_reader_flavor == SRCU_READ_FLAVOR_ATOMIC)) + return; // Leak the callback rather than corrupt SRCU state. if (should_rcu_defer()) { struct srcu_defer *sndp = this_cpu_ptr(&srcu_defer); struct srcu_data *sdp; @@ -1621,6 +1659,11 @@ static void __synchronize_srcu(struct srcu_struct *ssp, bool do_norm) if (rcu_scheduler_active == RCU_SCHEDULER_INACTIVE) return; + if (WARN_ON_ONCE(ssp->srcu_reader_flavor == SRCU_READ_FLAVOR_ATOMIC)) { + // This works, and exposes a possible bug. + synchronize_srcu_atomic(ssp); + return; + } might_sleep(); check_init_srcu_struct(ssp, false); init_completion(&rcu.completion); @@ -1741,10 +1784,21 @@ EXPORT_SYMBOL_GPL(get_state_synchronize_srcu); * period has elapsed in the meantime. Unlike get_state_synchronize_srcu(), * this function also ensures that any needed SRCU grace period will be * started. This convenience does come at a cost in terms of CPU overhead. + * + * This function cannot be used with atomic SRCU, which only has + * atomic grace periods. Give a warning if someone tries, and return + * the same cookie that would have been returned, but refrain from + * messing up state by starting a grace period. If someone somewhere + * somehow invokes synchronize_srcu_atomic(), passing this cookie to + * poll_state_synchronize_srcu() will return true. If no one ever invokes + * synchronize_srcu_atomic(), too bad. */ unsigned long start_poll_synchronize_srcu(struct srcu_struct *ssp) { - return srcu_gp_start_if_needed(ssp, NULL, true); + if (WARN_ON_ONCE(ssp->srcu_reader_flavor == SRCU_READ_FLAVOR_ATOMIC)) + return get_state_synchronize_srcu(ssp); + else + return srcu_gp_start_if_needed(ssp, NULL, true); } EXPORT_SYMBOL_GPL(start_poll_synchronize_srcu); @@ -1833,6 +1887,12 @@ void srcu_barrier(struct srcu_struct *ssp) unsigned long s; check_init_srcu_struct(ssp, false); + if (WARN_ON_ONCE(ssp->srcu_reader_flavor == SRCU_READ_FLAVOR_ATOMIC)) { + // There shouldn't be any callbacks for atomic SRCU, + // but just in case. + schedule_timeout_uninterruptible(HZ/10); + return; + } /* * Register any deferred callbacks before snapshotting the sequence. The @@ -1904,6 +1964,9 @@ static void srcu_expedite_current_cb(struct rcu_head *rhp) * no current grace period, one might be created. If the current grace * period is currently sleeping, that sleep will complete before expediting * will take effect. + * + * This function must not be invoked on srcu_struct structures that are + * used with srcu_read_lock_atomic() and synchronize_srcu_atomic(). */ void srcu_expedite_current(struct srcu_struct *ssp) { @@ -1911,6 +1974,9 @@ void srcu_expedite_current(struct srcu_struct *ssp) bool needcb = false; struct srcu_data *sdp; + // Atomic SRCU has no callbacks, so there is nothing to expedite. + if (WARN_ON_ONCE(ssp->srcu_reader_flavor == SRCU_READ_FLAVOR_ATOMIC)) + return; migrate_disable(); sdp = this_cpu_ptr(ssp->sda); raw_spin_lock_irqsave_sdp_contention(sdp, &flags); @@ -1978,8 +2044,10 @@ static void srcu_advance_state(struct srcu_struct *ssp, bool is_atomic) return; } idx = rcu_seq_state(READ_ONCE(ssp->srcu_sup->srcu_gp_seq)); - if (idx == SRCU_STATE_IDLE) + if (idx == SRCU_STATE_IDLE) { + WARN_ON_ONCE(ssp->srcu_reader_flavor & SRCU_READ_FLAVOR_ATOMIC); srcu_gp_start(ssp); + } raw_spin_unlock_irq_rcu_node(ssp->srcu_sup); if (idx != SRCU_STATE_IDLE) { if (!is_atomic) @@ -2015,9 +2083,78 @@ static void srcu_advance_state(struct srcu_struct *ssp, bool is_atomic) return; /* readers present, retry later. */ } ssp->srcu_sup->srcu_n_exp_nodelay = 0; - srcu_gp_end(ssp); /* Releases ->srcu_gp_mutex. */ + srcu_gp_end(ssp, is_atomic); /* Releases ->srcu_gp_mutex. */ + } +} + +/** + * synchronize_srcu_atomic - spin for prior SRCU read-side critical-section completion + * @ssp: srcu_struct with which to synchronize. + * + * Similar to synchronize_srcu(), but spins rather than blocking. + * Use only with srcu_read_lock_atomic() and srcu_read_unlock_atomic(), + * which are forbidden from voluntarily context switching. If + * synchronize_srcu_atomic() is invoked from a more restrictive context + * (for example, interrupts disabled) for a given srcu_struct structure, + * then for that structure, all calls to both srcu_read_lock_atomic() + * and srcu_read_unlock_atomic() must be invoked from that same context, + * or one that is even more strict. + * + * If synchronize_srcu_atomic() is invoked on a given srcu_struct + * structure, then none of call_srcu(), synchronize_srcu(), + * synchronize_srcu_expedited(), start_poll_synchronize_srcu(), + * srcu_barrier(), or srcu_expedite_current() may be invoked on that + * same structure. + * + * Because synchronize_srcu_atomic() is even more expedited than is + * synchronize_srcu_expedited(), there is no expedited counterpart to + * this function. + */ +void synchronize_srcu_atomic(struct srcu_struct *ssp) +{ + unsigned long srcu_state; + struct srcu_usage *sup = ssp->srcu_sup; + + // Initialize. Either init_srcu_struct() was invoked or + // DEFINE_SRCU() or similar was used. Therefore, no allocation + // will be done here. + check_init_srcu_struct(ssp, true); + srcu_check_read_flavor(ssp, SRCU_READ_FLAVOR_ATOMIC); + + // Perhaps others will do our work for us. + srcu_state = get_state_synchronize_srcu(ssp); + while (atomic_read(&sup->srcu_atomic_gp_flag) || + atomic_xchg(&sup->srcu_atomic_gp_flag, 1)) { + if (poll_state_synchronize_srcu(ssp, srcu_state)) + return; + cpu_relax(); + } + + // One last check for others doing our work for us under the lock. + raw_spin_lock_irq_rcu_node(sup); + if (poll_state_synchronize_srcu(ssp, srcu_state)) { + raw_spin_unlock_irq_rcu_node(sup); + atomic_set(&sup->srcu_atomic_gp_flag, 0); + return; + } + + // OK, we really have to do it ourselves. Start the grace period. + non_block_start(); // We must not voluntarily block! + smp_store_release(&sup->srcu_gp_seq_needed, srcu_state); // See srcu_funnel_gp_start(). + ASSERT_EXCLUSIVE_WRITER(ssp->srcu_sup->srcu_gp_seq); + srcu_gp_start(ssp); + raw_spin_unlock_irq_rcu_node(sup); + + // Wait for it to complete, helping it along. + while (!poll_state_synchronize_srcu(ssp, srcu_state)) { + cpu_relax(); + srcu_advance_state(ssp, true); } + ASSERT_EXCLUSIVE_WRITER(sup->srcu_atomic_gp_flag); + atomic_set_release(&sup->srcu_atomic_gp_flag, 0); + non_block_end(); } +EXPORT_SYMBOL_GPL(synchronize_srcu_atomic); /* * Invoke a limited number of SRCU callbacks that have passed through @@ -2098,6 +2235,7 @@ static void srcu_reschedule(struct srcu_struct *ssp, unsigned long delay) } } else if (!rcu_seq_state(ssp->srcu_sup->srcu_gp_seq)) { /* Outstanding request and no GP. Start one. */ + WARN_ON_ONCE(ssp->srcu_reader_flavor & SRCU_READ_FLAVOR_ATOMIC); srcu_gp_start(ssp); } raw_spin_unlock_irq_rcu_node(ssp->srcu_sup); From 206d3b443a537e06029dc538f48096ff9b8fff2e Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Tue, 18 Aug 2026 17:10:15 -0700 Subject: [PATCH 0090/1012] srcutiny: Add an atomic Tiny SRCU This commit adds the Tiny SRCU counterpart to Tree SRCU's synchronize_srcu_atomic(). One might hope that this could be as trivial as Tiny RCU's synchronize_rcu(), and there was a time when it would have been. But lazy preemption really can preempt an SRCU read-side critical section, which means that synchronize_srcu_atomic() really must be prepared to spin waiting for it. This spinning currently consists of cond_resched_tasks_rcu_qs() and cpu_relax(). It would be better to have some way of telling the scheduler that there is nothing useful for us to do. We cannot use the traditional wait_event() approach because synchronize_srcu_atomic() is not permitted to block. [ paulmck: Apply Kunwu Chan feedback. ] Co-developed-by: David Woodhouse Signed-off-by: David Woodhouse Assisted-by: Claude:claude-mythos-5 Signed-off-by: Paul E. McKenney --- include/linux/srcutiny.h | 6 +++ kernel/rcu/srcutiny.c | 79 ++++++++++++++++++++++++++++++++++++++++ 2 files changed, 85 insertions(+) diff --git a/include/linux/srcutiny.h b/include/linux/srcutiny.h index 85b5de438450b3..47a368f945e349 100644 --- a/include/linux/srcutiny.h +++ b/include/linux/srcutiny.h @@ -19,6 +19,7 @@ struct srcu_struct { short srcu_lock_nesting[2]; /* srcu_read_lock() nesting depth. */ u8 srcu_gp_running; /* GP workqueue running? */ u8 srcu_gp_waiting; /* GP waiting for readers? */ + u8 srcu_atomic_gp_flag; /* Serialize atomic GP work.*/ unsigned long srcu_idx; /* Current reader array element in bit 0x2. */ unsigned long srcu_idx_max; /* Furthest future srcu_idx request. */ struct swait_queue_head srcu_wq; @@ -64,15 +65,20 @@ void srcu_defer_drain(struct irq_work *irq_work); #define DEFINE_SRCU_FAST_UPDOWN(name) DEFINE_SRCU(name) #define DEFINE_STATIC_SRCU_FAST_UPDOWN(name) \ static struct srcu_struct name = __SRCU_STRUCT_INIT(name, name, name, name) +#define DEFINE_SRCU_ATOMIC(name) DEFINE_SRCU(name) +#define DEFINE_STATIC_SRCU_ATOMIC(name) \ + static struct srcu_struct name = __SRCU_STRUCT_INIT(name, name, name, name) // Dummy structure for srcu_notifier_head. struct srcu_usage { }; #define __SRCU_USAGE_INIT(name) { } #define __init_srcu_struct_fast __init_srcu_struct #define __init_srcu_struct_fast_updown __init_srcu_struct +#define __init_srcu_struct_atomic __init_srcu_struct #ifndef CONFIG_DEBUG_LOCK_ALLOC #define init_srcu_struct_fast init_srcu_struct #define init_srcu_struct_fast_updown init_srcu_struct +#define init_srcu_struct_atomic init_srcu_struct #endif // #ifndef CONFIG_DEBUG_LOCK_ALLOC void synchronize_srcu(struct srcu_struct *ssp); diff --git a/kernel/rcu/srcutiny.c b/kernel/rcu/srcutiny.c index 32b37d63d58aa4..c6a2b74ae9d63c 100644 --- a/kernel/rcu/srcutiny.c +++ b/kernel/rcu/srcutiny.c @@ -41,6 +41,7 @@ static int init_srcu_struct_fields(struct srcu_struct *ssp) ssp->srcu_cb_tail = &ssp->srcu_cb_head; ssp->srcu_gp_running = false; ssp->srcu_gp_waiting = false; + ssp->srcu_atomic_gp_flag = 0; ssp->srcu_idx = 0; ssp->srcu_idx_max = 0; INIT_WORK(&ssp->srcu_work, srcu_drive_gp); @@ -339,6 +340,79 @@ void synchronize_srcu(struct srcu_struct *ssp) } EXPORT_SYMBOL_GPL(synchronize_srcu); +/* + * synchronize_srcu_atomic - spinning grace period for atomic-reader domains + * @ssp: srcu_struct with which to synchronize. + * + * On !SMP this cannot spin: a reader observed mid-section is preempted + * or interrupted-out, and can only finish if we yield the CPU. But it + * is also never needed: an atomic-flavor reader (preemption disabled) + * cannot be observed mid-section from process context on the sole CPU. + * So a reader observed here has broken the atomic-domain promise, and + * the only correct wait for it is a real grace period. + * + * (Actual kernel-doc header is in Tree SRCU.) + */ +void synchronize_srcu_atomic(struct srcu_struct *ssp) +{ + int idx; + bool ret; + unsigned long srcu_state = get_state_synchronize_srcu(ssp); + + srcu_lock_sync(&ssp->dep_map); + + if (IS_ENABLED(CONFIG_PREEMPTION)) + synchronize_rcu(); // Needed for RCU Tasks Trace to imply RCU grace period. + // And in Tiny RCU, it is near zero cost and doesn't block. + + // Usually, there will be no readers. + preempt_disable(); // Guard against lazy preemption and some other grace period. + ret = !READ_ONCE(ssp->srcu_lock_nesting[0]) && !READ_ONCE(ssp->srcu_lock_nesting[1]); + if (ret) { + WRITE_ONCE(ssp->srcu_idx_max, ssp->srcu_idx + 2); + WRITE_ONCE(ssp->srcu_idx, ssp->srcu_idx + 2); + preempt_enable(); + return; + } + + // Wait to drive a grace period or for someone else to do it + // for us while we are lazily preempted. + while (ssp->srcu_atomic_gp_flag) { + if (poll_state_synchronize_srcu(ssp, srcu_state)) { + preempt_enable(); + return; + } + preempt_enable(); + cpu_relax(); + cond_resched_tasks_rcu_qs(); + preempt_disable(); + } + ssp->srcu_atomic_gp_flag = 1; + preempt_enable(); + + // We get here if a reader has been lazily preempted. + // First, wait for old readers, which are quite unlikely. + WRITE_ONCE(ssp->srcu_idx_max, get_state_synchronize_srcu(ssp)); + idx = !(((READ_ONCE(ssp->srcu_idx) + 1) & 0x2) >> 1); + while (READ_ONCE(ssp->srcu_lock_nesting[idx])) { + cond_resched_tasks_rcu_qs(); + cpu_relax(); + } + + // Next, flip the index and wait for the other group of readers. + WRITE_ONCE(ssp->srcu_idx, ssp->srcu_idx + 1); + idx = !idx; + while (READ_ONCE(ssp->srcu_lock_nesting[idx])) { + cond_resched_tasks_rcu_qs(); + cpu_relax(); + } + + // Finally, flip the index again for poll_state_synchronize_srcu(). + WRITE_ONCE(ssp->srcu_idx, ssp->srcu_idx + 1); + WARN_ON_ONCE(!poll_state_synchronize_srcu(ssp, srcu_state)); +} +EXPORT_SYMBOL_GPL(synchronize_srcu_atomic); + /* Register any deferred callbacks, then wait for all in-flight ones. */ void srcu_barrier(struct srcu_struct *ssp) { @@ -367,6 +441,11 @@ EXPORT_SYMBOL_GPL(get_state_synchronize_srcu); * The difference between this and get_state_synchronize_srcu() is that * this function ensures that the poll_state_synchronize_srcu() will * eventually return the value true. + * + * This function cannot be used with atomic SRCU, which only has + * atomic grace periods. Doing so will silently corrupt internal + * SRCU state. Tree SRCU has appropriate checking with splats, + * so please test with CONFIG_SMP=y as well as CONFIG_SMP=n. */ unsigned long start_poll_synchronize_srcu(struct srcu_struct *ssp) { From 43edbab9086ee1b5e486659f03afa3ddf251e5d7 Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Fri, 21 Aug 2026 16:30:31 -0700 Subject: [PATCH 0091/1012] rcutorture: Add support for testing synchronize_srcu_atomic() This commit adds support for the value 0x10 for reader_flavor, which specifies SRCU_READ_FLAVOR_ATOMIC, that is, srcu_read_lock_atomic(), srcu_read_unlock_atomic(), and synchronize_srcu_atomic(). [ paulmck: Apply Kunwu Chan feedback. ] Signed-off-by: Paul E. McKenney --- kernel/rcu/rcutorture.c | 46 +++++++++++++++++++++++++++++++++++------ 1 file changed, 40 insertions(+), 6 deletions(-) diff --git a/kernel/rcu/rcutorture.c b/kernel/rcu/rcutorture.c index 4d1be2a49f011b..182d47975efd18 100644 --- a/kernel/rcu/rcutorture.c +++ b/kernel/rcu/rcutorture.c @@ -710,10 +710,21 @@ static struct rcu_torture_ops rcu_busted_ops = { DEFINE_STATIC_SRCU(srcu_ctl); DEFINE_STATIC_SRCU_FAST(srcu_ctlf); DEFINE_STATIC_SRCU_FAST_UPDOWN(srcu_ctlfud); +DEFINE_STATIC_SRCU_ATOMIC(srcu_ctla); static struct srcu_struct srcu_ctld; static struct srcu_struct *srcu_ctlp = &srcu_ctl; static struct rcu_torture_ops srcud_ops; +// Restrict APIs permitted for atomic SRCU. +static void srcu_torture_init_forbidden_apis(void) +{ + cur_ops->call = NULL; + cur_ops->cb_barrier = NULL; + cur_ops->deferred_free = NULL; + cur_ops->exp_current = NULL; + cur_ops->start_gp_poll = NULL; +} + static void srcu_torture_init(void) { rcu_sync_torture_init(); @@ -729,6 +740,11 @@ static void srcu_torture_init(void) srcu_ctlp = &srcu_ctlfud; VERBOSE_TOROUT_STRING("srcu_torture_init fast-up/down SRCU"); } + if (reader_flavor & SRCU_READ_FLAVOR_ATOMIC) { + srcu_ctlp = &srcu_ctla; + VERBOSE_TOROUT_STRING("srcu_torture_init atomic SRCU"); + srcu_torture_init_forbidden_apis(); + } } static void srcu_get_gp_data(int *flags, unsigned long *gp_seq) @@ -766,6 +782,11 @@ static int srcu_torture_read_lock(void) WARN_ON_ONCE(idx & ~0x1); ret += idx << 3; } + if (reader_flavor & SRCU_READ_FLAVOR_ATOMIC) { + idx = srcu_read_lock_atomic(srcu_ctlp); + WARN_ON_ONCE(idx & ~0x1); + ret += idx << 4; + } return ret; } @@ -785,7 +806,8 @@ srcu_read_delay(struct torture_random_state *rrsp, struct rt_read_seg *rtrsp) delay = torture_random(rrsp) % (nrealreaders * 2 * longdelay * uspertick); - if (!delay && !in_atomic() && !rcu_preempt_depth() && !irqs_disabled()) { + if (!delay && !in_atomic() && !rcu_preempt_depth() && !irqs_disabled() && + !(reader_flavor & SRCU_READ_FLAVOR_ATOMIC)) { schedule_timeout_interruptible(longdelay); rtrsp->rt_delay_jiffies = longdelay; } else { @@ -796,6 +818,8 @@ srcu_read_delay(struct torture_random_state *rrsp, struct rt_read_seg *rtrsp) static void srcu_torture_read_unlock(int idx) { WARN_ON_ONCE((reader_flavor && (idx & ~reader_flavor)) || (!reader_flavor && (idx & ~0x1))); + if (reader_flavor & SRCU_READ_FLAVOR_ATOMIC) + srcu_read_unlock_atomic(srcu_ctlp, (idx & 0x10) >> 4); if (reader_flavor & SRCU_READ_FLAVOR_FAST_UPDOWN) srcu_read_unlock_fast_updown(srcu_ctlp, __srcu_ctr_to_ptr(srcu_ctlp, (idx & 0x8) >> 3)); @@ -875,7 +899,10 @@ static void srcu_torture_deferred_free(struct rcu_torture *rp) static void srcu_torture_synchronize(void) { - synchronize_srcu(srcu_ctlp); + if (reader_flavor & SRCU_READ_FLAVOR_ATOMIC) + synchronize_srcu_atomic(srcu_ctlp); + else + synchronize_srcu(srcu_ctlp); } static unsigned long srcu_torture_get_gp_state(void) @@ -911,7 +938,10 @@ static void srcu_torture_stats(void) static void srcu_torture_synchronize_expedited(void) { - synchronize_srcu_expedited(srcu_ctlp); + if (reader_flavor & SRCU_READ_FLAVOR_ATOMIC) + synchronize_srcu_atomic(srcu_ctlp); + else + synchronize_srcu_expedited(srcu_ctlp); } static void srcu_torture_expedite_current(void) @@ -969,6 +999,10 @@ static void srcud_torture_init(void) } else if (reader_flavor & SRCU_READ_FLAVOR_FAST_UPDOWN) { WARN_ON(init_srcu_struct_fast_updown(&srcu_ctld)); VERBOSE_TOROUT_STRING("srcud_torture_init fast-up/down SRCU"); + } else if (reader_flavor & SRCU_READ_FLAVOR_ATOMIC) { + WARN_ON(init_srcu_struct_atomic(&srcu_ctld)); + VERBOSE_TOROUT_STRING("srcud_torture_init atomic SRCU"); + srcu_torture_init_forbidden_apis(); } else { WARN_ON(init_srcu_struct(&srcu_ctld)); } @@ -1749,7 +1783,7 @@ rcu_torture_writer(void *arg) pr_alert("%s" TORTURE_FLAG " Waited %lu jiffies for boot to complete.\n", torture_type, jiffies - j); - if (IS_ENABLED(CONFIG_RCU_LAZY)) + if (IS_ENABLED(CONFIG_RCU_LAZY) && cur_ops->call) INIT_WORK_ONSTACK(&lazy_work, rcu_torture_writer_work); do { @@ -1944,7 +1978,7 @@ rcu_torture_writer(void *arg) !rcu_gp_is_normal(); } rcu_torture_writer_state = RTWS_STUTTER; - if (IS_ENABLED(CONFIG_RCU_LAZY)) + if (IS_ENABLED(CONFIG_RCU_LAZY) && cur_ops->call) queue_work(system_percpu_wq, &lazy_work); stutter_waited = stutter_wait("rcu_torture_writer"); if (stutter_waited && @@ -1977,7 +2011,7 @@ rcu_torture_writer(void *arg) " Dynamic grace-period expediting was disabled.\n", torture_type); - if (IS_ENABLED(CONFIG_RCU_LAZY)) { + if (IS_ENABLED(CONFIG_RCU_LAZY) && cur_ops->call) { cancel_work_sync(&lazy_work); destroy_work_on_stack(&lazy_work); } From 8d684851ef7b909fb7b3c17ca9824dc78107daf0 Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Wed, 26 Aug 2026 09:03:57 -0700 Subject: [PATCH 0092/1012] srcutree: Disable preemption across synchronize_srcu_atomic() Because synchronize_srcu_atomic() cannot sleep, if a pair of them run concurrently pinned to the same CPU, it is possible that one will spin uselessly waiting for the other while at the same time preventing that other from running. This commit therefore disables preemption to prevent this failure mode. Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutree.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index 570d068d184029..61c2375ba2ecfe 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -2123,11 +2123,14 @@ void synchronize_srcu_atomic(struct srcu_struct *ssp) // Perhaps others will do our work for us. srcu_state = get_state_synchronize_srcu(ssp); + preempt_disable(); while (atomic_read(&sup->srcu_atomic_gp_flag) || atomic_xchg(&sup->srcu_atomic_gp_flag, 1)) { + preempt_enable(); if (poll_state_synchronize_srcu(ssp, srcu_state)) return; cpu_relax(); + preempt_disable(); } // One last check for others doing our work for us under the lock. @@ -2135,6 +2138,7 @@ void synchronize_srcu_atomic(struct srcu_struct *ssp) if (poll_state_synchronize_srcu(ssp, srcu_state)) { raw_spin_unlock_irq_rcu_node(sup); atomic_set(&sup->srcu_atomic_gp_flag, 0); + preempt_enable(); return; } @@ -2152,6 +2156,7 @@ void synchronize_srcu_atomic(struct srcu_struct *ssp) } ASSERT_EXCLUSIVE_WRITER(sup->srcu_atomic_gp_flag); atomic_set_release(&sup->srcu_atomic_gp_flag, 0); + preempt_enable(); non_block_end(); } EXPORT_SYMBOL_GPL(synchronize_srcu_atomic); From b02bb477756f2c3115ecc495bbe7e2ece34b8ccd Mon Sep 17 00:00:00 2001 From: Sebastian Andrzej Siewior Date: Fri, 21 Aug 2026 20:38:09 +0200 Subject: [PATCH 0093/1012] srcu: Use IRQ_WORK_INIT_HARD for srcu's irq_work MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The irq_work in srcu is used to schedule a delayed work. The work item is not scheduled directly because it is not always possible wake a thread directly. On PREEMPT_RT the default irq_work is initialized with IRQ_WORK_LAZY and is delayed to the irq_work thread. A system booted with the command line "trace_event=…" will freeze during boot because the irq_work thread is not yet deployed (and tracing uses synchronize_srcu() in tp_rcu_cond_sync()). The irq_work performs just a wakeup a thread, there is nothing wrong with doing this from hardirq context on PREEMPT_RT. Use IRQ_WORK_INIT_HARD for srcu's irq_work. Fixes: 7c405fb3279b3 ("rcu: Use an intermediate irq_work to start process_srcu()") Signed-off-by: Sebastian Andrzej Siewior Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutree.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index 61c2375ba2ecfe..6625420cfb43f4 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -259,7 +259,12 @@ static int init_srcu_struct_fields(struct srcu_struct *ssp, bool is_static, bool mutex_init(&ssp->srcu_sup->srcu_barrier_mutex); atomic_set(&ssp->srcu_sup->srcu_barrier_cpu_cnt, 0); INIT_DELAYED_WORK(&ssp->srcu_sup->work, process_srcu); - init_irq_work(&ssp->srcu_sup->irq_work, srcu_irq_work); + /* + * trace events started on the command line require SRCU before + * the irq_work kthread starts. Since all it does is a simple + * wakeup, having it as a hard irq, even on PREEMPT_RT is fine. + */ + ssp->srcu_sup->irq_work = IRQ_WORK_INIT_HARD(srcu_irq_work); ssp->srcu_sup->sda_is_static = is_static; if (!is_static) { ssp->sda = alloc_percpu(struct srcu_data); From 251c1c13f426529258d6cdd8f651c97821b2deb9 Mon Sep 17 00:00:00 2001 From: Sunho Park Date: Sun, 30 Aug 2026 18:56:05 +0900 Subject: [PATCH 0094/1012] srcu: Fix WARN_ON() for rcu_segcblist_n_cbs() in cleanup_srcu_struct() The WARN_ON() added by commit 78a38cbf6f20 ("srcu: Queue sdp->work when the delay timer is successfully deleted") uses rcu_segcblist_n_cbs() to detect callbacks that srcu_barrier() failed to wait for. However, the ->len counter is decremented only at the end of srcu_invoke_callbacks(), after the invoking loop has finished. Since srcu_barrier() can return right after the barrier callback is invoked, cleanup_srcu_struct() can see a non-zero n_cbs even though the cblist is already physically empty, falsely triggering the WARN_ON() together with a still-pending delay_work timer. This can be triggered as follows, as seen in the syzbot report against kvm_destroy_vm() -> cleanup_srcu_struct(&kvm->srcu): 1. call_srcu(&kvm->srcu, &bus->rcu, __free_bus) starts SRCU grace period GP1. 2. GP1 ends: a delay timer is armed and sdp->work is queued, but sdp->work has not run yet. 3. Another call_srcu(&kvm->srcu, &bus->rcu, __free_bus) call invokes srcu_segcblist_advance(), which moves the GP1 callback to RCU_DONE_TAIL, and starts SRCU grace period GP2. 4. srcu_barrier() is called. It queues its barrier callback after the GP2 callback and waits for srcu_invoke_callbacks() to invoke it. 5. GP2 ends: another delay timer is armed, and the sdp->work queued in step 2 begins to run. Its srcu_invoke_callbacks() call invokes srcu_segcblist_advance() again, moving the GP2 and barrier callbacks to RCU_DONE_TAIL as well, and then invokes all of them. However, rcu_segcblist_add_len(), which updates srcu_cblist's ->len, has not run yet at this point. 6. srcu_barrier() returns once its callback has been invoked, and cleanup_srcu_struct() starts running. It finds the delay timer armed in step 5 still pending and srcu_cblist's ->len still non-zero (because step 5 has not reached rcu_segcblist_add_len() yet), and WARN_ON() fires even though every callback has actually been invoked. Use rcu_segcblist_empty(), which checks the actual head of the cblist, instead of rcu_segcblist_n_cbs(), which checks the racy ->len counter. Callbacks that have genuinely not been invoked yet still leave the list non-empty, so the WARN_ON() still catches callers that skip srcu_barrier() or queue callbacks after it. Link: https://lore.kernel.org/rcu/e6350377085ddd85d6ef00d8e9a67bd50c762d3c@linux.dev/T/#t Reported-by: syzbot+d4faf7db59e11f6fd1ab@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=d4faf7db59e11f6fd1ab Fixes: 78a38cbf6f20 ("srcu: Queue sdp->work when the delay timer is successfully deleted") Suggested-by: Zqiang Signed-off-by: Sunho Park Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutree.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index 6625420cfb43f4..64081c8eacc84a 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -787,7 +787,7 @@ void cleanup_srcu_struct(struct srcu_struct *ssp) // Call srcu_barrier() before this cleanup_srcu_struct() // to avoid triggering this WARN_ON(). if (WARN_ON(timer_delete_sync(&sdp->delay_work) && - rcu_segcblist_n_cbs(&sdp->srcu_cblist)) && + !rcu_segcblist_empty(&sdp->srcu_cblist)) && rcu_cpu_beenfullyonline(sdp->cpu)) queue_work_on(sdp->cpu, rcu_gp_wq, &sdp->work); flush_work(&sdp->work); From 321eab145d17a1ea12b1a87b7123cecaf91dfb6d Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Wed, 2 Sep 2026 08:30:38 -0700 Subject: [PATCH 0095/1012] srcutree: Warn if Tiny SRCU readers are preempted The fastpath of synchronize_srcu_atomic() should always be taken because readers always disable preemption. Therefore, the fact that synchronize_srcu_atomic() is running at all should mean that all readers have ended. But there are always bugs. And responding to a usage bug with a too-short SRCU grace period, and thus possibly corrupting memory is at best a sadistic response so such a bug. For this reason, synchronize_srcu_atomic() explicitly waits for readers. Except that it does so silently, possibly failing to flag this bug. This commit therefore adds a splat if synchronize_srcu_atomic() fails to take the early exit. Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutiny.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/kernel/rcu/srcutiny.c b/kernel/rcu/srcutiny.c index c6a2b74ae9d63c..99f8bfd98b045e 100644 --- a/kernel/rcu/srcutiny.c +++ b/kernel/rcu/srcutiny.c @@ -375,6 +375,14 @@ void synchronize_srcu_atomic(struct srcu_struct *ssp) return; } + // Because readers disable preemption, we should never get here. + // However, a splat and some spinning is usually preferable to + // memory corruption due to a too-short grace period. There is + // the possibility that this will hang if the preempted reader is + // not looked upon favorably by the scheduler, but this is still + // preferable to memory corruption. + WARN_ON_ONCE(1); + // Wait to drive a grace period or for someone else to do it // for us while we are lazily preempted. while (ssp->srcu_atomic_gp_flag) { From b865bcbb3f1a7974a6082dc1ff8358490159687f Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Wed, 2 Sep 2026 12:58:24 -0700 Subject: [PATCH 0096/1012] srcutree: Explicitly note DEFINE_SRCU() needs for srcu_barrier() In the core kernel, an srcu_struct structure created by DEFINE_SRCU() or friends lives as long as the kernel does, so there are no particular requirements surrounding the end of that structure's life. In contrast, when DEFINE_SRCU() and friends are used within a module, the corresponding srcu_struct structures' lifetimes end when that module exits. This in turn means that if such a structure was passed to call_srcu(), then srcu_barrier() must be invoked after the last call_srcu() invocation but before the module exits. This commit therefore adds a comment stating this. Reported-by: Alexander Aring Signed-off-by: Paul E. McKenney --- include/linux/srcutree.h | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/include/linux/srcutree.h b/include/linux/srcutree.h index ad9d9658b0a2f5..93b76e54389432 100644 --- a/include/linux/srcutree.h +++ b/include/linux/srcutree.h @@ -214,6 +214,12 @@ struct srcu_struct { * instead of smp_mb(), and given that the first (for example) * srcu_read_lock_fast() might race with the first synchronize_srcu(), * this different must be specified at initialization time. + * + * If you use any of the DEFINE_SRCU() functions within a module, the + * module-entry code will invoke init_srcu_struct() and the module-exit + * code will invoke cleanup_srcu_struct(). This means that if your module + * passes the resulting srcu_struct structure to call_srcu(), you will + * need to also pass this structure to srcu_barrier() prior to module exit. */ #ifdef MODULE # define __DEFINE_SRCU(name, fast, is_static) \ From b3331e96d2d887a8364db6b987638a20f1ae4ee1 Mon Sep 17 00:00:00 2001 From: Kunwu Chan Date: Mon, 7 Sep 2026 15:58:20 +0800 Subject: [PATCH 0097/1012] rcutorture: Add atomic-SRCU support to torture.sh Add the --do-atomic-srcu argument to torture.sh, which runs the SRCU-N, SRCU-P, and SRCU-T scenarios, thus covering both Tree SRCU (SRCU-N and SRCU-P) and Tiny SRCU (SRCU-T), with rcutorture.reader_flavor=0x10 appended to the boot parameters so that it takes precedence over each scenario's own reader-flavor setting. This exercises srcu_read_lock_atomic(), srcu_read_unlock_atomic(), and synchronize_srcu_atomic(). As with other torture.sh tests, the --do-kcsan argument runs a KCSAN+PROVE_LOCKING variant of this test. [ paulmck: Make --do-atomic-srcu be default-on. ] Signed-off-by: Kunwu Chan Signed-off-by: Paul E. McKenney --- .../selftests/rcutorture/bin/torture.sh | 24 +++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/tools/testing/selftests/rcutorture/bin/torture.sh b/tools/testing/selftests/rcutorture/bin/torture.sh index f0083891ee8147..e7b578ec152f16 100755 --- a/tools/testing/selftests/rcutorture/bin/torture.sh +++ b/tools/testing/selftests/rcutorture/bin/torture.sh @@ -68,6 +68,7 @@ do_clocksourcewd="${ifnotaarch64}" do_rt=yes do_rcutasksflavors="${ifnotaarch64}" # FIXME: Back to "yes" when SMP=n auto-avoided do_srcu_lockdep=yes +do_atomic_srcu=yes do_rcu_rust=no # doyesno - Helper function for yes/no arguments @@ -103,6 +104,7 @@ usage () { echo " --do-rcu-rust / --do-no-rcu-rust / --no-rcu-rust" echo " --do-scftorture / --do-no-scftorture / --no-scftorture" echo " --do-srcu-lockdep / --do-no-srcu-lockdep / --no-srcu-lockdep" + echo " --do-atomic-srcu / --do-no-atomic-srcu / --no-atomic-srcu" echo " --duration [ | h | d ]" echo " --guest-cpu-limit N" echo " --kcsan-kmake-arg kernel-make-arguments" @@ -148,6 +150,7 @@ do do_kcsan=yes do_clocksourcewd="${ifnotaarch64}" do_srcu_lockdep=yes + do_atomic_srcu=yes ;; --do-allmodconfig|--do-no-allmodconfig|--no-allmodconfig) do_allmodconfig=`doyesno "$1" --do-allmodconfig` @@ -183,6 +186,7 @@ do do_kcsan=no do_clocksourcewd=no do_srcu_lockdep=no + do_atomic_srcu=no ;; --do-normal|--do-norm|--do-no-normal|--do-no-norm|--no-normal|--no-norm) do_normal=`doyesno "$1" --do-normal` @@ -212,6 +216,9 @@ do --do-srcu-lockdep|--do-no-srcu-lockdep|--no-srcu-lockdep) do_srcu_lockdep=`doyesno "$1" --do-srcu-lockdep` ;; + --do-atomic-srcu|--do-no-atomic-srcu|--no-atomic-srcu) + do_atomic_srcu=`doyesno "$1" --do-atomic-srcu` + ;; --duration) checkarg --duration "(minutes)" $# "$2" '^[0-9][0-9]*\(m\|h\|d\|\)$' '^error' mult=1 @@ -497,6 +504,23 @@ then torture_set "rcutorture" tools/testing/selftests/rcutorture/bin/kvm.sh --allcpus --duration "$duration_rcutorture" --configs "$configs_rcutorture" --trust-make fi +# Test atomic SRCU across Tree SRCU (SRCU-N and SRCU-P) and Tiny SRCU +# (SRCU-T). The reader flavor selects srcu_read_lock_atomic() and +# synchronize_srcu_atomic(). Tiny SRCU requires SMP=n, which aarch64 +# does not support. +if test "$do_atomic_srcu" = "yes" +then + torture_bootargs="rcutorture.reader_flavor=0x10" + configs_atomic_srcu="SRCU-N SRCU-P" + if test "$ifnotaarch64" = yes + then + configs_atomic_srcu="$configs_atomic_srcu SRCU-T" + fi + torture_set "atomic-srcu" tools/testing/selftests/rcutorture/bin/kvm.sh \ + --allcpus --duration "$duration_rcutorture" \ + --configs "$configs_atomic_srcu" --trust-make +fi + if test "$do_locktorture" = "yes" then torture_bootargs="torture.disable_onoff_at_boot" From 3a9fa2af6396bba22899322c6e8d9c94f8c99d9c Mon Sep 17 00:00:00 2001 From: Kunwu Chan Date: Mon, 7 Sep 2026 15:58:19 +0800 Subject: [PATCH 0098/1012] srcutree: Add reader-free fastpath to synchronize_srcu_atomic() synchronize_srcu_atomic() is restricted to srcu_read_lock_atomic() and srcu_read_unlock_atomic(), whose read-side critical sections disable preemption. In the common case where there are no readers at all, the grace period therefore need not do the index flip. Add a fastpath that sums both ranks of the per-CPU ->srcu_ctrs[] counters and, if the lock counts match the unlock counts on both ranks, ends the grace period immediately, skipping the srcu_advance_state() scans, mirroring the similar Tiny SRCU fastpath. Correctness requires the counter-sum proof to follow the grace-period anchor written by srcu_gp_start(); placing it before the anchor could let this grace period miss a pre-existing reader and return without waiting for it. The smp_mb() between the unlock and lock sums pairs with the smp_mb() in __srcu_read_lock(). The grace period is ended manually under ->lock and ->srcu_atomic_gp_flag. In theory, this is slower than David Woodhouse's earlier patch, but David's measurements showed that the performance was close enough that it makes sense to keep the get_state_synchronize_srcu() and poll_state_synchronize_srcu semantics. Link: https://lore.kernel.org/all/20260907075829.2073224-4-kunwu.chan@linux.dev/ Link: https://lore.kernel.org/all/0f4dea21bac43685d3286df329401177e4452b36.camel@infradead.org/ Link: https://lore.kernel.org/all/0d4af6318ac67486858be1df8d436147b444a2d2.camel@infradead.org/ Signed-off-by: Kunwu Chan Signed-off-by: Paul E. McKenney Cc: David Woodhouse --- kernel/rcu/srcutree.c | 48 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index 64081c8eacc84a..93b8d1088abe83 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -2119,6 +2119,8 @@ void synchronize_srcu_atomic(struct srcu_struct *ssp) { unsigned long srcu_state; struct srcu_usage *sup = ssp->srcu_sup; + unsigned long rdm0, rdm1; + unsigned long unlocks0, unlocks1; // Initialize. Either init_srcu_struct() was invoked or // DEFINE_SRCU() or similar was used. Therefore, no allocation @@ -2154,6 +2156,52 @@ void synchronize_srcu_atomic(struct srcu_struct *ssp) srcu_gp_start(ssp); raw_spin_unlock_irq_rcu_node(sup); + // + // Fastpath: If there are no readers at all, neither grace-period + // scan need wait, so both can be satisfied at once without doing + // the index flip. The counter-sum proof is the same as that of + // srcu_readers_active_idx_check(), but spanning both indices. + // Atomic SRCU guarantees that all readers are of + // SRCU_READ_FLAVOR_ATOMIC, so the SLOWGP check never triggers and + // the ->srcu_reader_flavor masks returned by + // srcu_readers_unlock_idx() are unused. + // + // This proof must follow the grace-period anchor written by the + // srcu_gp_start() above, never precede it. With the anchor first, + // a reader whose lock increment is missed by the sums below cannot + // have incremented its lock counter before the anchor, and therefore + // cannot be a pre-existing reader of this grace period. Placing the + // proof before the anchor would let this grace period miss a + // pre-existing reader and return without waiting for it. + // + // The smp_mb() pairs with the smp_mb() in __srcu_read_lock() + // (store-buffering pattern), which guarantees that a lock is always + // counted if the corresponding unlock is counted, the same + // memory-ordering guarantee as is provided by + // srcu_readers_active_idx_check(). + // + unlocks0 = srcu_readers_unlock_idx(ssp, 0, &rdm0); + unlocks1 = srcu_readers_unlock_idx(ssp, 1, &rdm1); + smp_mb(); /* A */ + if (srcu_readers_lock_idx(ssp, 0, false, unlocks0) && + srcu_readers_lock_idx(ssp, 1, false, unlocks1)) { + // No readers, so end this grace period manually, skipping + // the index flip. Advancing the sequence number via + // rcu_seq_start() in srcu_gp_start() above and rcu_seq_end() + // below keeps get_state_synchronize_srcu() and + // poll_state_synchronize_srcu() working, all under ->lock + // and ->srcu_atomic_gp_flag, which excludes concurrent + // sequence-number updates. + raw_spin_lock_irq_rcu_node(sup); + rcu_seq_end(&sup->srcu_gp_seq); + raw_spin_unlock_irq_rcu_node(sup); + WARN_ON_ONCE(!poll_state_synchronize_srcu(ssp, srcu_state)); + atomic_set_release(&sup->srcu_atomic_gp_flag, 0); + preempt_enable(); + non_block_end(); + return; + } + // Wait for it to complete, helping it along. while (!poll_state_synchronize_srcu(ssp, srcu_state)) { cpu_relax(); From 840b8874bf2a71d515318bcd86507bc098d5e2fe Mon Sep 17 00:00:00 2001 From: Kunwu Chan Date: Mon, 7 Sep 2026 15:58:26 +0800 Subject: [PATCH 0099/1012] srcutree: Skip callback scheduling for atomic SRCU grace periods call_srcu() is forbidden on atomic srcu_struct, so srcu_gp_end() never has callbacks to invoke for them. Yet it schedules callback invocation, which for atomic SRCU's SRCU_SIZE_SMALL state arms the boot CPU's ->delay_work timer every grace period, only for srcu_invoke_callbacks() to find nothing to do. Skip this for atomic SRCU. Signed-off-by: Kunwu Chan Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutree.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index 93b8d1088abe83..0954486880c7d1 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -1016,10 +1016,10 @@ static void srcu_gp_end(struct srcu_struct *ssp, bool is_atomic) /* Initiate callback invocation as needed. */ ss_state = smp_load_acquire(&sup->srcu_size_state); - if (ss_state < SRCU_SIZE_WAIT_BARRIER) { + if (!is_atomic && ss_state < SRCU_SIZE_WAIT_BARRIER) { srcu_schedule_cbs_sdp(per_cpu_ptr(ssp->sda, get_boot_cpu_id()), cbdelay); - } else { + } else if (!is_atomic) { idx = rcu_seq_ctr(gpseq) % ARRAY_SIZE(snp->srcu_have_cbs); srcu_for_each_node_breadth_first(ssp, snp) { raw_spin_lock_irq_rcu_node(snp); From b3e2518af3fe58d67a4d0f76941901fc3c1e9312 Mon Sep 17 00:00:00 2001 From: Kunwu Chan Date: Mon, 7 Sep 2026 15:58:27 +0800 Subject: [PATCH 0100/1012] srcutree: Remove srcu_barrier() sleep for atomic SRCU The atomic-SRCU path in srcu_barrier() sleeps for 100 milliseconds just in case there are callbacks to wait for. But call_srcu() refuses atomic SRCU with a WARN_ON_ONCE() before reaching the deferred-enqueue path, so there can be no callbacks, deferred or otherwise. Drop the sleep. Signed-off-by: Kunwu Chan Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutree.c | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index 0954486880c7d1..82e61421142e5d 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -1892,12 +1892,9 @@ void srcu_barrier(struct srcu_struct *ssp) unsigned long s; check_init_srcu_struct(ssp, false); - if (WARN_ON_ONCE(ssp->srcu_reader_flavor == SRCU_READ_FLAVOR_ATOMIC)) { - // There shouldn't be any callbacks for atomic SRCU, - // but just in case. - schedule_timeout_uninterruptible(HZ/10); + // Atomic SRCU has no callbacks, so there is nothing to wait on. + if (WARN_ON_ONCE(ssp->srcu_reader_flavor == SRCU_READ_FLAVOR_ATOMIC)) return; - } /* * Register any deferred callbacks before snapshotting the sequence. The From 11bde0eae691e382218a408a029a012132da69d0 Mon Sep 17 00:00:00 2001 From: Kunwu Chan Date: Mon, 7 Sep 2026 15:58:29 +0800 Subject: [PATCH 0101/1012] srcu: Restrict atomic-SRCU non_block annotation to task context srcu_read_lock_atomic() and srcu_read_unlock_atomic() arm and disarm might_sleep() checks via non_block_start()/non_block_end(), skipping hardirq so the update does not land on the interrupted task's ->non_block_count. Inline softirqs run on the interrupted task's stack as well, so a timer callback running atomic-SRCU readers races with the interrupted task's own ->non_block_count updates, as KCSAN reports: BUG: KCSAN: data-race in srcu_torture_read_lock / srcu_torture_read_unlock write to 0xffffa00f818ea418 of 4 bytes by interrupt on cpu 0: srcu_torture_read_lock+0x422/0x470 rcutorture_one_extend+0xdc/0x600 rcu_torture_one_read+0xd1/0x330 rcu_torture_timer+0x75/0x140 call_timer_fn+0xe6/0x2f0 ... run_timer_softirq+0xb7/0x130 handle_softirqs+0xfc/0x3f0 __irq_exit_rcu+0x8e/0x100 Use in_task() so the annotation is applied only in task context; it is redundant elsewhere because might_sleep() already warns about sleeping from atomic context. Signed-off-by: Kunwu Chan Signed-off-by: Paul E. McKenney --- include/linux/srcu.h | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/include/linux/srcu.h b/include/linux/srcu.h index 3f232f2e052446..1a8a465a5650dc 100644 --- a/include/linux/srcu.h +++ b/include/linux/srcu.h @@ -346,11 +346,13 @@ static inline int srcu_read_lock_atomic(struct srcu_struct *ssp) /* * Arm might_sleep() to catch even a *potentially* sleeping call * in the section, not just an actual schedule: the atomic-domain - * promise must hold on every path, contended or not. In hardirq - * the annotation would land on the interrupted task; it is also + * promise must hold on every path, contended or not. In hardirq, + * softirq, or NMI the annotation would land on the interrupted + * task, and can also result in data races against that task's + * own non_block_start()/non_block_end() invocations; it is also * redundant there, so skip it. */ - if (!in_hardirq()) + if (in_task()) non_block_start(); srcu_check_read_flavor(ssp, SRCU_READ_FLAVOR_ATOMIC); retval = __srcu_read_lock(ssp); @@ -562,7 +564,7 @@ static inline void srcu_read_unlock_atomic(struct srcu_struct *ssp, int idx) srcu_check_read_flavor(ssp, SRCU_READ_FLAVOR_ATOMIC); srcu_lock_release(&ssp->dep_map); __srcu_read_unlock(ssp, idx); - if (!in_hardirq()) + if (in_task()) non_block_end(); preempt_enable(); } From 6308a115b8a6bf24f680337a9fe2b6e95a8f0a9d Mon Sep 17 00:00:00 2001 From: Kunwu Chan Date: Mon, 7 Sep 2026 15:58:22 +0800 Subject: [PATCH 0102/1012] srcutree: Make init_srcu_struct_atomic() prevent transition to big The is_atomic parameter of init_srcu_struct_fields() exists so that atomic SRCU never transitions to big, but neither init_srcu_struct_atomic() nor its lockdep counterpart __init_srcu_struct_atomic() sets it. On systems where srcutree.convert_to_big selects SRCU_SIZING_INIT, this needlessly allocates a full srcu_node combining tree for any dynamically initialized atomic srcu_struct, despite atomic SRCU having neither callbacks nor srcu_barrier() operations. Pass true from both atomic entry points, adding an is_atomic parameter to __init_srcu_struct_common(). Signed-off-by: Kunwu Chan Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutree.c | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index 82e61421142e5d..4e9a0b8ea34494 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -305,26 +305,27 @@ static int init_srcu_struct_fields(struct srcu_struct *ssp, bool is_static, bool #ifdef CONFIG_DEBUG_LOCK_ALLOC static int -__init_srcu_struct_common(struct srcu_struct *ssp, const char *name, struct lock_class_key *key) +__init_srcu_struct_common(struct srcu_struct *ssp, const char *name, + struct lock_class_key *key, bool is_atomic) { /* Don't re-initialize a lock while it is held. */ debug_check_no_locks_freed((void *)ssp, sizeof(*ssp)); lockdep_init_map(&ssp->dep_map, name, key, 0); - return init_srcu_struct_fields(ssp, false, false); + return init_srcu_struct_fields(ssp, false, is_atomic); } int init_srcu_struct_lockdep(struct srcu_struct *ssp, const char *name, struct lock_class_key *key) { ssp->srcu_reader_flavor = 0; - return __init_srcu_struct_common(ssp, name, key); + return __init_srcu_struct_common(ssp, name, key, false); } EXPORT_SYMBOL_GPL(init_srcu_struct_lockdep); int __init_srcu_struct_fast(struct srcu_struct *ssp, const char *name, struct lock_class_key *key) { ssp->srcu_reader_flavor = SRCU_READ_FLAVOR_FAST; - return __init_srcu_struct_common(ssp, name, key); + return __init_srcu_struct_common(ssp, name, key, false); } EXPORT_SYMBOL_GPL(__init_srcu_struct_fast); @@ -332,14 +333,14 @@ int __init_srcu_struct_fast_updown(struct srcu_struct *ssp, const char *name, struct lock_class_key *key) { ssp->srcu_reader_flavor = SRCU_READ_FLAVOR_FAST_UPDOWN; - return __init_srcu_struct_common(ssp, name, key); + return __init_srcu_struct_common(ssp, name, key, false); } EXPORT_SYMBOL_GPL(__init_srcu_struct_fast_updown); int __init_srcu_struct_atomic(struct srcu_struct *ssp, const char *name, struct lock_class_key *key) { ssp->srcu_reader_flavor = SRCU_READ_FLAVOR_ATOMIC; - return __init_srcu_struct_common(ssp, name, key); + return __init_srcu_struct_common(ssp, name, key, true); } EXPORT_SYMBOL_GPL(__init_srcu_struct_atomic); @@ -414,7 +415,7 @@ EXPORT_SYMBOL_GPL(init_srcu_struct_fast_updown); int init_srcu_struct_atomic(struct srcu_struct *ssp) { ssp->srcu_reader_flavor = SRCU_READ_FLAVOR_ATOMIC; - return init_srcu_struct_fields(ssp, false, false); + return init_srcu_struct_fields(ssp, false, true); } EXPORT_SYMBOL_GPL(init_srcu_struct_atomic); From b19888cb203839c08932281fcfd4678074f21833 Mon Sep 17 00:00:00 2001 From: Kunwu Chan Date: Fri, 11 Sep 2026 15:21:25 +0800 Subject: [PATCH 0103/1012] srcutree: Don't transition atomic SRCU to big in srcu_gp_end() Atomic SRCU must remain in the small size state. Warn if this invariant is violated and avoid transitioning to big in that case. [ paulmck: Folded "if" onto one line for 100-character limit. ] Signed-off-by: Kunwu Chan Signed-off-by: Paul E. McKenney Reviewed-by: Bradley Morgan --- kernel/rcu/srcutree.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index 4e9a0b8ea34494..6a05bd8a6f390a 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -1073,8 +1073,10 @@ static void srcu_gp_end(struct srcu_struct *ssp, bool is_atomic) raw_spin_unlock_irq_rcu_node(sup); } - /* Transition to big if needed. */ - if (ss_state != SRCU_SIZE_SMALL && ss_state != SRCU_SIZE_BIG) { + /* Transition to big if needed, but never for atomic SRCU. */ + if (ssp->srcu_reader_flavor == SRCU_READ_FLAVOR_ATOMIC && ss_state != SRCU_SIZE_SMALL) { + WARN_ON_ONCE(1); + } else if (ss_state != SRCU_SIZE_SMALL && ss_state != SRCU_SIZE_BIG) { if (ss_state == SRCU_SIZE_ALLOC) init_srcu_struct_nodes(ssp, GFP_KERNEL); else From 89aa2d12f6d53f0f94d17582d1832d6f2ebe603f Mon Sep 17 00:00:00 2001 From: Kunwu Chan Date: Fri, 11 Sep 2026 12:02:51 +0800 Subject: [PATCH 0104/1012] srcutree: Skip torture to-big transition for atomic SRCU Atomic SRCU does not use the srcu_node combining tree. Exclude it from the SRCU_SIZING_IS_TORTURE() transition in srcu_torture_stats_print(), and warn if such a transition is requested. Signed-off-by: Kunwu Chan Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutree.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index 6a05bd8a6f390a..6e37d53b412e1a 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -2422,8 +2422,10 @@ void srcu_torture_stats_print(struct srcu_struct *ssp, char *tt, char *tf) } pr_cont(" T(%ld,%ld)\n", s0, s1); } - if (SRCU_SIZING_IS_TORTURE()) - srcu_transition_to_big(ssp); + if (SRCU_SIZING_IS_TORTURE()) { + if (!WARN_ON_ONCE(ssp->srcu_reader_flavor == SRCU_READ_FLAVOR_ATOMIC)) + srcu_transition_to_big(ssp); + } } EXPORT_SYMBOL_GPL(srcu_torture_stats_print); From f009ad97675774aaf3c08249ddf30b73d61cb822 Mon Sep 17 00:00:00 2001 From: Thierry Reding Date: Wed, 2 Sep 2026 12:17:37 +0200 Subject: [PATCH 0105/1012] ARM: tegra: Clean up AHUB on Tegra124 Use #address-cells = <1> and #size-cells = <1> because we don't need 64-bit register addressing for this hardware. While at it, also adjust the ranges property to encompass the entire AHUB range as per the TRM. Signed-off-by: Thierry Reding --- arch/arm/boot/dts/nvidia/tegra124.dtsi | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/arch/arm/boot/dts/nvidia/tegra124.dtsi b/arch/arm/boot/dts/nvidia/tegra124.dtsi index ce4efa1de509b7..86956b94ca0b03 100644 --- a/arch/arm/boot/dts/nvidia/tegra124.dtsi +++ b/arch/arm/boot/dts/nvidia/tegra124.dtsi @@ -1114,13 +1114,13 @@ "rx3", "tx3", "rx4", "tx4", "rx5", "tx5", "rx6", "tx6", "rx7", "tx7", "rx8", "tx8", "rx9", "tx9"; - ranges; - #address-cells = <2>; - #size-cells = <2>; + ranges = <0x70300000 0x0 0x70300000 0x10000>; + #address-cells = <1>; + #size-cells = <1>; tegra_i2s0: i2s@70301000 { compatible = "nvidia,tegra124-i2s"; - reg = <0x0 0x70301000 0x0 0x100>; + reg = <0x70301000 0x100>; nvidia,ahub-cif-ids = <4 4>; clocks = <&tegra_car TEGRA124_CLK_I2S0>; resets = <&tegra_car 30>; @@ -1130,7 +1130,7 @@ tegra_i2s1: i2s@70301100 { compatible = "nvidia,tegra124-i2s"; - reg = <0x0 0x70301100 0x0 0x100>; + reg = <0x70301100 0x100>; nvidia,ahub-cif-ids = <5 5>; clocks = <&tegra_car TEGRA124_CLK_I2S1>; resets = <&tegra_car 11>; @@ -1140,7 +1140,7 @@ tegra_i2s2: i2s@70301200 { compatible = "nvidia,tegra124-i2s"; - reg = <0x0 0x70301200 0x0 0x100>; + reg = <0x70301200 0x100>; nvidia,ahub-cif-ids = <6 6>; clocks = <&tegra_car TEGRA124_CLK_I2S2>; resets = <&tegra_car 18>; @@ -1150,7 +1150,7 @@ tegra_i2s3: i2s@70301300 { compatible = "nvidia,tegra124-i2s"; - reg = <0x0 0x70301300 0x0 0x100>; + reg = <0x70301300 0x100>; nvidia,ahub-cif-ids = <7 7>; clocks = <&tegra_car TEGRA124_CLK_I2S3>; resets = <&tegra_car 101>; @@ -1160,7 +1160,7 @@ tegra_i2s4: i2s@70301400 { compatible = "nvidia,tegra124-i2s"; - reg = <0x0 0x70301400 0x0 0x100>; + reg = <0x70301400 0x100>; nvidia,ahub-cif-ids = <8 8>; clocks = <&tegra_car TEGRA124_CLK_I2S4>; resets = <&tegra_car 102>; From 5e10d31007a5713b67495b5f156f35ca2078cd32 Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Wed, 9 Sep 2026 15:27:57 -0700 Subject: [PATCH 0106/1012] torture.sh: Add hazptr torturing This commit adds hazard-pointer torturing to the torture.sh script as a default-on option. This uses the same duration as rcutorture. While in the area, change kvm.sh to tolerate the empty-string arguments that torture.sh can generate, for example, if a given test needs no boot arguments. Signed-off-by: Paul E. McKenney Cc: Mathieu Desnoyers Cc: Boqun Feng Cc: Bradley Morgan --- tools/testing/selftests/rcutorture/bin/kvm.sh | 4 +++ .../selftests/rcutorture/bin/torture.sh | 26 ++++++++++++++++++- 2 files changed, 29 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/rcutorture/bin/kvm.sh b/tools/testing/selftests/rcutorture/bin/kvm.sh index 32c3199976388d..2f386cf536e361 100755 --- a/tools/testing/selftests/rcutorture/bin/kvm.sh +++ b/tools/testing/selftests/rcutorture/bin/kvm.sh @@ -98,6 +98,7 @@ usage () { while test $# -gt 0 do + echo Argument: :$1: case "$1" in --allcpus) cpus=$TORTURE_ALLOTED_CPUS @@ -271,6 +272,9 @@ do --trust-make) TORTURE_TRUST_MAKE="y" ;; + "") + # torture.sh can pass empty arguments. Ignore them. + ;; *) echo Unknown argument $1 usage diff --git a/tools/testing/selftests/rcutorture/bin/torture.sh b/tools/testing/selftests/rcutorture/bin/torture.sh index f0083891ee8147..2824dab14d41ed 100755 --- a/tools/testing/selftests/rcutorture/bin/torture.sh +++ b/tools/testing/selftests/rcutorture/bin/torture.sh @@ -43,6 +43,7 @@ fi configs_rcutorture= configs_locktorture= configs_scftorture= +configs_hazptr= kcsan_kmake_args= # Default compression, duration, and apportionment. @@ -69,6 +70,7 @@ do_rt=yes do_rcutasksflavors="${ifnotaarch64}" # FIXME: Back to "yes" when SMP=n auto-avoided do_srcu_lockdep=yes do_rcu_rust=no +do_hazptr=yes # doyesno - Helper function for yes/no arguments function doyesno () { @@ -89,6 +91,7 @@ usage () { echo " --do-all" echo " --do-allmodconfig / --do-no-allmodconfig / --no-allmodconfig" echo " --do-clocksourcewd / --do-no-clocksourcewd / --no-clocksourcewd" + echo " --do-hazptr / --do-no-hazptr / --no-hazptr" echo " --do-kasan / --do-no-kasan / --no-kasan" echo " --do-kcsan / --do-no-kcsan / --no-kcsan" echo " --do-kvfree / --do-no-kvfree / --no-kvfree" @@ -117,6 +120,11 @@ do compress_concurrency=$2 shift ;; + --config-hazptr|--configs-hazptr) + checkarg --configs-hazptr "(list of config files)" "$#" "$2" '^[^/]\+$' '^--' + configs_hazptr="$configs_hazptr $2" + shift + ;; --config-rcutorture|--configs-rcutorture) checkarg --configs-rcutorture "(list of config files)" "$#" "$2" '^[^/]\+$' '^--' configs_rcutorture="$configs_rcutorture $2" @@ -147,6 +155,7 @@ do do_kasan=yes do_kcsan=yes do_clocksourcewd="${ifnotaarch64}" + do_hazptr=yes do_srcu_lockdep=yes ;; --do-allmodconfig|--do-no-allmodconfig|--no-allmodconfig) @@ -155,6 +164,9 @@ do --do-clocksourcewd|--do-no-clocksourcewd|--no-clocksourcewd) do_clocksourcewd=`doyesno "$1" --do-clocksourcewd` ;; + --do-hazptr|--do-no-hazptr|--no-hazptr) + do_hazptr=`doyesno "$1" --do-hazptr` + ;; --do-kasan|--do-no-kasan|--no-kasan) do_kasan=`doyesno "$1" --do-kasan` ;; @@ -182,6 +194,7 @@ do do_kasan=no do_kcsan=no do_clocksourcewd=no + do_hazptr=no do_srcu_lockdep=no ;; --do-normal|--do-norm|--do-no-normal|--do-no-norm|--no-normal|--no-norm) @@ -343,7 +356,7 @@ function torture_one { boottag="--bootargs" cur_bootargs="$torture_bootargs" fi - "$@" $boottag "$cur_bootargs" --datestamp "$ds/results-$curflavor" > $T/$curflavor.out 2>&1 + "$@" "${boottag}" "$cur_bootargs" --datestamp "$ds/results-$curflavor" > $T/$curflavor.out 2>&1 retcode=$? resdir="`grep '^Results directory: ' $T/$curflavor.out | tail -1 | sed -e 's/^Results directory: //'`" if test -z "$resdir" @@ -704,6 +717,17 @@ then fi fi +# Calculate hazptr defaults and apportion time +if test -z "$configs_hazptr" +then + configs_hazptr=CFLIST +fi +if test "$do_hazptr" = "yes" +then + torture_bootargs="" + torture_set "hazptr" tools/testing/selftests/rcutorture/bin/kvm.sh --torture hazptr --allcpus --duration "$duration_rcutorture" --configs "$configs_hazptr" --trust-make +fi + echo " --- " $scriptname $args echo " --- " Done `date` | tee -a $T/log ret=0 From 1eccf02fb8d5359c4143cb50e6f52313b65fe7cd Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Thu, 3 Sep 2026 15:26:24 -0700 Subject: [PATCH 0107/1012] rcu: Add running and boosted indications to RCU task stall dump Currently, rcu_print_task_stall() will dump out the PID, RCU reader nesting level, the rcu_special structure's flags, and whether or not that reader is on the ->blkd_tasks list. When debugging RCU priority boosting, it is also good to know whether the stalled RCU reader is currently running and whether it is currently being RCU priority boosted. This commit therefore adds this information to the output. [ paulmck: Apply kernel test robot feedback. ] Signed-off-by: Paul E. McKenney --- kernel/rcu/tree_stall.h | 23 +++++++++++++++++++++-- 1 file changed, 21 insertions(+), 2 deletions(-) diff --git a/kernel/rcu/tree_stall.h b/kernel/rcu/tree_stall.h index 93ba31a619b670..5dded1e8919759 100644 --- a/kernel/rcu/tree_stall.h +++ b/kernel/rcu/tree_stall.h @@ -11,6 +11,7 @@ #include #include #include +#include ////////////////////////////////////////////////////////////////////////////// // @@ -305,6 +306,8 @@ struct rcu_stall_chk_rdr { int nesting; union rcu_special rs; bool on_blkd_list; + bool rcu_rdr_running; + int rcu_rdr_boosted; }; /* @@ -313,6 +316,7 @@ struct rcu_stall_chk_rdr { */ static int check_slow_task(struct task_struct *t, void *arg) { + struct rcu_node *rnp; struct rcu_stall_chk_rdr *rscrp = arg; if (task_curr(t)) @@ -320,6 +324,19 @@ static int check_slow_task(struct task_struct *t, void *arg) rscrp->nesting = t->rcu_read_lock_nesting; rscrp->rs = t->rcu_read_unlock_special; rscrp->on_blkd_list = !list_empty(&t->rcu_node_entry); + rscrp->rcu_rdr_running = task_curr(t); + rscrp->rcu_rdr_boosted = 0; + if (rscrp->on_blkd_list) { + rnp = READ_ONCE(t->rcu_blocked_node); + raw_spin_lock_rcu_node(rnp); /* irqs already disabled. */ + if (rnp == READ_ONCE(t->rcu_blocked_node)) { + if (rt_mutex_owner(&rnp->boost_mtx.rtmutex) == t) + rscrp->rcu_rdr_boosted = 1; + } else { + rscrp->rcu_rdr_boosted = 2; + } + raw_spin_unlock_rcu_node(rnp); /* irqs remain disabled. */ + } return 0; } @@ -357,12 +374,14 @@ static int rcu_print_task_stall(struct rcu_node *rnp, unsigned long flags) if (task_call_func(t, check_slow_task, &rscr)) pr_cont(" P%d", t->pid); else - pr_cont(" P%d/%d:%c%c%c%c", + pr_cont(" P%d/%d:%c%c%c%c%c%c", t->pid, rscr.nesting, ".b"[rscr.rs.b.blocked], ".q"[rscr.rs.b.need_qs], ".e"[rscr.rs.b.exp_hint], - ".l"[rscr.on_blkd_list]); + ".l"[rscr.on_blkd_list], + ".R"[rscr.rcu_rdr_running], + ".B?"[rscr.rcu_rdr_boosted]); lockdep_assert_irqs_disabled(); put_task_struct(t); ndetected++; From 42b4285017966d0805035cd3b1acf94d9b85ef92 Mon Sep 17 00:00:00 2001 From: Hemanth Selam Date: Mon, 7 Sep 2026 12:06:22 +0530 Subject: [PATCH 0108/1012] rcu: Fix typo "upto" in comment Correct "upto" to "up to", reported by scripts/checkpatch.pl using the misspelling list in scripts/spelling.txt. Only touches comments, no code changes. Assisted-by: Cursor:claude-opus-5 Signed-off-by: Hemanth Selam Signed-off-by: Paul E. McKenney --- kernel/rcu/srcutree.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/rcu/srcutree.c b/kernel/rcu/srcutree.c index ed204b3f4b8440..01f19bb17dd6d6 100644 --- a/kernel/rcu/srcutree.c +++ b/kernel/rcu/srcutree.c @@ -624,7 +624,7 @@ module_param(srcu_retry_check_delay, ulong, 0444); #define SRCU_UL_CLAMP_LO(val, low) ((val) > (low) ? (val) : (low)) #define SRCU_UL_CLAMP_HI(val, high) ((val) < (high) ? (val) : (high)) #define SRCU_UL_CLAMP(val, low, high) SRCU_UL_CLAMP_HI(SRCU_UL_CLAMP_LO((val), (low)), (high)) -// per-GP-phase no-delay instances adjusted to allow non-sleeping poll upto +// per-GP-phase no-delay instances adjusted to allow non-sleeping poll up to // one jiffies time duration. Mult by 2 is done to factor in the srcu_get_delay() // called from process_srcu(). #define SRCU_DEFAULT_MAX_NODELAY_PHASE_ADJUSTED \ From e21d364f5eb918912d539bbda763688ef06d1fec Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Thu, 10 Sep 2026 17:25:37 +0000 Subject: [PATCH 0109/1012] rcu: Drop the private tick-internal.h include from tree.c Nothing in the tree.c translation unit uses anything from the private timekeeping header ../time/tick-internal.h, included since commit 48d07c04b4cc ("rcu: Enable elimination of Tree-RCU softirq processing"). The tick symbols RCU uses, tick_dep_set(), tick_dep_clear(), their _cpu and _task variants, tick_nohz_full_cpu() and TICK_DEP_BIT_RCU, are all declared in the public linux/tick.h, which tree.c already includes. Drop the include. Signed-off-by: Bradley Morgan Signed-off-by: Paul E. McKenney --- kernel/rcu/tree.c | 1 - 1 file changed, 1 deletion(-) diff --git a/kernel/rcu/tree.c b/kernel/rcu/tree.c index 96848fc1f02b8f..338737b9781cba 100644 --- a/kernel/rcu/tree.c +++ b/kernel/rcu/tree.c @@ -64,7 +64,6 @@ #include #include #include -#include "../time/tick-internal.h" #include "tree.h" #include "rcu.h" From 8998488e9a0be01609c09bce284e45d8fe340d00 Mon Sep 17 00:00:00 2001 From: Matthias Goergens Date: Fri, 11 Sep 2026 11:40:59 +0800 Subject: [PATCH 0110/1012] rcu: Make userspace barrier hook drain kvfree_rcu work The bcachefs ktest allocation-leak check writes rcutree.do_rcu_barrier before reading /proc/allocinfo. While testing bcachefs performance changes, small objects released with kfree_rcu() remained visible after repeated writes to the hook and 20 seconds of waiting, causing otherwise clean tests to fail their leak check. The test assumes a stronger contract than the hook currently documents: rcu_barrier() waits for ordinary callbacks, but does not flush objects still held in kfree_rcu() batching or per-CPU SLUB sheaves. The retained population eventually fell as a sheaf filled; there is no evidence here of unbounded growth or OOM. Changing the hook to drain kvfree_rcu() work let the same unmodified bcachefs workload pass its allocation check. All eight checkpoints in one VM, after 50 through 400 option changes, reported zero retained reconcile_scan objects. This motivated the separate private-cache test used to isolate the incomplete drain from bcachefs. Calling kvfree_rcu_barrier() from rcu_barrier_throttled() was proposed when kvfree_rcu_barrier() was added in 2024, to restore a clean baseline between userspace benchmark runs. The discussion concluded that keeping the existing hook name, adding the second operation and documenting both was the safest compatibility choice, but the follow-up was not added. Add that drain and document the stronger test interface. Always retain the existing start-rate limit and perform the kvfree_rcu() drain: an unrelated ordinary barrier does not establish that this work completed. Retain the entry ordinary-barrier sequence snapshot. After draining, skip the final ordinary barrier only if that snapshot is complete, preserving the memory barrier on the completion path. Otherwise, invoke rcu_barrier() explicitly. This keeps the ordinary-callback guarantee independent of whether kvfree_rcu_barrier() embeds an ordinary barrier. Clarify that the documented completion guarantee covers work queued before the request, without preventing new work from being queued. Earlier validation of the unconditional-drain version used four fresh VM pairs with a private-cache fixture: controls retained the queued object (60 to 60 active objects), and treatments drained it (60 to 59). An ordinary-callback test passed on both kernels. Those runs predated the guarded skip and do not validate that change. No elapsed-time improvement is claimed. Link: https://lore.kernel.org/all/20240820155935.1167988-1-urezki@gmail.com/ Signed-off-by: Matthias Goergens Signed-off-by: Paul E. McKenney --- .../admin-guide/kernel-parameters.txt | 9 ++++-- kernel/rcu/tree.c | 30 ++++++++++++------- 2 files changed, 26 insertions(+), 13 deletions(-) diff --git a/Documentation/admin-guide/kernel-parameters.txt b/Documentation/admin-guide/kernel-parameters.txt index 68647ff4bdd24b..914b65ae941346 100644 --- a/Documentation/admin-guide/kernel-parameters.txt +++ b/Documentation/admin-guide/kernel-parameters.txt @@ -5699,9 +5699,12 @@ Kernel parameters there is an ongoing too-long CSD-lock wait. rcutree.do_rcu_barrier= [KNL] - Request a call to rcu_barrier(). This is - throttled so that userspace tests can safely - hammer on the sysfs variable if they so choose. + Wait for deferred kfree_rcu() frees and ordinary + call_rcu() callbacks queued before this request to + complete. This does not prevent new work from being + queued concurrently. Requests are throttled so that + userspace tests can safely hammer on the sysfs + variable if they so choose. If triggered before the RCU grace-period machinery is fully active, this will error out with EAGAIN. diff --git a/kernel/rcu/tree.c b/kernel/rcu/tree.c index 338737b9781cba..f60252390d5dea 100644 --- a/kernel/rcu/tree.c +++ b/kernel/rcu/tree.c @@ -3988,12 +3988,12 @@ EXPORT_SYMBOL_GPL(rcu_barrier); static unsigned long rcu_barrier_last_throttle; /** - * rcu_barrier_throttled - Do rcu_barrier(), but limit to one per second + * rcu_barrier_throttled - Drain deferred RCU frees, but rate-limit starts * - * This can be thought of as guard rails around rcu_barrier() that - * permits unrestricted userspace use, at least assuming the hardware's - * try_cmpxchg() is robust. There will be at most one call per second to - * rcu_barrier() system-wide from use of this function, which means that + * This can be thought of as guard rails around the deferred-free barriers + * that permit unrestricted userspace use, at least assuming the hardware's + * try_cmpxchg() is robust. There will be at most one drain operation started + * per sixteenth of a second from use of this function, which means that * callers might needlessly wait a second or three. * * This is intended for use by test suites to avoid OOM by flushing RCU @@ -4015,14 +4015,24 @@ static void rcu_barrier_throttled(void) while (time_in_range(j, old, old + HZ / 16) || !try_cmpxchg(&rcu_barrier_last_throttle, &old, j)) { schedule_timeout_idle(HZ / 16); - if (rcu_seq_done(&rcu_state.barrier_sequence, s)) { - smp_mb(); /* caller's subsequent code after above check. */ - return; - } j = jiffies; old = READ_ONCE(rcu_barrier_last_throttle); } - rcu_barrier(); + /* + * kfree_rcu() can retain objects outside the ordinary callback lists in + * per-CPU SLUB sheaves and kvfree_rcu batches. Always drain those queues: + * an ordinary barrier does not establish that this work was drained. + */ + kvfree_rcu_barrier(); + /* + * A completed barrier can still cover ordinary callbacks queued before + * our entry snapshot. Otherwise, retain an explicit ordinary barrier + * without depending on the implementation of kvfree_rcu_barrier(). + */ + if (rcu_seq_done(&rcu_state.barrier_sequence, s)) + smp_mb(); /* caller's subsequent code after above check. */ + else + rcu_barrier(); } /* From 11fd060b915850d5e256e85a21b454df161bc1bc Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Fri, 11 Sep 2026 23:36:43 +0200 Subject: [PATCH 0111/1012] rcuref: Fix the rcuread_is_dead reference in rcuref_read() kernel-doc The kernel-doc comment of rcuref_read() refers to rcuread_is_dead, which does not exist. The name is rcuref_is_dead. Say rcuref_is_dead. Fixes: 3efa66ce6ee1 ("rcuref: Provide rcuref_is_dead()") Assisted-by: LLM Signed-off-by: Karl Mehltretter Reviewed-by: Sebastian Andrzej Siewior Signed-off-by: Paul E. McKenney --- include/linux/rcuref.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/include/linux/rcuref.h b/include/linux/rcuref.h index 2fb2af6d982497..01fee161b67c3e 100644 --- a/include/linux/rcuref.h +++ b/include/linux/rcuref.h @@ -34,7 +34,7 @@ static inline void rcuref_init(rcuref_t *ref, unsigned int cnt) * indicate that it is safe to schedule the object, protected by this reference * counter, for deconstruction. * If you want to know if the reference counter has been marked DEAD (as - * signaled by rcuref_put()) please use rcuread_is_dead(). + * signaled by rcuref_put()) please use rcuref_is_dead(). */ static inline unsigned int rcuref_read(rcuref_t *ref) { From 64b7dbc3e2226f1259577b32d5e7675c2839d93d Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Tue, 11 Aug 2026 11:44:00 -0700 Subject: [PATCH 0112/1012] rcu-tasks: Disable callback contend/collapse messages by default New workloads can do large bursts of call_rcu_tasks() invocations in a short time period, followed by a quiet time period long enough to drain all of the callbacks, followed by another burst of call_rcu_tasks() invocations. This can cause RCU Tasks to switch back and forth between queuing callbacks only on CPU 0 (during quiet periods) and on all CPUs (during bursts). Which is fine. Except for the fact that each cycle from CPU-0-only to all-CPUs queuing and back generates three console messages, one announcing the shift to all-CPUs queuing, another announcing the start of the shift back to CPU-0-only queuing, and the third announcing completion of this shift after an RCU grace period. And these console messages can overrun console-log communications channels and obscure other console-message-based debugging information. And the only known use for these console messages is debugging RCU Tasks itself. This commit therefore adds a rcupdate.rcu_task_collapse_debug module parameter that defaults to false (suppressing these console messages). Those debugging or otherwise playing with RCU Tasks callback queuing auto-adjustment can set this parameter to the value true. [ paulmck: Apply Breno Leitao feedback. ] Reported-by: Breno Leitao Reported-by: David Dai Signed-off-by: Paul E. McKenney Reviewed-by: Breno Leitao --- Documentation/admin-guide/kernel-parameters.txt | 7 +++++++ kernel/rcu/tasks.h | 15 ++++++++++++--- 2 files changed, 19 insertions(+), 3 deletions(-) diff --git a/Documentation/admin-guide/kernel-parameters.txt b/Documentation/admin-guide/kernel-parameters.txt index 914b65ae941346..6cc6d45b59d68a 100644 --- a/Documentation/admin-guide/kernel-parameters.txt +++ b/Documentation/admin-guide/kernel-parameters.txt @@ -6405,6 +6405,13 @@ Kernel parameters period to instead use normal non-expedited grace-period processing. + rcupdate.rcu_task_collapse_debug= [KNL] + Enable debugging prints that record when RCU Tasks + and RCU Tasks Trace expand to per-CPU callback + queuing and collapse back to CPU-0 queuing. + This is default-disabled due to the fact that + some workloads can make it quite noisy. + rcupdate.rcu_task_collapse_lim= [KNL] Set the maximum number of callbacks present at the beginning of a grace period that allows diff --git a/kernel/rcu/tasks.h b/kernel/rcu/tasks.h index 627295396cd91d..fcac7361ec51ec 100644 --- a/kernel/rcu/tasks.h +++ b/kernel/rcu/tasks.h @@ -178,6 +178,8 @@ static int rcu_task_contend_lim __read_mostly = 100; module_param(rcu_task_contend_lim, int, 0444); static int rcu_task_collapse_lim __read_mostly = 10; module_param(rcu_task_collapse_lim, int, 0444); +static bool rcu_task_collapse_debug __read_mostly; +module_param(rcu_task_collapse_debug, bool, 0644); static int rcu_task_lazy_lim __read_mostly = 32; module_param(rcu_task_lazy_lim, int, 0444); @@ -390,7 +392,8 @@ static void call_rcu_tasks_generic(struct rcu_head *rhp, rcu_callback_t func, WRITE_ONCE(rtp->percpu_enqueue_shift, 0); WRITE_ONCE(rtp->percpu_dequeue_lim, rcu_task_cpu_ids); smp_store_release(&rtp->percpu_enqueue_lim, rcu_task_cpu_ids); - pr_info("Switching %s to per-CPU callback queuing.\n", rtp->name); + if (data_race(rcu_task_collapse_debug)) + pr_info("Switching %s to per-CPU callback queuing.\n", rtp->name); } raw_spin_unlock_irqrestore(&rtp->cbs_gbl_lock, flags); } @@ -511,7 +514,9 @@ static int rcu_tasks_need_gpcb(struct rcu_tasks *rtp) smp_store_release(&rtp->percpu_enqueue_lim, 1); rtp->percpu_dequeue_gpseq = get_state_synchronize_rcu(); gpdone = false; - pr_info("Starting switch %s to CPU-0 callback queuing.\n", rtp->name); + if (data_race(rcu_task_collapse_debug)) + pr_info("Starting switch %s to CPU-0 callback queuing.\n", + rtp->name); } raw_spin_unlock_irqrestore(&rtp->cbs_gbl_lock, flags); } @@ -519,7 +524,9 @@ static int rcu_tasks_need_gpcb(struct rcu_tasks *rtp) raw_spin_lock_irqsave(&rtp->cbs_gbl_lock, flags); if (rtp->percpu_enqueue_lim < rtp->percpu_dequeue_lim) { WRITE_ONCE(rtp->percpu_dequeue_lim, 1); - pr_info("Completing switch %s to CPU-0 callback queuing.\n", rtp->name); + if (data_race(rcu_task_collapse_debug)) + pr_info("Completing switch %s to CPU-0 callback queuing.\n", + rtp->name); } if (rtp->percpu_dequeue_lim == 1) { for (cpu = rtp->percpu_dequeue_lim; cpu < rcu_task_cpu_ids; cpu++) { @@ -704,6 +711,8 @@ static void __init rcu_tasks_bootup_oddness(void) pr_info("\tTasks-RCU CPU stall info multiplier clamped to %d (rcu_task_stall_info_mult).\n", rtsimc); rcu_task_stall_info_mult = rtsimc; } + if (rcu_task_collapse_debug) + pr_info("\tTasks-RCU callback contend/collapse debug enabled.\n"); #endif /* #ifdef CONFIG_TASKS_RCU */ #ifdef CONFIG_TASKS_RCU pr_info("\tTrampoline variant of Tasks RCU enabled.\n"); From b309290a246f34995c2e212ef51b620847722f39 Mon Sep 17 00:00:00 2001 From: Kunwu Chan Date: Thu, 20 Aug 2026 17:42:28 +0800 Subject: [PATCH 0113/1012] rcutorture: Fix divide-by-zero with fwd_progress_div=1 When fwd_progress_div=1, the forward-progress test computes: sd4 = (sd + div - 1) / div = sd dur = sd4 + torture_random(&trs) % (sd - sd4) = sd4 + % 0 The modulo operation with a zero divisor triggers an integer division by zero (undefined behavior at the C level, #DE trap on x86), causing a kernel Oops and panic. On x86_64, this manifests as: rcu_torture_fwd_prog_nr: Starting forward-progress test 0 Oops: divide error: 0000 [#1] SMP PTI RIP: 0010:rcu_torture_fwd_prog+0x90b/0x1160 R12: 0000000000000000 The existing guard only handles non-positive values. However, fwd_progress_div=1 also makes the random range empty because sd4 == sd. Change the guard to reject values below 2. The forward-progress test only reaches this calculation when stall_dur() is positive, so sd = stall_dur() + 1 >= 2. For fwd_progress_div >= 2, sd4 < sd, ensuring that sd - sd4 is at least 1. Keep the existing fallback to the default value of 4 for invalid values. Verified with QEMU/KVM: a 138-second run with fwd_progress_div=1 completed 81 forward-progress test cycles without a crash. Fixes: 1b27291b1ea4f ("rcutorture: Add forward-progress tests for RCU grace periods") Signed-off-by: Kunwu Chan Signed-off-by: Paul E. McKenney --- kernel/rcu/rcutorture.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/rcu/rcutorture.c b/kernel/rcu/rcutorture.c index 182d47975efd18..79807475b67245 100644 --- a/kernel/rcu/rcutorture.c +++ b/kernel/rcu/rcutorture.c @@ -4024,7 +4024,7 @@ static int __init rcu_torture_fwd_prog_init(void) } if (fwd_progress_holdoff <= 0) fwd_progress_holdoff = 1; - if (fwd_progress_div <= 0) + if (fwd_progress_div < 2) fwd_progress_div = 4; rfp = kzalloc_objs(*rfp, fwd_progress); fwd_prog_tasks = kzalloc_objs(*fwd_prog_tasks, fwd_progress); From 834cecb21e378bc3a70cab08e5678adfde7a581a Mon Sep 17 00:00:00 2001 From: Zqiang Date: Mon, 24 Aug 2026 17:24:27 +0800 Subject: [PATCH 0114/1012] rcutorture: Synchronously wait for all rcu_torture_irq() callbacks to complete The rcu_torture_reader() drives RCU readers from interrupt context via smp_call_function_single(cpu, rcu_torture_irq, NULL, 0) with wait=0, to runs rcu_torture_irq() on a remote CPU. this is async, nothing waits for the remote handler to run. On shutdown, torture_stop_kthread() only waits for each reader kthread to return, and the reader's timer_delete_sync() only drains its timer. Neither waits for a rcu_torture_irq() which still pending or executing on a remote CPU, so it can run after all readers have exited and rcu_torture_cleanup() has already advanced. 1. rcu_torture_irq() may issue cur_ops->call(rhp, rcu_torture_timer_cb) after cur_ops->cb_barrier() has been waiting for all outstanding callbacks complete. once the module is unloaded, fires into freed module text, a use-after-free happen. 2. rcu_torture_irq() may still be inside rcu_torture_one_read(), holding a read-side critical section, when cur_ops->cleanup() tears the flavor down (e.g. cleanup_srcu_struct()), triggering an active-reader warning or use-after-free of the torn-down structure. This commit therefore issue a kick_all_cpus_sync() after all readers kthread have returned and before cur_ops->cb_barrier(), synchronous IPI round trip to every CPU guarantees that every rcu_torture_irq() which previously issued by any reader has completed. Signed-off-by: Zqiang Signed-off-by: Paul E. McKenney --- kernel/rcu/rcutorture.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/kernel/rcu/rcutorture.c b/kernel/rcu/rcutorture.c index 79807475b67245..51ed35f5aab171 100644 --- a/kernel/rcu/rcutorture.c +++ b/kernel/rcu/rcutorture.c @@ -4491,6 +4491,8 @@ rcu_torture_cleanup(void) for (i = 0; i < nrealreaders; i++) torture_stop_kthread(rcu_torture_reader, reader_tasks[i]); + if (irqreader && cur_ops->irq_capable) + kick_all_cpus_sync(); kfree(reader_tasks); reader_tasks = NULL; } From c3f9b7bf999c2d7f5f0fdb696ec80895895e37ba Mon Sep 17 00:00:00 2001 From: "Paul E. McKenney" Date: Thu, 27 Aug 2026 11:25:51 -0700 Subject: [PATCH 0115/1012] torture: Allow specifying alternative ssh command to kvm-remote.sh Some environments require use of alternative commands to access the test hosts. This commit therefore adds a KVM_REMOTE_SSH environment variable for this purpose. If this variable is unset, ssh is used. Any alternative ssh command must support the usual ssh arguments, including the command to be executed remotely. In some cases, you may need a wrapper script to make the alternative ssh-like command look enough like ssh to satisfy kvm-remote.sh. Signed-off-by: Paul E. McKenney --- .../selftests/rcutorture/bin/kvm-remote.sh | 22 ++++++++++++++----- 1 file changed, 16 insertions(+), 6 deletions(-) diff --git a/tools/testing/selftests/rcutorture/bin/kvm-remote.sh b/tools/testing/selftests/rcutorture/bin/kvm-remote.sh index 48a8052d5dae37..a8397f86e49dc7 100755 --- a/tools/testing/selftests/rcutorture/bin/kvm-remote.sh +++ b/tools/testing/selftests/rcutorture/bin/kvm-remote.sh @@ -6,6 +6,11 @@ # Usage: kvm-remote.sh "systems" [ ] # kvm-remote.sh "systems" /path/to/old/run [ ] # +# The caller may set the KVM_REMOTE_SSH environment in order to specify +# an alternative ssh command, which is necessary in some environments +# for authentication purposes. This alternative ssn command must support +# ssh's usual arguments. +# # Copyright (C) 2021 Facebook, Inc. # # Authors: Paul E. McKenney @@ -13,6 +18,11 @@ scriptname=$0 args="$*" +if test -z "${KVM_REMOTE_SSH}" +then + KVM_REMOTE_SSH=ssh; export KVM_REMOTE_SSH +fi + if ! test -d tools/testing/selftests/rcutorture/bin then echo $scriptname must be run from top-level directory of kernel source tree. @@ -137,7 +147,7 @@ chmod +x $T/bin/kvm-remote-*.sh # Check first to avoid the need for cleanup for system-name typos for i in $systems do - ssh -o BatchMode=yes $i getconf _NPROCESSORS_ONLN > $T/ssh.stdout 2> $T/ssh.stderr + ${KVM_REMOTE_SSH} -o BatchMode=yes $i getconf _NPROCESSORS_ONLN > $T/ssh.stdout 2> $T/ssh.stderr ret=$? if test "$ret" -ne 0 then @@ -158,14 +168,14 @@ echo Build-products tarball: `du -h $T/binres.tgz` | tee -a "$oldrun/remote-log" for i in $systems do echo Downloading tarball to $i `date` | tee -a "$oldrun/remote-log" - cat $T/binres.tgz | ssh -o BatchMode=yes $i "cd /tmp; tar -xzf -" + cat $T/binres.tgz | ${KVM_REMOTE_SSH} -o BatchMode=yes $i "cd /tmp; tar -xzf -" ret=$? tries=0 while test "$ret" -ne 0 do echo Unable to download $T/binres.tgz to system $i, waiting and then retrying. $tries prior retries. | tee -a "$oldrun/remote-log" sleep 60 - cat $T/binres.tgz | ssh -o BatchMode=yes $i "cd /tmp; tar -xzf -" + cat $T/binres.tgz | ${KVM_REMOTE_SSH} -o BatchMode=yes $i "cd /tmp; tar -xzf -" ret=$? if test "$ret" -ne 0 then @@ -191,7 +201,7 @@ checkremotefile () { while : do - ssh -o BatchMode=yes $1 "test -f \"$2\"" + ${KVM_REMOTE_SSH} -o BatchMode=yes $1 "test -f \"$2\"" ret=$? if test "$ret" -eq 255 then @@ -239,7 +249,7 @@ startbatches () { then continue # System still running last test, skip. fi - ssh -o BatchMode=yes "$i" "cd \"$resdir/$ds\"; touch remote.run; PATH=\"$T/bin:$PATH\" nohup kvm-remote-$curbatch.sh > kvm-remote-$curbatch.sh.out 2>&1 &" 1>&2 + ${KVM_REMOTE_SSH} -o BatchMode=yes "$i" "cd \"$resdir/$ds\"; touch remote.run; PATH=\"$T/bin:$PATH\" nohup kvm-remote-$curbatch.sh > kvm-remote-$curbatch.sh.out 2>&1 &" 1>&2 ret=$? if test "$ret" -ne 0 then @@ -281,7 +291,7 @@ do if test "$ret" -eq 1 then echo " ---" Collecting results from $i `date` | tee -a "$oldrun/remote-log" - ( cd "$oldrun"; ssh -o BatchMode=yes $i "cd $rundir; tar -czf - kvm-remote-*.sh.out */console.log */kvm-test-1-run*.sh.out */qemu[_-]pid */qemu-retval */qemu-affinity; rm -rf $T > /dev/null 2>&1" | tar -xzf - ) + ( cd "$oldrun"; ${KVM_REMOTE_SSH} -o BatchMode=yes $i "cd $rundir; tar -czf - kvm-remote-*.sh.out */console.log */kvm-test-1-run*.sh.out */qemu[_-]pid */qemu-retval */qemu-affinity; rm -rf $T > /dev/null 2>&1" | tar -xzf - ) break; fi if test "$ret" -eq 255 From 8364c973ed19972cb84e4c61c69d09cc6f921572 Mon Sep 17 00:00:00 2001 From: Marc Zyngier Date: Sat, 19 Sep 2026 13:16:56 +0100 Subject: [PATCH 0116/1012] KVM: arm64: vgic-v5: Correctly handle host ISTE __le32 conversion The vgic-v5 defines the host ISTE (h_iste) as a __le32, but uses get_user/put_user on a userspace buffer declared as u32 *. This leads to sparse having yet another fit. Instead, define h_iste as a u32, and perform the conversion at the point of doing the access on the architectural state. Reported-by: kernel test robot Closes: https://lore.kernel.org/oe-kbuild-all/202609191215.to1MBYak-lkp@intel.com/ Signed-off-by: Marc Zyngier --- arch/arm64/kvm/vgic/vgic-v5-tables.c | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/arch/arm64/kvm/vgic/vgic-v5-tables.c b/arch/arm64/kvm/vgic/vgic-v5-tables.c index 2c8d0f360506cc..aae5cf05f3907e 100644 --- a/arch/arm64/kvm/vgic/vgic-v5-tables.c +++ b/arch/arm64/kvm/vgic/vgic-v5-tables.c @@ -1457,7 +1457,6 @@ static int vgic_v5_get_lpi_ist_desc(struct kvm *kvm, static int vgic_v5_save_linear_ist(const struct vgic_v5_ist_desc *ist, u32 __user *uaddr, size_t nr_entries) { - __le32 h_iste; size_t index; int ret; @@ -1466,8 +1465,9 @@ static int vgic_v5_save_linear_ist(const struct vgic_v5_ist_desc *ist, for (index = 0; index < nr_entries; index++) { __le32 *h_iste_addr = ist->base + index * ist->iste_size; + u32 h_iste; - h_iste = READ_ONCE(*h_iste_addr); + h_iste = le32_to_cpu(READ_ONCE(*h_iste_addr)); ret = put_user(h_iste, uaddr); if (ret) return ret; @@ -1520,7 +1520,7 @@ static int vgic_v5_save_two_level_ist(const struct vgic_v5_ist_desc *ist, h_iste = *(__le32 *)(h_l2_ist_base + h_l2_index * ist->iste_size); - ret = put_user(h_iste, uaddr); + ret = put_user(le32_to_cpu(h_iste), uaddr); if (ret) return ret; @@ -1710,19 +1710,19 @@ static int vgic_v5_restore_linear_ist(struct kvm *kvm, u32 __user *uaddr, size_t nr_entries, u32 intid_type) { - __le32 h_iste; size_t index; int ret; for (index = 0; index < nr_entries; index++) { void *h_iste_addr = ist->base + index * ist->iste_size; + u32 h_iste; ret = get_user(h_iste, uaddr); if (ret) return ret; ret = vgic_v5_restore_ist_entry(kvm, ist, h_iste_addr, - h_iste, index, intid_type); + cpu_to_le32(h_iste), index, intid_type); if (ret) return ret; @@ -1744,7 +1744,6 @@ static int vgic_v5_restore_two_level_ist(struct kvm *kvm, struct vgic_v5_two_level_ist_shape shape; size_t h_l1_index, h_l2_index; void *h_l2_ist_base; - __le32 h_iste; int ret; shape = vgic_v5_two_level_ist_shape(ist); @@ -1771,13 +1770,14 @@ static int vgic_v5_restore_two_level_ist(struct kvm *kvm, void *h_iste_addr = h_l2_ist_base + h_l2_index * ist->iste_size; u32 intid = h_l1_index * shape.l2_entries + h_l2_index; + u32 h_iste; ret = get_user(h_iste, uaddr); if (ret) return ret; ret = vgic_v5_restore_ist_entry(kvm, ist, h_iste_addr, - h_iste, intid, + cpu_to_le32(h_iste), intid, intid_type); if (ret) return ret; From 7c5397a6b7ac4fda87462d874bf39fe9f0ded55f Mon Sep 17 00:00:00 2001 From: Marc Zyngier Date: Sat, 19 Sep 2026 16:48:10 +0100 Subject: [PATCH 0117/1012] fixup! KVM: arm64: vgic-v5: Correctly handle host ISTE __le32 conversion Signed-off-by: Marc Zyngier --- arch/arm64/kvm/vgic/vgic-v5-tables.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/arch/arm64/kvm/vgic/vgic-v5-tables.c b/arch/arm64/kvm/vgic/vgic-v5-tables.c index aae5cf05f3907e..d577e485ebe9e8 100644 --- a/arch/arm64/kvm/vgic/vgic-v5-tables.c +++ b/arch/arm64/kvm/vgic/vgic-v5-tables.c @@ -1490,7 +1490,6 @@ static int vgic_v5_save_two_level_ist(const struct vgic_v5_ist_desc *ist, struct vgic_v5_two_level_ist_shape shape; size_t h_l1_index, h_l2_index; void *h_l2_ist_base; - __le32 h_iste; int ret; shape = vgic_v5_two_level_ist_shape(ist); @@ -1517,10 +1516,10 @@ static int vgic_v5_save_two_level_ist(const struct vgic_v5_ist_desc *ist, shape.l2_entries * ist->iste_size); for (h_l2_index = 0; h_l2_index < shape.l2_entries; h_l2_index++) { - h_iste = *(__le32 *)(h_l2_ist_base + - h_l2_index * ist->iste_size); + void *h_iste_addr = h_l2_ist_base + h_l2_index * ist->iste_size; + u32 h_iste = le32_to_cpu(*(__le32 *)h_iste_addr); - ret = put_user(le32_to_cpu(h_iste), uaddr); + ret = put_user(h_iste, uaddr); if (ret) return ret; From f394f933b7887f36f3af06c12d778a7bf8b21829 Mon Sep 17 00:00:00 2001 From: Yibo Tan Date: Sun, 13 Sep 2026 15:29:10 +0800 Subject: [PATCH 0118/1012] HID: sensor-hub: Fail unfinished multi-value reads on removal sensor_hub_remove() completes pending reads after stopping the HID device, but does not record why they completed. A successful completion wait therefore returns zero even if no complete input report was received. Multi-value IIO callers then format their untouched automatic buffer as a successful result. With a valid four-element signed 32-bit quaternion report descriptor, an unprivileged reader received all 16 bytes of the untouched buffer. Across 11 independent KASLR-enabled boots, four reads exposed exact pointers to dev_rot_channels or dev_sysfs_ops. Subtracting the matching link-time symbol address recovered the kernel KASLR slide in all four cases. The reader ran as UID/GID 65534 with no effective capabilities through the mode-0644 IIO attribute. The test used a privileged UHID broker to create and remove the provider; it does not demonstrate unprivileged provider removal. Mark a pending request as shut down before completing it from the removal path, and return -ENODEV from a multi-value read that observes the marker after a successful wait. Let removal win even if a response raced with teardown, since the device is no longer available. The Root B-only repair returned -ENODEV with no payload or kernel diagnostic in 3/3 matching signed-32-bit runs. The source reproducer, complete vulnerable and fixed serial logs, result tables, and checksums are available in [1]. Link: https://github.com/kimaiden1984-boop/linux-kernel-poc-collections/tree/main/cases/hid-sensor-quaternion-root-b-kaslr [1] Fixes: f784fcea4506 ("HID: sensor-hub: Add sensor_hub_input_attr_read_values() for multi-byte reads") Cc: stable@vger.kernel.org Suggested-by: Jonathan Cameron Assisted-by: LLM Signed-off-by: Yibo Tan Acked-by: Srinivas Pandruvada Signed-off-by: Jonathan Cameron --- drivers/hid/hid-sensor-hub.c | 6 +++++- include/linux/hid-sensor-hub.h | 2 ++ 2 files changed, 7 insertions(+), 1 deletion(-) diff --git a/drivers/hid/hid-sensor-hub.c b/drivers/hid/hid-sensor-hub.c index 6470a290ebfc5b..a9bd72218c070c 100644 --- a/drivers/hid/hid-sensor-hub.c +++ b/drivers/hid/hid-sensor-hub.c @@ -334,6 +334,8 @@ int sensor_hub_input_attr_read_values(struct hid_sensor_hub_device *hsdev, ret = -ETIMEDOUT; else if (cycles < 0) ret = cycles; + else if (hsdev->pending.shutdown) + ret = -ENODEV; hsdev->pending.status = false; } @@ -805,8 +807,10 @@ static int sensor_hub_finalize_pending_fn(struct device *dev, void *data) { struct hid_sensor_hub_device *hsdev = dev->platform_data; - if (hsdev->pending.status) + if (hsdev->pending.status) { + hsdev->pending.shutdown = true; complete(&hsdev->pending.ready); + } return 0; } diff --git a/include/linux/hid-sensor-hub.h b/include/linux/hid-sensor-hub.h index ab5cc8db3fbb3b..5aecf447418398 100644 --- a/include/linux/hid-sensor-hub.h +++ b/include/linux/hid-sensor-hub.h @@ -38,6 +38,7 @@ struct hid_sensor_hub_attribute_info { /** * struct sensor_hub_pending - Synchronous read pending information * @status: Pending status true/false. + * @shutdown: The device is being removed. * @ready: Completion synchronization data. * @usage_id: Usage id for physical device, e.g. gyro usage id. * @attr_usage_id: Usage Id of a field, e.g. X-axis for a gyro. @@ -48,6 +49,7 @@ struct hid_sensor_hub_attribute_info { */ struct sensor_hub_pending { bool status; + bool shutdown; struct completion ready; u32 usage_id; u32 attr_usage_id; From 81c671d59425effeca645a33a3d8b637526e97da Mon Sep 17 00:00:00 2001 From: Linmao Li Date: Mon, 24 Aug 2026 11:55:30 +0800 Subject: [PATCH 0119/1012] iio: imu: inv_icm42607: propagate runtime suspend errors The runtime suspend callback returns success when updating PWR_MGMT0 fails. The PM core then marks the device suspended even though the shutdown outcome is unknown. If the write did not reach the sensor, the sensors remain running and draw power while the device is idle. The condition need not persist. If the bus error clears, a later sensor access either reads the current PWR_MGMT0 value from hardware or retries programming the requested enabled mode, allowing normal operation to resume. Return the underlying errno to the PM core. -EAGAIN and -EBUSY retain their transient-error semantics; other errors put runtime PM into an error state and cause later PM acquires to fail until the status is explicitly reset. In the normal idle case, a successful system suspend can perform that reset. This leaves the transient-versus-fatal classification to the PM core and matches the sibling ICM-42600 driver. Keep a void wrapper for the managed teardown action, where errors can only be logged. Fixes: 3007c1530f96 ("iio: imu: inv_icm42607: Add PM support for icm42607") Signed-off-by: Linmao Li Tested-by: Kanak Shilledar Tested-by: Chris Morgan Signed-off-by: Jonathan Cameron --- drivers/iio/imu/inv_icm42607/inv_icm42607_core.c | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/drivers/iio/imu/inv_icm42607/inv_icm42607_core.c b/drivers/iio/imu/inv_icm42607/inv_icm42607_core.c index 190e998f7b8ef2..0da362967f63b2 100644 --- a/drivers/iio/imu/inv_icm42607/inv_icm42607_core.c +++ b/drivers/iio/imu/inv_icm42607/inv_icm42607_core.c @@ -537,9 +537,8 @@ static int inv_icm42607_enable_vddio_reg(struct inv_icm42607_state *st) return 0; } -static void inv_icm42607_sensors_off(void *_data) +static int inv_icm42607_sensors_off(struct inv_icm42607_state *st) { - struct inv_icm42607_state *st = _data; const struct device *dev = regmap_get_device(st->map); int ret; @@ -552,6 +551,13 @@ static void inv_icm42607_sensors_off(void *_data) st->conf.accel.mode); if (ret) dev_err(dev, "Unable to turn off sensors\n"); + + return ret; +} + +static void inv_icm42607_sensors_off_action(void *data) +{ + inv_icm42607_sensors_off(data); } static void inv_icm42607_disable_vddio_reg(void *_data) @@ -619,7 +625,7 @@ int inv_icm42607_core_probe(struct regmap *regmap, * Ensure if sensors get turned on at some point, they're turned off * as part of teardown. */ - ret = devm_add_action_or_reset(dev, inv_icm42607_sensors_off, st); + ret = devm_add_action_or_reset(dev, inv_icm42607_sensors_off_action, st); if (ret) return ret; @@ -688,8 +694,7 @@ static int inv_icm42607_runtime_suspend(struct device *dev) * however the tradeoff is that an unused sensor won't be * turned off until the entire chip is no longer in use. */ - inv_icm42607_sensors_off(st); - return 0; + return inv_icm42607_sensors_off(st); } EXPORT_NS_GPL_DEV_PM_OPS(inv_icm42607_pm_ops, IIO_ICM42607) = { From d1dd46640d415aa3ccdbb599e991b9b0b5029922 Mon Sep 17 00:00:00 2001 From: Linmao Li Date: Mon, 24 Aug 2026 11:55:31 +0800 Subject: [PATCH 0120/1012] iio: imu: inv_icm42607: restore runtime PM on system resume errors pm_runtime_force_suspend() leaves runtime PM disabled after it succeeds and expects pm_runtime_force_resume() to restore runtime PM management during system resume. The resume callback returns early if enabling the vddio regulator or synchronizing the register cache fails, skipping the matching pm_runtime_force_resume() call. Runtime PM consequently remains disabled after the system has resumed, so runtime autosuspend can no longer turn off sensors enabled afterward. Call pm_runtime_force_resume() on both error paths. Keep the first error as the return value and report a runtime PM restore failure separately. Fixes: 3007c1530f96 ("iio: imu: inv_icm42607: Add PM support for icm42607") Signed-off-by: Linmao Li Tested-by: Kanak Shilledar Tested-by: Chris Morgan Signed-off-by: Jonathan Cameron --- .../iio/imu/inv_icm42607/inv_icm42607_core.c | 23 +++++++++++++++---- 1 file changed, 19 insertions(+), 4 deletions(-) diff --git a/drivers/iio/imu/inv_icm42607/inv_icm42607_core.c b/drivers/iio/imu/inv_icm42607/inv_icm42607_core.c index 0da362967f63b2..f4ef75da22c762 100644 --- a/drivers/iio/imu/inv_icm42607/inv_icm42607_core.c +++ b/drivers/iio/imu/inv_icm42607/inv_icm42607_core.c @@ -664,9 +664,8 @@ static int inv_icm42607_suspend(struct device *dev) return 0; } -static int inv_icm42607_resume(struct device *dev) +static int inv_icm42607_resume_core(struct inv_icm42607_state *st) { - struct inv_icm42607_state *st = dev_get_drvdata(dev); int ret; ret = inv_icm42607_enable_vddio_reg(st); @@ -675,9 +674,25 @@ static int inv_icm42607_resume(struct device *dev) /* Sync the regcache again after regulator shutdown. */ regcache_mark_dirty(st->map); - ret = regcache_sync(st->map); - if (ret) + + return regcache_sync(st->map); +} + +static int inv_icm42607_resume(struct device *dev) +{ + struct inv_icm42607_state *st = dev_get_drvdata(dev); + int ret; + + ret = inv_icm42607_resume_core(st); + if (ret) { + int rc; + + rc = pm_runtime_force_resume(dev); + if (rc) + dev_warn(dev, "Failed to restore runtime PM state: %d\n", rc); + return ret; + } return pm_runtime_force_resume(dev); } From 411c85f2497a5c9256618c8543313fef2e140c43 Mon Sep 17 00:00:00 2001 From: Salah Triki Date: Thu, 17 Sep 2026 14:00:52 +0100 Subject: [PATCH 0121/1012] iio: proximity: isl29501: Fix return type of isl29501_register_write isl29501_register_write() was returning a u32 instead of an int. Since the function returns negative error codes (such as -ERANGE or return values from i2c_smbus_write_byte_data()), returning an unsigned integer type prevents callers from correctly checking for negative error conditions. Fix this by changing the function return type from u32 to int. This was found through manual code review. Fixes: 1c28799257bc ("iio: light: isl29501: Add support for the ISL29501 ToF sensor.") Signed-off-by: Salah Triki Reviewed-by: Joshua Crofts Cc: stable@vger.kernel.org Signed-off-by: Jonathan Cameron --- drivers/iio/proximity/isl29501.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/iio/proximity/isl29501.c b/drivers/iio/proximity/isl29501.c index 95fb7238f67887..a98a9975323c60 100644 --- a/drivers/iio/proximity/isl29501.c +++ b/drivers/iio/proximity/isl29501.c @@ -226,7 +226,7 @@ static int isl29501_register_read(struct isl29501_private *isl29501, return ret; } -static u32 isl29501_register_write(struct isl29501_private *isl29501, +static int isl29501_register_write(struct isl29501_private *isl29501, enum isl29501_register_name name, u32 value) { From 6af9b1ecbc21da9dcf493ee73bcbb3c806844fe4 Mon Sep 17 00:00:00 2001 From: Fabrice Gasnier Date: Wed, 16 Sep 2026 19:18:38 +0200 Subject: [PATCH 0122/1012] iio: adc: stm32-adc: fix check on internal channel availability If an unsupported internal channel like vddgpu is requested, the driver prints a warning but falls through and assigns it a valid int_ch below. This causes a problem later during setup: stm32_adc_int_ch_enable() { ... case STM32_ADC_INT_CH_VDDGPU: stm32_adc_set_bits(adc, adc->cfg->regs->or_vddgpu.reg, adc->cfg->regs->or_vddgpu.mask); ... } Because the register offset is uninitialized (0), this performs a read-modify-write on offset 0, which corresponds to the ISR register. Fix this by returning before a valid int_ch is assigned. Choice is made to keep current driver behavior to warn about the channel name. Fixes: cf0fb80ae167 ("iio: adc: stm32-adc: add stm32mp13 support") Reported-by: Sashiko Closes: https://lore.kernel.org/all/20260911162602.D323F1F000FF@smtp.kernel.org/ Cc: stable@vger.kernel.org Reviewed-by: Andy Shevchenko Signed-off-by: Fabrice Gasnier Signed-off-by: Jonathan Cameron --- drivers/iio/adc/stm32-adc.c | 34 +++++++++++++++++++--------------- 1 file changed, 19 insertions(+), 15 deletions(-) diff --git a/drivers/iio/adc/stm32-adc.c b/drivers/iio/adc/stm32-adc.c index 5c6c06b269be6e..b77f39f305d894 100644 --- a/drivers/iio/adc/stm32-adc.c +++ b/drivers/iio/adc/stm32-adc.c @@ -2263,33 +2263,37 @@ static int stm32_adc_populate_int_ch(struct iio_dev *indio_dev, const char *ch_n for (i = 0; i < STM32_ADC_INT_CH_NB; i++) { if (!strncmp(stm32_adc_ic[i].name, ch_name, STM32_ADC_CH_SZ)) { + bool na; + /* Check internal channel availability */ switch (i) { case STM32_ADC_INT_CH_VDDCORE: - if (!adc->cfg->regs->or_vddcore.reg) - dev_warn(&indio_dev->dev, - "%s channel not available\n", ch_name); + na = !adc->cfg->regs->or_vddcore.reg; break; case STM32_ADC_INT_CH_VDDCPU: - if (!adc->cfg->regs->or_vddcpu.reg) - dev_warn(&indio_dev->dev, - "%s channel not available\n", ch_name); + na = !adc->cfg->regs->or_vddcpu.reg; break; case STM32_ADC_INT_CH_VDDQ_DDR: - if (!adc->cfg->regs->or_vddq_ddr.reg) - dev_warn(&indio_dev->dev, - "%s channel not available\n", ch_name); + na = !adc->cfg->regs->or_vddq_ddr.reg; break; case STM32_ADC_INT_CH_VREFINT: - if (!adc->cfg->regs->ccr_vref.reg) - dev_warn(&indio_dev->dev, - "%s channel not available\n", ch_name); + na = !adc->cfg->regs->ccr_vref.reg; break; case STM32_ADC_INT_CH_VBAT: - if (!adc->cfg->regs->ccr_vbat.reg) - dev_warn(&indio_dev->dev, - "%s channel not available\n", ch_name); + na = !adc->cfg->regs->ccr_vbat.reg; break; + default: + return -EINVAL; + } + + if (na) { + /* + * Channel label matches an internal STM32 ADC channel. + * Warn about it, as there's normally no restriction on the + * name but that's not among available internal channels. + */ + dev_warn(&indio_dev->dev, "no %s internal channel\n", ch_name); + return 0; } if (stm32_adc_ic[i].idx != STM32_ADC_INT_CH_VREFINT) { From 60a3152c340993714fd693bba0fcf9335b29403e Mon Sep 17 00:00:00 2001 From: Fabrice Gasnier Date: Wed, 16 Sep 2026 19:15:04 +0200 Subject: [PATCH 0123/1012] iio: adc: stm32-adc: fix possible division by zero in processed channel In case the conversion has failed or returned zero, processing *val can lead to a division by zero. Need to check for errors, or converted value is zero, before processing the data. In case the converted value is zero, e.g. the Vrefint channel, this should be considered as invalid in all cases. Fixes: 0e346b2cfa85 ("iio: adc: stm32-adc: add vrefint calibration support") Reported-by: Sashiko Closes: https://lore.kernel.org/all/20260911161555.244F31F000FF@smtp.kernel.org/ Cc: stable@vger.kernel.org Signed-off-by: Fabrice Gasnier Reviewed-by: Andy Shevchenko Signed-off-by: Jonathan Cameron --- drivers/iio/adc/stm32-adc.c | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/drivers/iio/adc/stm32-adc.c b/drivers/iio/adc/stm32-adc.c index b77f39f305d894..90f0e257e6d1b6 100644 --- a/drivers/iio/adc/stm32-adc.c +++ b/drivers/iio/adc/stm32-adc.c @@ -1608,11 +1608,16 @@ static int stm32_adc_read_raw(struct iio_dev *indio_dev, ret = stm32_adc_single_conv(indio_dev, chan, val); else ret = -EINVAL; + iio_device_release_direct(indio_dev); + if (ret < 0) + return ret; - if (mask == IIO_CHAN_INFO_PROCESSED) + if (mask == IIO_CHAN_INFO_PROCESSED) { + if (*val == 0) + return -EINVAL; *val = STM32_ADC_VREFINT_VOLTAGE * adc->vrefint.vrefint_cal / *val; + } - iio_device_release_direct(indio_dev); return ret; case IIO_CHAN_INFO_SCALE: From 81e86a30f7618e688fa66ace9f67185f87545956 Mon Sep 17 00:00:00 2001 From: Marcelo Schmitt Date: Wed, 16 Sep 2026 14:01:34 -0300 Subject: [PATCH 0124/1012] iio: adc: ad7173: Fix digital filter configuration Filter enable and filter type selection masks were swapped on data preparation for filter configuration register write. Use the correct masks to set each property of post filter configuration. Fixes: ff06b39be1a1 ("iio: adc: ad7173: support changing filter type") Signed-off-by: Marcelo Schmitt Cc: stable@vger.kernel.org Signed-off-by: Jonathan Cameron --- drivers/iio/adc/ad7173.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/iio/adc/ad7173.c b/drivers/iio/adc/ad7173.c index eb47175a052870..faebb7b34a1519 100644 --- a/drivers/iio/adc/ad7173.c +++ b/drivers/iio/adc/ad7173.c @@ -775,9 +775,9 @@ static int ad7173_load_config(struct ad7173_state *st, return ad_sd_write_reg(&st->sd, AD7173_REG_FILTER(free_cfg_slot), 2, FIELD_PREP(AD7173_FILTER_SINC3_MAP, 0) | - FIELD_PREP(AD7173_FILTER_ENHFILT_MASK, - post_filter_enable) | FIELD_PREP(AD7173_FILTER_ENHFILTEN, + post_filter_enable) | + FIELD_PREP(AD7173_FILTER_ENHFILT_MASK, post_filter_select) | FIELD_PREP(AD7173_FILTER_ORDER, 0) | FIELD_PREP(AD7173_FILTER_ODR_MASK, From 182c47735554925da9605dc9dd90a8c15a1bbae5 Mon Sep 17 00:00:00 2001 From: Salah Triki Date: Thu, 3 Sep 2026 12:43:09 +0100 Subject: [PATCH 0125/1012] iio: adc: ad4030: fix invalid oversampling_ratio validation In ad4030_set_avg_frame_len(), the logarithm is calculated before input validation. Passing zero or negative values leads to an undefined result from ilog2(). Validate that the input is strictly positive prior to computing its logarithm. Fixes: 949abd1ca5a4 ("iio: adc: ad4030: add averaging support") Assisted-by: LLM Signed-off-by: Salah Triki Reviewed-by: Andy Shevchenko Cc: stable@vger.kernel.org Signed-off-by: Jonathan Cameron --- drivers/iio/adc/ad4030.c | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/drivers/iio/adc/ad4030.c b/drivers/iio/adc/ad4030.c index e97400a1aa4812..877b06007396f0 100644 --- a/drivers/iio/adc/ad4030.c +++ b/drivers/iio/adc/ad4030.c @@ -746,14 +746,21 @@ static int ad4030_set_chan_calibbias(struct iio_dev *indio_dev, static int ad4030_set_avg_frame_len(struct iio_dev *dev, int avg_val) { struct ad4030_state *st = iio_priv(dev); - unsigned int avg_log2 = ilog2(avg_val); unsigned int last_avg_idx = ARRAY_SIZE(ad4030_average_modes) - 1; + unsigned int avg_log2; int freq_hz; int ret; - if (avg_val < 0 || avg_val > ad4030_average_modes[last_avg_idx]) + /* Reject unsupported modes */ + if (avg_val > ad4030_average_modes[last_avg_idx]) + return -EINVAL; + + /* Avoid invalid values for logarithm since it's undefined */ + if (avg_val < 1) return -EINVAL; + avg_log2 = ilog2(avg_val); + if (st->offload_trigger) { /* * The sample averaging and sampling frequency configurations From d249ff58f054148ca55c9cddfa5c2d6c13fd3366 Mon Sep 17 00:00:00 2001 From: Linmao Li Date: Wed, 26 Aug 2026 16:31:56 +0800 Subject: [PATCH 0126/1012] iio: adc: ade9000: fix NULL pointer dereference in clkout registration ade9000_setup_clkout() passes NULL as the register address when registering a divider clock. During clock registration, the common clock framework calls clk_divider_recalc_rate(), which dereferences the address through readl(). As a result, probing an ADE9000 configured as a clock provider with an external input clock crashes. CLKOUT passes CLKIN through without changing its rate. Register it as a 1:1 fixed-factor clock, which does not require register access. This change does not affect the configuration using the internal clock, for which the driver does not register a clock provider. Fixes: 81de7b4619fc ("iio: adc: add ade9000 support") Cc: stable@vger.kernel.org Signed-off-by: Linmao Li Reviewed-by: Antoniu Miclaus Signed-off-by: Jonathan Cameron --- drivers/iio/adc/ade9000.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/drivers/iio/adc/ade9000.c b/drivers/iio/adc/ade9000.c index da6caabfe2a41c..4fc0eb7e70a887 100644 --- a/drivers/iio/adc/ade9000.c +++ b/drivers/iio/adc/ade9000.c @@ -1647,8 +1647,9 @@ static int ade9000_setup_clkout(struct device *dev, struct ade9000_state *st) return 0; /* CLKOUT passes through CLKIN with divider of 1 */ - clkout_hw = devm_clk_hw_register_divider(dev, "clkout", __clk_get_name(st->clkin), - CLK_SET_RATE_PARENT, NULL, 0, 1, 0, NULL); + clkout_hw = devm_clk_hw_register_fixed_factor(dev, "clkout", + __clk_get_name(st->clkin), + CLK_SET_RATE_PARENT, 1, 1); if (IS_ERR(clkout_hw)) return dev_err_probe(dev, PTR_ERR(clkout_hw), "Failed to register clkout"); From 5b5cb8c87066a335d320b5efedf5a46d0b72052f Mon Sep 17 00:00:00 2001 From: Thomas Huth Date: Wed, 16 Sep 2026 11:50:03 +0200 Subject: [PATCH 0127/1012] lib/crypto: aes: Provide functions for zeroizing aes_key and aes_enckey Some crypto functions need to zeroize their local aes_key or aes_enckey structures after use to avoid leaking sensitive material on the stack. Provide aes_zeroize_key() and aes_zeroize_enckey() helper functions that can be used with __cleanup() to automatically zeroize the structs when they go out of scope. While we're at it, replace the memzero_explicit() calls in lib/crypto/aes.c with the new helper functions. Signed-off-by: Thomas Huth Link: https://patch.msgid.link/20260916095022.604354-2-thuth@redhat.com Signed-off-by: Eric Biggers --- include/crypto/aes.h | 18 ++++++++++++++++++ lib/crypto/aes.c | 10 +++++----- 2 files changed, 23 insertions(+), 5 deletions(-) diff --git a/include/crypto/aes.h b/include/crypto/aes.h index 3279cfa5460854..9fe868161e1d38 100644 --- a/include/crypto/aes.h +++ b/include/crypto/aes.h @@ -101,6 +101,15 @@ struct aes_enckey { union aes_enckey_arch k; }; +/** + * aes_zeroize_enckey() - Zeroize an aes_enckey structure + * @key: The aes_enckey to zeroize + */ +static inline void aes_zeroize_enckey(struct aes_enckey *key) +{ + memzero_explicit(key, sizeof(*key)); +} + /** * struct aes_key - An AES key prepared for encryption and decryption * @aes_enckey: Common fields and the key prepared for encryption @@ -115,6 +124,15 @@ struct aes_key { union aes_invkey_arch inv_k; }; +/** + * aes_zeroize_key() - Zeroize an aes_key structure + * @key: The aes_key to zeroize + */ +static inline void aes_zeroize_key(struct aes_key *key) +{ + memzero_explicit(key, sizeof(*key)); +} + /* * Please ensure that the first two fields are 16-byte aligned * relative to the start of the structure, i.e., don't move them! diff --git a/lib/crypto/aes.c b/lib/crypto/aes.c index f1549839b3de0c..07c1d912ac365b 100644 --- a/lib/crypto/aes.c +++ b/lib/crypto/aes.c @@ -539,7 +539,7 @@ static void __init aes_fips_test(void) if (memcmp(fips_test_data, data, sizeof(data)) != 0) panic("aes: FIPS self-test failed (wrong plaintext)\n"); - memzero_explicit(&key, sizeof(key)); + aes_zeroize_key(&key); } #if IS_ENABLED(CONFIG_CRYPTO_LIB_AES_CBC_MACS) @@ -827,7 +827,7 @@ static void __init aes_ecb_fips_test(void) if (memcmp(fips_test_data, data, sizeof(data)) != 0) panic("aes: ECB FIPS self-test failed (wrong plaintext)\n"); - memzero_explicit(&key, sizeof(key)); + aes_zeroize_key(&key); } #else /* CONFIG_CRYPTO_LIB_AES_ECB */ static inline void aes_ecb_fips_test(void) @@ -1040,7 +1040,7 @@ static void __init aes_cbc_fips_test(void) if (memcmp(fips_test_data, data, sizeof(data)) != 0) panic("aes: CBC FIPS self-test failed (wrong plaintext)\n"); - memzero_explicit(&key, sizeof(key)); + aes_zeroize_key(&key); } /* FIPS cryptographic algorithm self-test for AES-CBC-CTS */ @@ -1069,7 +1069,7 @@ static void __init aes_cbc_cts_fips_test(void) if (memcmp(ptext, data, data_len) != 0) panic("aes: CBC-CTS FIPS self-test failed (wrong plaintext)\n"); - memzero_explicit(&key, sizeof(key)); + aes_zeroize_key(&key); } #else /* CONFIG_CRYPTO_LIB_AES_CBC */ static inline void aes_cbc_fips_test(void) @@ -1194,7 +1194,7 @@ static void __init aes_ctr_fips_test(void) if (memcmp(fips_test_data, data, sizeof(data)) != 0) panic("aes: CTR FIPS self-test failed (wrong plaintext)\n"); - memzero_explicit(&key, sizeof(key)); + aes_zeroize_enckey(&key); } #else /* CONFIG_CRYPTO_LIB_AES_CTR */ static inline void aes_ctr_fips_test(void) From 4fa5b8784e6c3607ded1f63a0a8e1ce2c57c41a0 Mon Sep 17 00:00:00 2001 From: Thomas Huth Date: Wed, 16 Sep 2026 11:50:04 +0200 Subject: [PATCH 0128/1012] lib/crypto: aes-xts: Provide function for zeroizing aes_xts_key In certain cases crypto code functions need to zeroize their local aes_xts_key structures after use to avoid leaking sensitive material. Provide an aes_xts_zeroize_key() helper function that e.g. can be used with __cleanup() to automatically zeroize the struct when it goes out of scope. While we're at it, replace the related memzero_explicit() call in lib/crypto/aes.c with a call to the new helper function. Signed-off-by: Thomas Huth Link: https://patch.msgid.link/20260916095022.604354-3-thuth@redhat.com Signed-off-by: Eric Biggers --- include/crypto/aes-xts.h | 13 +++++++++++-- lib/crypto/aes.c | 2 +- 2 files changed, 12 insertions(+), 3 deletions(-) diff --git a/include/crypto/aes-xts.h b/include/crypto/aes-xts.h index b9e828265e58ab..3a52e1cf40b57a 100644 --- a/include/crypto/aes-xts.h +++ b/include/crypto/aes-xts.h @@ -22,6 +22,15 @@ struct aes_xts_key { struct aes_enckey tweak_key; }; +/** + * aes_xts_zeroize_key() - Zeroize an aes_xts_key structure + * @key: The aes_xts_key to zeroize + */ +static inline void aes_xts_zeroize_key(struct aes_xts_key *key) +{ + memzero_explicit(key, sizeof(*key)); +} + /** * aes_xts_preparekey() - Prepare a key for AES-XTS encryption and decryption * @key: (output) The key structure to initialize @@ -30,8 +39,8 @@ struct aes_xts_key { * @flags: Optional flag XTS_FORBID_WEAK_KEYS to forbid keys whose two halves * are the same. * - * Users should use memzero_explicit() to zeroize the key struct at the end of - * its lifetime. (But if this function fails, zeroization is unnecessary.) + * Users should use aes_xts_zeroize_key() to zeroize the key struct at the end + * of its lifetime. (But if this function fails, zeroization is unnecessary.) * * Context: Any context. * Return: diff --git a/lib/crypto/aes.c b/lib/crypto/aes.c index 07c1d912ac365b..34ef5deca0a79f 100644 --- a/lib/crypto/aes.c +++ b/lib/crypto/aes.c @@ -1223,7 +1223,7 @@ int aes_xts_preparekey(struct aes_xts_key *key, const u8 *in_key, return 0; out_zeroize: - memzero_explicit(key, sizeof(*key)); + aes_xts_zeroize_key(key); return err; } EXPORT_SYMBOL_GPL(aes_xts_preparekey); From 662b355994ccc184550ca9fa2be481288007df5d Mon Sep 17 00:00:00 2001 From: Thomas Huth Date: Wed, 16 Sep 2026 11:50:05 +0200 Subject: [PATCH 0129/1012] lib/crypto: aes-gcm: Provide functions for zeroizing aes_gcm* structures In certain cases crypto code needs to zeroize their local aes_gcm_key or aes_gcm_ctx structures after use to avoid leaking sensitive material. Provide aes_gcm_zeroize_key() and aes_gcm_zeroize_ctx() helper functions that e.g. can be used with __cleanup() to automatically zeroize the structures when they go out of scope. While we're at it, replace the related memzero_explicit() calls in lib/crypto/aes.c with calls to the new helper functions. Signed-off-by: Thomas Huth Link: https://patch.msgid.link/20260916095022.604354-4-thuth@redhat.com Signed-off-by: Eric Biggers --- include/crypto/aes-gcm.h | 22 ++++++++++++++++++++-- lib/crypto/aes.c | 8 +++----- 2 files changed, 23 insertions(+), 7 deletions(-) diff --git a/include/crypto/aes-gcm.h b/include/crypto/aes-gcm.h index 2aee62f0198919..a81b00fd8e27f6 100644 --- a/include/crypto/aes-gcm.h +++ b/include/crypto/aes-gcm.h @@ -21,6 +21,15 @@ struct aes_gcm_key { size_t authtag_len; /* Length of authentication tags in bytes */ }; +/** + * aes_gcm_zeroize_key() - Zeroize an aes_gcm_key structure + * @key: The aes_gcm_key to zeroize + */ +static inline void aes_gcm_zeroize_key(struct aes_gcm_key *key) +{ + memzero_explicit(key, sizeof(*key)); +} + /** * struct aes_gcm_ctx - Context for incrementally en/decrypting a message */ @@ -58,6 +67,15 @@ struct aes_gcm_ctx { u64 data_len; }; +/** + * aes_gcm_zeroize_ctx() - Zeroize an aes_gcm_ctx structure + * @ctx: The aes_gcm_ctx to zeroize + */ +static inline void aes_gcm_zeroize_ctx(struct aes_gcm_ctx *ctx) +{ + memzero_explicit(ctx, sizeof(*ctx)); +} + /** * aes_gcm_preparekey() - Prepare a key for AES-GCM encryption and decryption * @key: (output) The key structure to initialize @@ -66,8 +84,8 @@ struct aes_gcm_ctx { * @authtag_len: Length of the authentication tag in bytes: * 4, 8, 12, 13, 14, 15, or 16. 16 is recommended. * - * Users should use memzero_explicit() to zeroize the key struct at the end of - * its lifetime. (But if this function fails, zeroization is unnecessary.) + * Users should use aes_gcm_zeroize_key() to zeroize the key struct at the end + * of its lifetime. (But if this function fails, zeroization is unnecessary.) * * Context: Any context. * Return: diff --git a/lib/crypto/aes.c b/lib/crypto/aes.c index 34ef5deca0a79f..0cb5d7355926e2 100644 --- a/lib/crypto/aes.c +++ b/lib/crypto/aes.c @@ -1670,7 +1670,7 @@ void aes_gcm_encrypt_final(struct aes_gcm_ctx *ctx, u8 *authtag) ghash_final(&ctx->ghash, ctx->ctr); /* Use ctr as temp buffer */ crypto_xor_cpy(authtag, ctx->ctr, ctx->j0_enc, ctx->key->authtag_len); - memzero_explicit(ctx, sizeof(*ctx)); + aes_gcm_zeroize_ctx(ctx); } EXPORT_SYMBOL_GPL(aes_gcm_encrypt_final); @@ -1697,7 +1697,7 @@ int aes_gcm_decrypt_final(struct aes_gcm_ctx *ctx, const u8 *authtag) -EBADMSG : 0; out: - memzero_explicit(ctx, sizeof(*ctx)); + aes_gcm_zeroize_ctx(ctx); return err; } EXPORT_SYMBOL_GPL(aes_gcm_decrypt_final); @@ -1742,7 +1742,7 @@ static void __init aes_gcm_fips_test(void) { const size_t data_len = sizeof(fips_test_data); u8 buf[sizeof(fips_test_data) + AES_BLOCK_SIZE]; - struct aes_gcm_key key; + struct aes_gcm_key key __cleanup(aes_gcm_zeroize_key); int err; if (aes_gcm_preparekey(&key, fips_test_key, sizeof(fips_test_key), @@ -1760,8 +1760,6 @@ static void __init aes_gcm_fips_test(void) panic("aes: GCM FIPS self-test failed (decryption failed)\n"); if (memcmp(fips_test_data, buf, data_len) != 0) panic("aes: GCM FIPS self-test failed (wrong plaintext)\n"); - - memzero_explicit(&key, sizeof(key)); } #else /* CONFIG_CRYPTO_LIB_AES_GCM */ static inline void aes_gcm_fips_test(void) From df38b4a06183f93feb0ea2e73036c2418278c197 Mon Sep 17 00:00:00 2001 From: Thomas Huth Date: Wed, 16 Sep 2026 11:50:06 +0200 Subject: [PATCH 0130/1012] lib/crypto: aes-ccm: Provide functions for zeroizing aes_ccm* structures In certain cases crypto code needs to zeroize their local aes_ccm_key or aes_ccm_ctx structures after use to avoid leaking sensitive material. Provide aes_ccm_zeroize_key() and aes_ccm_zeroize_ctx() helper functions that e.g. can be used with __cleanup() to automatically zeroize the structures when they go out of scope. While we're at it, replace the related memzero_explicit() calls in lib/crypto/aes.c with calls to the new helper functions. Signed-off-by: Thomas Huth Link: https://patch.msgid.link/20260916095022.604354-5-thuth@redhat.com Signed-off-by: Eric Biggers --- include/crypto/aes-ccm.h | 22 ++++++++++++++++++++-- lib/crypto/aes.c | 8 +++----- 2 files changed, 23 insertions(+), 7 deletions(-) diff --git a/include/crypto/aes-ccm.h b/include/crypto/aes-ccm.h index 8b00859ac4d6bd..c52982dd91d2c8 100644 --- a/include/crypto/aes-ccm.h +++ b/include/crypto/aes-ccm.h @@ -18,6 +18,15 @@ struct aes_ccm_key { size_t authtag_len; /* Length of authentication tags in bytes */ }; +/** + * aes_ccm_zeroize_key() - Zeroize an aes_ccm_key structure + * @key: The aes_ccm_key to zeroize + */ +static inline void aes_ccm_zeroize_key(struct aes_ccm_key *key) +{ + memzero_explicit(key, sizeof(*key)); +} + /** * struct aes_ccm_ctx - Context for incrementally en/decrypting a message */ @@ -50,6 +59,15 @@ struct aes_ccm_ctx { bool ad_padded; }; +/** + * aes_ccm_zeroize_ctx() - Zeroize an aes_ccm_ctx structure + * @ctx: The aes_ccm_ctx to zeroize + */ +static inline void aes_ccm_zeroize_ctx(struct aes_ccm_ctx *ctx) +{ + memzero_explicit(ctx, sizeof(*ctx)); +} + /** * aes_ccm_preparekey() - Prepare a key for AES-CCM encryption and decryption * @key: (output) The key structure to initialize @@ -58,8 +76,8 @@ struct aes_ccm_ctx { * @authtag_len: Length of the authentication tag in bytes: * 4, 6, 8, 10, 12, 14, or 16. 16 is recommended. * - * Users should use memzero_explicit() to zeroize the key struct at the end of - * its lifetime. (But if this function fails, zeroization is unnecessary.) + * Users should use aes_ccm_zeroize_key() to zeroize the key struct at the end + * of its lifetime. (But if this function fails, zeroization is unnecessary.) * * Context: Any context. * Return: diff --git a/lib/crypto/aes.c b/lib/crypto/aes.c index 0cb5d7355926e2..2d29adca795328 100644 --- a/lib/crypto/aes.c +++ b/lib/crypto/aes.c @@ -2011,7 +2011,7 @@ void aes_ccm_encrypt_final(struct aes_ccm_ctx *ctx, u8 *authtag) if (ctx->partial_len) aes_encrypt(&ctx->key->aes, ctx->mac, ctx->mac); crypto_xor_cpy(authtag, ctx->mac, ctx->s0, ctx->key->authtag_len); - memzero_explicit(ctx, sizeof(*ctx)); + aes_ccm_zeroize_ctx(ctx); } EXPORT_SYMBOL_GPL(aes_ccm_encrypt_final); @@ -2032,7 +2032,7 @@ int aes_ccm_decrypt_final(struct aes_ccm_ctx *ctx, const u8 *authtag) -EBADMSG : 0; out: - memzero_explicit(ctx, sizeof(*ctx)); + aes_ccm_zeroize_ctx(ctx); return err; } EXPORT_SYMBOL_GPL(aes_ccm_decrypt_final); @@ -2084,7 +2084,7 @@ static void __init aes_ccm_fips_test(void) const size_t data_len = sizeof(fips_test_data); const size_t nonce_len = 13; u8 buf[sizeof(fips_test_data) + AES_BLOCK_SIZE]; - struct aes_ccm_key key; + struct aes_ccm_key key __cleanup(aes_ccm_zeroize_key); int err; if (aes_ccm_preparekey(&key, fips_test_key, sizeof(fips_test_key), @@ -2106,8 +2106,6 @@ static void __init aes_ccm_fips_test(void) panic("aes: CCM FIPS self-test failed (decryption failed)\n"); if (memcmp(fips_test_data, buf, data_len) != 0) panic("aes: CCM FIPS self-test failed (wrong plaintext)\n"); - - memzero_explicit(&key, sizeof(key)); } #else /* CONFIG_CRYPTO_LIB_AES_CCM */ static inline void aes_ccm_fips_test(void) From bd1ed2bb0c0b292fe8297db09144c44ba6e1922d Mon Sep 17 00:00:00 2001 From: Thomas Huth Date: Wed, 16 Sep 2026 11:50:07 +0200 Subject: [PATCH 0131/1012] lib/crypto: md5: Provide functions for zeroizing hmac_md5 structures In certain cases crypto code functions need to zeroize their local hmac_md5_key or hmac_md5_ctx structures after use to avoid leaking sensitive material on the stack. Provide hmac_md5_zeroize_key() and hmac_md5_zeroize_ctx() helper functions that e.g. can be used with __cleanup() to automatically zeroize the structure when it goes out of scope. While we're at it, replace the related memzero_explicit() call in lib/crypto/md5.c with a call to the new helper function. Signed-off-by: Thomas Huth Link: https://patch.msgid.link/20260916095022.604354-6-thuth@redhat.com Signed-off-by: Eric Biggers --- include/crypto/md5.h | 19 +++++++++++++++++++ lib/crypto/md5.c | 2 +- 2 files changed, 20 insertions(+), 1 deletion(-) diff --git a/include/crypto/md5.h b/include/crypto/md5.h index c47aedfe67ecd0..1ed89c15b662c9 100644 --- a/include/crypto/md5.h +++ b/include/crypto/md5.h @@ -4,6 +4,7 @@ #include #include +#include #define MD5_DIGEST_SIZE 16 #define MD5_HMAC_BLOCK_SIZE 64 @@ -98,6 +99,15 @@ struct hmac_md5_key { struct md5_block_state ostate; }; +/** + * hmac_md5_zeroize_key() - Zeroize an hmac_md5_key structure + * @key: The hmac_md5_key to zeroize + */ +static inline void hmac_md5_zeroize_key(struct hmac_md5_key *key) +{ + memzero_explicit(key, sizeof(*key)); +} + /** * struct hmac_md5_ctx - Context for computing HMAC-MD5 of a message * @hash_ctx: private @@ -108,6 +118,15 @@ struct hmac_md5_ctx { struct md5_block_state ostate; }; +/** + * hmac_md5_zeroize_ctx() - Zeroize an hmac_md5_ctx structure + * @ctx: The hmac_md5_ctx context to zeroize + */ +static inline void hmac_md5_zeroize_ctx(struct hmac_md5_ctx *ctx) +{ + memzero_explicit(ctx, sizeof(*ctx)); +} + /** * hmac_md5_preparekey() - Prepare a key for HMAC-MD5 * @key: (output) the key structure to initialize diff --git a/lib/crypto/md5.c b/lib/crypto/md5.c index 3d2b017a0525ae..a8ee57600012d5 100644 --- a/lib/crypto/md5.c +++ b/lib/crypto/md5.c @@ -271,7 +271,7 @@ void hmac_md5_final(struct hmac_md5_ctx *ctx, u8 out[MD5_DIGEST_SIZE]) cpu_to_le32_array(ctx->ostate.h, ARRAY_SIZE(ctx->ostate.h)); memcpy(out, ctx->ostate.h, MD5_DIGEST_SIZE); - memzero_explicit(ctx, sizeof(*ctx)); + hmac_md5_zeroize_ctx(ctx); } EXPORT_SYMBOL_GPL(hmac_md5_final); From 258a4ae2d8f1d29f703f0e5f80ac9b657237b547 Mon Sep 17 00:00:00 2001 From: Thomas Huth Date: Wed, 16 Sep 2026 11:50:08 +0200 Subject: [PATCH 0132/1012] lib/crypto: sm3: Provide a function for zeroizing the sm3_ctx structure In certain cases crypto code functions need to zeroize their local sm3_ctx structure after use to avoid leaking sensitive material. Provide a sm3_zeroize_ctx() helper function that e.g. can be used with __cleanup() to automatically zeroize the structure when it goes out of scope. While we're at it, replace the related memzero_explicit() call in lib/crypto/sm3.c with a call to the new helper function. Signed-off-by: Thomas Huth Link: https://patch.msgid.link/20260916095022.604354-7-thuth@redhat.com Signed-off-by: Eric Biggers --- include/crypto/sm3.h | 10 ++++++++++ lib/crypto/sm3.c | 2 +- 2 files changed, 11 insertions(+), 1 deletion(-) diff --git a/include/crypto/sm3.h b/include/crypto/sm3.h index 371e8a66170546..d044ca80213ba2 100644 --- a/include/crypto/sm3.h +++ b/include/crypto/sm3.h @@ -11,6 +11,7 @@ #define _CRYPTO_SM3_H #include +#include #define SM3_DIGEST_SIZE 32 #define SM3_BLOCK_SIZE 64 @@ -41,6 +42,15 @@ struct sm3_ctx { u8 buf[SM3_BLOCK_SIZE] __aligned(__alignof__(__be64)); }; +/** + * sm3_zeroize_ctx() - Zeroize an sm3_ctx structure + * @ctx: The sm3_ctx to zeroize + */ +static inline void sm3_zeroize_ctx(struct sm3_ctx *ctx) +{ + memzero_explicit(ctx, sizeof(*ctx)); +} + /** * sm3_init() - Initialize an SM3 context for a new message * @ctx: the context to initialize diff --git a/lib/crypto/sm3.c b/lib/crypto/sm3.c index b02b8a247adf23..23059347b4493d 100644 --- a/lib/crypto/sm3.c +++ b/lib/crypto/sm3.c @@ -258,7 +258,7 @@ static void __sm3_final(struct sm3_ctx *ctx, u8 out[SM3_DIGEST_SIZE]) void sm3_final(struct sm3_ctx *ctx, u8 out[SM3_DIGEST_SIZE]) { __sm3_final(ctx, out); - memzero_explicit(ctx, sizeof(*ctx)); + sm3_zeroize_ctx(ctx); } EXPORT_SYMBOL_GPL(sm3_final); From 142595952334a782f77e61c11eb15712ff2fbdce Mon Sep 17 00:00:00 2001 From: Thomas Huth Date: Wed, 16 Sep 2026 11:50:09 +0200 Subject: [PATCH 0133/1012] lib/crypto: blake2: Provide functions for zeroizing blake2*_ctx structures In certain cases crypto code needs to zeroize their local blake2b_ctx or blake2s_ctx structures after use to avoid leaking sensitive material. Provide blake2b_zeroize_ctx() and blake2s_zeroize_ctx() helper functions that e.g. can be used with __cleanup() to automatically zeroize the structures when they go out of scope. While we're at it, replace the related memzero_explicit() calls in lib/crypto/blake2*.c with calls to the new helper functions. Signed-off-by: Thomas Huth Link: https://patch.msgid.link/20260916095022.604354-8-thuth@redhat.com Signed-off-by: Eric Biggers --- include/crypto/blake2b.h | 9 +++++++++ include/crypto/blake2s.h | 9 +++++++++ lib/crypto/blake2b.c | 2 +- lib/crypto/blake2s.c | 2 +- 4 files changed, 20 insertions(+), 2 deletions(-) diff --git a/include/crypto/blake2b.h b/include/crypto/blake2b.h index 3bc37fd103a7ab..eda1604bce780e 100644 --- a/include/crypto/blake2b.h +++ b/include/crypto/blake2b.h @@ -37,6 +37,15 @@ struct blake2b_ctx { unsigned int outlen; }; +/** + * blake2b_zeroize_ctx() - Zeroize a blake2b_ctx structure + * @ctx: The blake2b_ctx to zeroize + */ +static inline void blake2b_zeroize_ctx(struct blake2b_ctx *ctx) +{ + memzero_explicit(ctx, sizeof(*ctx)); +} + enum blake2b_iv { BLAKE2B_IV0 = 0x6A09E667F3BCC908ULL, BLAKE2B_IV1 = 0xBB67AE8584CAA73BULL, diff --git a/include/crypto/blake2s.h b/include/crypto/blake2s.h index 648cb782435887..bb4e6870ed1960 100644 --- a/include/crypto/blake2s.h +++ b/include/crypto/blake2s.h @@ -41,6 +41,15 @@ struct blake2s_ctx { unsigned int outlen; }; +/** + * blake2s_zeroize_ctx() - Zeroize a blake2s_ctx structure + * @ctx: The blake2s_ctx to zeroize + */ +static inline void blake2s_zeroize_ctx(struct blake2s_ctx *ctx) +{ + memzero_explicit(ctx, sizeof(*ctx)); +} + enum blake2s_iv { BLAKE2S_IV0 = 0x6A09E667UL, BLAKE2S_IV1 = 0xBB67AE85UL, diff --git a/lib/crypto/blake2b.c b/lib/crypto/blake2b.c index 581b7f8486fae8..55d6c437f311ff 100644 --- a/lib/crypto/blake2b.c +++ b/lib/crypto/blake2b.c @@ -148,7 +148,7 @@ void blake2b_final(struct blake2b_ctx *ctx, u8 *out) blake2b_compress(ctx, ctx->buf, 1, ctx->buflen); cpu_to_le64_array(ctx->h, ARRAY_SIZE(ctx->h)); memcpy(out, ctx->h, ctx->outlen); - memzero_explicit(ctx, sizeof(*ctx)); + blake2b_zeroize_ctx(ctx); } EXPORT_SYMBOL(blake2b_final); diff --git a/lib/crypto/blake2s.c b/lib/crypto/blake2s.c index 71578a08474233..24f7f34334b010 100644 --- a/lib/crypto/blake2s.c +++ b/lib/crypto/blake2s.c @@ -142,7 +142,7 @@ void blake2s_final(struct blake2s_ctx *ctx, u8 *out) blake2s_compress(ctx, ctx->buf, 1, ctx->buflen); cpu_to_le32_array(ctx->h, ARRAY_SIZE(ctx->h)); memcpy(out, ctx->h, ctx->outlen); - memzero_explicit(ctx, sizeof(*ctx)); + blake2s_zeroize_ctx(ctx); } EXPORT_SYMBOL(blake2s_final); From ff32b230b2de68385349cb5428594de76ba164c4 Mon Sep 17 00:00:00 2001 From: Thomas Huth Date: Wed, 16 Sep 2026 11:50:10 +0200 Subject: [PATCH 0134/1012] lib/crypto: sha1: Provide functions for zeroizing hmac_sha1 structures In certain cases crypto code needs to zeroize their local hmac_sha1_key or hmac_sha1_ctx structures after use to avoid leaking sensitive material. Provide hmac_sha1_zeroize_key() and hmac_sha1_zeroize_ctx() helper functions that e.g. can be used with __cleanup() to automatically zeroize the structures when they go out of scope. While we're at it, replace the related memzero_explicit() call in lib/crypto/sha1.c with a call to the new helper function. Signed-off-by: Thomas Huth Link: https://patch.msgid.link/20260916095022.604354-9-thuth@redhat.com Signed-off-by: Eric Biggers --- include/crypto/sha1.h | 19 +++++++++++++++++++ lib/crypto/sha1.c | 2 +- 2 files changed, 20 insertions(+), 1 deletion(-) diff --git a/include/crypto/sha1.h b/include/crypto/sha1.h index 4d973e016cd696..bc0046bffeaeef 100644 --- a/include/crypto/sha1.h +++ b/include/crypto/sha1.h @@ -7,6 +7,7 @@ #define _CRYPTO_SHA1_H #include +#include #define SHA1_DIGEST_SIZE 20 #define SHA1_BLOCK_SIZE 64 @@ -96,6 +97,15 @@ struct hmac_sha1_key { struct sha1_block_state ostate; }; +/** + * hmac_sha1_zeroize_key() - Zeroize an hmac_sha1_key structure + * @key: The hmac_sha1_key to zeroize + */ +static inline void hmac_sha1_zeroize_key(struct hmac_sha1_key *key) +{ + memzero_explicit(key, sizeof(*key)); +} + /** * struct hmac_sha1_ctx - Context for computing HMAC-SHA1 of a message * @sha_ctx: private @@ -106,6 +116,15 @@ struct hmac_sha1_ctx { struct sha1_block_state ostate; }; +/** + * hmac_sha1_zeroize_ctx() - Zeroize an hmac_sha1_ctx structure + * @ctx: The hmac_sha1_ctx context to zeroize + */ +static inline void hmac_sha1_zeroize_ctx(struct hmac_sha1_ctx *ctx) +{ + memzero_explicit(ctx, sizeof(*ctx)); +} + /** * hmac_sha1_preparekey() - Prepare a key for HMAC-SHA1 * @key: (output) the key structure to initialize diff --git a/lib/crypto/sha1.c b/lib/crypto/sha1.c index b687b89d97cb44..c4361ef77166ef 100644 --- a/lib/crypto/sha1.c +++ b/lib/crypto/sha1.c @@ -275,7 +275,7 @@ void hmac_sha1_final(struct hmac_sha1_ctx *ctx, u8 out[SHA1_DIGEST_SIZE]) for (size_t i = 0; i < SHA1_DIGEST_SIZE; i += 4) put_unaligned_be32(ctx->ostate.h[i / 4], out + i); - memzero_explicit(ctx, sizeof(*ctx)); + hmac_sha1_zeroize_ctx(ctx); } EXPORT_SYMBOL_GPL(hmac_sha1_final); From 9d654db44a30fb1a34593573ea763b3e404d9d00 Mon Sep 17 00:00:00 2001 From: Thomas Huth Date: Wed, 16 Sep 2026 11:50:11 +0200 Subject: [PATCH 0135/1012] security: keys: trusted: always clear the hmac_sha1_ctx before returning Clear the hmac_sha1_ctx structure via __cleanup(hmac_sha1_zeroize_ctx) to make sure that the function does not leak sensitive data on the stack when returning without calling hmac_sha1_final(). Reviewed-by: Jarkko Sakkinen Signed-off-by: Thomas Huth Link: https://patch.msgid.link/20260916095022.604354-10-thuth@redhat.com Signed-off-by: Eric Biggers --- security/keys/trusted-keys/trusted_tpm1.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/security/keys/trusted-keys/trusted_tpm1.c b/security/keys/trusted-keys/trusted_tpm1.c index bf0bf7f3697052..e5a904b5c19469 100644 --- a/security/keys/trusted-keys/trusted_tpm1.c +++ b/security/keys/trusted-keys/trusted_tpm1.c @@ -101,7 +101,7 @@ static inline void dump_tpm_buf(unsigned char *buf) static int TSS_rawhmac(unsigned char *digest, const unsigned char *key, unsigned int keylen, ...) { - struct hmac_sha1_ctx hmac_ctx; + struct hmac_sha1_ctx hmac_ctx __cleanup(hmac_sha1_zeroize_ctx); va_list argp; unsigned int dlen; unsigned char *data; From 57adc2fa748ac09f8b5bfd3b8fb7a23115b36804 Mon Sep 17 00:00:00 2001 From: Thomas Huth Date: Wed, 16 Sep 2026 11:50:15 +0200 Subject: [PATCH 0136/1012] lib/crypto: Add documentation about zeroization of key and context data Add a central document about zeroization in libcrypto so we don't have to repeat this information in the individual kernel docs of the zeroization functions all over the place. Signed-off-by: Thomas Huth Link: https://patch.msgid.link/20260916095022.604354-14-thuth@redhat.com Signed-off-by: Eric Biggers --- .../crypto/libcrypto-zeroization.rst | 150 ++++++++++++++++++ Documentation/crypto/libcrypto.rst | 1 + 2 files changed, 151 insertions(+) create mode 100644 Documentation/crypto/libcrypto-zeroization.rst diff --git a/Documentation/crypto/libcrypto-zeroization.rst b/Documentation/crypto/libcrypto-zeroization.rst new file mode 100644 index 00000000000000..76b65067116347 --- /dev/null +++ b/Documentation/crypto/libcrypto-zeroization.rst @@ -0,0 +1,150 @@ +.. SPDX-License-Identifier: GPL-2.0-or-later + +Crypto Key Zeroization +====================== + +This document describes the conventions for zeroizing crypto structures in the +kernel. + +Note: the kernel follows traditional cryptographic terminology by using +the term "zeroizing" to mean erasing sensitive parameters to prevent +their disclosure if the system is later compromised. This distinguishes +it from zeroing memory for other purposes such as initialization. + +.. contents:: + +Overview +-------- + +Cryptographic key material and intermediate state (such as HMAC contexts) must +be zeroized after use to prevent sensitive data from lingering on the stack or +heap, where it could be leaked through memory disclosure vulnerabilities, +crash dumps, or cold-boot attacks. + +For memory that has been allocated with kmalloc() or a similar function, +kfree_sensitive() should be used instead of kfree() to release the memory. + +For other cases, the kernel provides memzero_explicit() for clearing the +memory. Unlike plain memset(), memzero_explicit() is guaranteed not +to be optimized away by the compiler, even when the memory being cleared +appears to be dead. + +The crypto library builds on memzero_explicit() by providing typed +zeroization helpers for each key and context structure. These helpers serve +two purposes: + +1. They make __cleanup() annotations possible, so that structures on + the stack are automatically zeroized when they go out of scope. + +2. They improve readability by replacing ``memzero_explicit(&key, sizeof(key))`` + with a self-documenting call like ``aes_zeroize_key(&key)``. + + +What to zeroize +--------------- + +The following types of structures hold sensitive material and should be +zeroized after use: + +- **Key structures** (e.g. ``struct aes_key``, ``struct hmac_sha256_key``): + contain expanded round keys or prepared key material. + +- **HMAC/MAC context structures** (e.g. ``struct hmac_sha256_ctx``, + ``struct aes_cmac_ctx``): contain inner and outer hash states derived from + the key. + +- **Hash context structures** (e.g. ``struct sha256_ctx``): may contain + sensitive data being hashed. + +Not all of these require explicit cleanup by callers. Many ``..._final()`` +functions already zeroize their context internally (see `Automatic vs. manual +zeroization`_ below). + + +Zeroization helpers +------------------- + +Each crypto structure that callers may need to zeroize should have a +corresponding inline helper function. The naming convention is:: + + _zeroize_(struct _ *p); + +For example:: + + void aes_zeroize_key(struct aes_key *key); + void aes_zeroize_enckey(struct aes_enckey *key); + void hmac_sha256_zeroize_ctx(struct hmac_sha256_ctx *ctx); + void aes_cmac_zeroize_key(struct aes_cmac_key *key); + void aes_cmac_zeroize_ctx(struct aes_cmac_ctx *ctx); + +Each helper is a ``static inline`` function in the algorithm's header that +wraps ``memzero_explicit()``, for example:: + + static inline void hmac_sha256_zeroize_ctx(struct hmac_sha256_ctx *ctx) + { + memzero_explicit(ctx, sizeof(*ctx)); + } + + +Using __cleanup for automatic zeroization +----------------------------------------- + +The preferred way to zeroize stack-allocated key and context structures is +with the __cleanup() attribute. This ensures zeroization happens on all +exit paths, including error returns and early exits. For example:: + + static int my_aesxts_setkey(..., const u8 *key, unsigned int len) + { + struct crypto_aes_ctx aes __cleanup(aes_zeroize_ctx); + ... + + /* Only half of the key data is cipher key */ + keylen = (len >> 1); + ret = aes_expandkey(&aes, key, keylen); + if (ret) + return ret; + + ... do something with the cipher key ... + + /* The other half is the tweak key */ + ret = aes_expandkey(&aes, (u8 *)(key + keylen), keylen); + if (ret) + return ret; /* <-- Could leak cipher key without __cleanup */ + + ... do something with the tweak key ... + + /* No need for memzero_explicit() at the end thanks to the __cleanup */ + return 0; + } + +Note that __cleanup() attributes should not be used in functions that use +"goto" statements. The benefit of cleanup helpers is the removal of "gotos", +and that "goto" statements can jump between scopes, so the expectation is +that usage of "goto" and cleanup helpers is never mixed in the same function. + + +Automatic vs. manual zeroization +-------------------------------- + +Many ``..._final()`` functions in the crypto library automatically zeroize +their context before returning. When this is the case, the kernel-doc for the +function documents it:: + + After finishing, this zeroizes @ctx. So the caller does not need to do it. + +In these cases, callers on simple code paths (where ``..._final()`` is always +reached) do not need to add __cleanup() or explicit zeroization. +However, __cleanup() is still recommended whenever there are error paths +that bypass ``..._final()``, as it ensures zeroization on all paths. + +For algorithms where ``..._final()`` does *not* zeroize the context (such as +the SHAKE XOFs, where ``shake_squeeze()`` can be called multiple times), +callers must explicitly zeroize the context by calling the appropriate helper +or using __cleanup(), for example:: + + struct shake_ctx ctx __cleanup(shake_zeroize_ctx); + + shake256_init(&ctx); + shake_update(&ctx, data, data_len); + shake_squeeze(&ctx, out, out_len); + /* ctx is automatically zeroized at end of scope */ diff --git a/Documentation/crypto/libcrypto.rst b/Documentation/crypto/libcrypto.rst index e911e05215979c..9533c12caa79dc 100644 --- a/Documentation/crypto/libcrypto.rst +++ b/Documentation/crypto/libcrypto.rst @@ -165,4 +165,5 @@ API documentation libcrypto-signature libcrypto-unauth-encryption libcrypto-utils + libcrypto-zeroization sha3 From e44867c61d3204e24ce0857ce9307ad9c625ecf0 Mon Sep 17 00:00:00 2001 From: Pengpeng Hou Date: Sun, 20 Sep 2026 11:33:03 +0800 Subject: [PATCH 0137/1012] iio: cdc: ad7150: fix OF matching and publish module aliases The OF match entries use positional initializers, which initialize the name field rather than compatible. Consequently the advertised AD7150, AD7151 and AD7156 compatible strings do not describe OF compatible matches. The table is also not exported for module alias generation. Use designated compatible initializers and publish the OF table. Keep the existing I2C ID table and its device-variant selection unchanged. The issue was found by our static-analysis tool. Fixes: 89f2d5b080bc ("staging:iio:cdc:ad7150: Add of_match_table") Assisted-by: LLM Cc: stable@vger.kernel.org Signed-off-by: Pengpeng Hou Signed-off-by: Jonathan Cameron --- drivers/iio/cdc/ad7150.c | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/drivers/iio/cdc/ad7150.c b/drivers/iio/cdc/ad7150.c index 2f35c6d2f9cee5..b36ac4e2dae800 100644 --- a/drivers/iio/cdc/ad7150.c +++ b/drivers/iio/cdc/ad7150.c @@ -636,11 +636,13 @@ static const struct i2c_device_id ad7150_id[] = { MODULE_DEVICE_TABLE(i2c, ad7150_id); static const struct of_device_id ad7150_of_match[] = { - { "adi,ad7150" }, - { "adi,ad7151" }, - { "adi,ad7156" }, + { .compatible = "adi,ad7150" }, + { .compatible = "adi,ad7151" }, + { .compatible = "adi,ad7156" }, { } }; +MODULE_DEVICE_TABLE(of, ad7150_of_match); + static struct i2c_driver ad7150_driver = { .driver = { .name = "ad7150", From b7ffbdb8de0b2becf45bd7b99dd3b13d9a4de02a Mon Sep 17 00:00:00 2001 From: Jinseob Kim Date: Mon, 21 Sep 2026 15:24:21 +0900 Subject: [PATCH 0138/1012] iio: buffer: serialize buffer teardown with mode claims MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Buffer-mode claims hold mlock to guarantee that the device remains in buffer mode until the claim is released. Normal buffer updates take info_exist_lock followed by mlock in iio_update_buffers(). However, iio_device_unregister() disables and deactivates all buffers without taking mlock. This can invalidate buffer state, including active_scan_mask, while a buffer-mode claim is held. Take mlock in iio_disable_all_buffers() so that unregister honors the mode-claim lifetime guarantee. The info_exist_lock -> mlock ordering matches iio_update_buffers(). Fixes: 0a8565425afd ("iio: core: introduce iio_device_{claim|release}_buffer_mode() APIs") Suggested-by: Jonathan Cameron Signed-off-by: Jinseob Kim Reviewed-by: Joshua Crofts Reviewed-by: Nuno Sá Cc: stable@vger.kernel.org Signed-off-by: Jonathan Cameron --- drivers/iio/industrialio-buffer.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/drivers/iio/industrialio-buffer.c b/drivers/iio/industrialio-buffer.c index 902401be0b6790..4294101810fd71 100644 --- a/drivers/iio/industrialio-buffer.c +++ b/drivers/iio/industrialio-buffer.c @@ -1377,6 +1377,10 @@ EXPORT_SYMBOL_GPL(iio_update_buffers); void iio_disable_all_buffers(struct iio_dev *indio_dev) { + struct iio_dev_opaque *iio_dev_opaque = to_iio_dev_opaque(indio_dev); + + guard(mutex)(&iio_dev_opaque->mlock); + iio_disable_buffers(indio_dev); iio_buffer_deactivate_all(indio_dev); } From cc6a4f34b6166fadcc944ebcbc25d32f1822d665 Mon Sep 17 00:00:00 2001 From: Shao-Fu Chen Date: Fri, 18 Sep 2026 18:10:46 +0800 Subject: [PATCH 0139/1012] optee: register TEE devices only once fully initialized optee_probe() called tee_device_register() on both the client and the supplicant device before the rest of struct optee had been initialized. tee_device_register() calls cdev_device_add(), which does two things at once: it creates /dev/tee0 and /dev/teepriv0, and it links the device into the tee class so that class_find_device() can find it. From that moment on the device is reachable both from user space via tee_open() and from kernel space via tee_client_open_context(). The only gate in teedev_open() is tee_device_get(), which merely checks that teedev->desc is non-NULL, which was already set by tee_device_alloc(). Therefore, there is effectively no gate at all. A context opened in that window can run against a struct optee where - optee->call_queue.mutex is not initialized by optee_cq_init() - optee->supp mutex and completions is not initialized by optee_supp_init() - optee->rpmb_dev_mutex is not initialized yet, - the message argument cache is not initialized by optee_shm_arg_cache_init() - optee->ctx is still NULL Any open session or invoke during this window takes uninitialized mutexes and dereferences a NULL pointer. Fix it by moving both tee_device_register() calls down to the point where all of struct optee is set up. Signed-off-by: Shao-Fu Chen Signed-off-by: Jens Wiklander --- drivers/tee/optee/ffa_abi.c | 16 ++++++++-------- drivers/tee/optee/smc_abi.c | 17 ++++++++--------- 2 files changed, 16 insertions(+), 17 deletions(-) diff --git a/drivers/tee/optee/ffa_abi.c b/drivers/tee/optee/ffa_abi.c index 633715b98625cd..d3cc7c5fc1e994 100644 --- a/drivers/tee/optee/ffa_abi.c +++ b/drivers/tee/optee/ffa_abi.c @@ -1123,14 +1123,6 @@ static int optee_ffa_probe(struct ffa_device *ffa_dev) optee_set_dev_group(optee); - rc = tee_device_register(optee->teedev); - if (rc) - goto err_unreg_supp_teedev; - - rc = tee_device_register(optee->supp_teedev); - if (rc) - goto err_unreg_supp_teedev; - rc = rhashtable_init(&optee->ffa.global_ids, &shm_rhash_params); if (rc) goto err_unreg_supp_teedev; @@ -1159,6 +1151,14 @@ static int optee_ffa_probe(struct ffa_device *ffa_dev) if (optee_ffa_protmem_pool_init(optee, sec_caps)) pr_info("Protected memory service not available\n"); + rc = tee_device_register(optee->teedev); + if (rc) + goto err_unregister_devices; + + rc = tee_device_register(optee->supp_teedev); + if (rc) + goto err_unregister_devices; + rc = optee_enumerate_devices(PTA_CMD_GET_DEVICES); if (rc) goto err_unregister_devices; diff --git a/drivers/tee/optee/smc_abi.c b/drivers/tee/optee/smc_abi.c index b8a2bdac3208f9..51624443359bfb 100644 --- a/drivers/tee/optee/smc_abi.c +++ b/drivers/tee/optee/smc_abi.c @@ -1849,14 +1849,6 @@ static int optee_probe(struct platform_device *pdev) optee_set_dev_group(optee); - rc = tee_device_register(optee->teedev); - if (rc) - goto err_unreg_supp_teedev; - - rc = tee_device_register(optee->supp_teedev); - if (rc) - goto err_unreg_supp_teedev; - optee_cq_init(&optee->call_queue, thread_count); optee_supp_init(&optee->supp); optee->smc.memremaped_shm = memremaped_shm; @@ -1916,6 +1908,14 @@ static int optee_probe(struct platform_device *pdev) if (optee->smc.sec_caps & OPTEE_SMC_SEC_CAP_DYNAMIC_SHM) pr_info("dynamic shared memory is enabled\n"); + rc = tee_device_register(optee->teedev); + if (rc) + goto err_disable_shm_cache; + + rc = tee_device_register(optee->supp_teedev); + if (rc) + goto err_disable_shm_cache; + rc = optee_enumerate_devices(PTA_CMD_GET_DEVICES); if (rc) goto err_disable_shm_cache; @@ -1942,7 +1942,6 @@ static int optee_probe(struct platform_device *pdev) optee_shm_arg_cache_uninit(optee); optee_supp_uninit(&optee->supp); mutex_destroy(&optee->call_queue.mutex); -err_unreg_supp_teedev: tee_device_unregister(optee->supp_teedev); err_unreg_teedev: tee_device_unregister(optee->teedev); From 049791c97bd4a59bc8fe9de302f2c496b6115f0a Mon Sep 17 00:00:00 2001 From: Marouene Boubakri Date: Thu, 24 Sep 2026 12:51:45 +0200 Subject: [PATCH 0140/1012] tee: optee: build the Arm-specific code only on Arm The OP-TEE driver reaches OP-TEE through the SMC ABI or the FF-A ABI, both specific to Arm, yet builds both unconditionally together with the SMC Calling Convention definitions they rely on: ffa_abi.c is always compiled and only its registration is conditioned on IS_REACHABLE(CONFIG_ARM_FFA_TRANSPORT), and optee_private.h includes and defines the SMC and FF-A specific types for every file of the driver. This is fine as long as the driver depends on HAVE_ARM_SMCCC, but it keeps the driver from being built for an architecture without SMCCC, such as RISC-V. Build smc_abi.c only when HAVE_ARM_SMCCC is set and ffa_abi.c only when the FF-A transport is enabled, and provide stubs for their registration otherwise, so that it fails with -EOPNOTSUPP as the FF-A ABI already does when the FF-A transport is not reachable. Keep the SMCCC header, the SMC invoke function type, the SMC and FF-A specific structures and the SMC RPC register parameters in optee_private.h under the same conditions, and drop the unused include from notif.c. Make OPTEE depend on ARM_FFA_TRANSPORT || !ARM_FFA_TRANSPORT, as it already does for RPMB, so that the driver is limited to a module when the FF-A transport is one, rather than built in without FF-A support. ffa_abi.c is thus built exactly when the FF-A transport is reachable from the driver, and the IS_REACHABLE() checks in optee_ffa_abi_register() and optee_ffa_abi_unregister() are always true, so drop them. OPTEE still depends on HAVE_ARM_SMCCC, so smc_abi.c is still always built. The only visible change is that OPTEE=y can no longer be combined with ARM_FFA_TRANSPORT=m: such a configuration now resolves to OPTEE=m, with the FF-A ABI available. Signed-off-by: Marouene Boubakri Signed-off-by: Jens Wiklander --- drivers/tee/optee/Kconfig | 1 + drivers/tee/optee/Makefile | 4 ++-- drivers/tee/optee/ffa_abi.c | 8 ++----- drivers/tee/optee/notif.c | 1 - drivers/tee/optee/optee_private.h | 39 ++++++++++++++++++++++++++++++- 5 files changed, 43 insertions(+), 10 deletions(-) diff --git a/drivers/tee/optee/Kconfig b/drivers/tee/optee/Kconfig index 50d2051f7f20b7..891dac63cab85a 100644 --- a/drivers/tee/optee/Kconfig +++ b/drivers/tee/optee/Kconfig @@ -5,6 +5,7 @@ config OPTEE depends on HAVE_ARM_SMCCC depends on MMU depends on RPMB || !RPMB + depends on ARM_FFA_TRANSPORT || !ARM_FFA_TRANSPORT help This implements the OP-TEE Trusted Execution Environment (TEE) driver. diff --git a/drivers/tee/optee/Makefile b/drivers/tee/optee/Makefile index ad7049c1c10721..183cdde1ac045c 100644 --- a/drivers/tee/optee/Makefile +++ b/drivers/tee/optee/Makefile @@ -7,8 +7,8 @@ optee-objs += rpc.o optee-objs += protmem.o optee-objs += supp.o optee-objs += device.o -optee-objs += smc_abi.o -optee-objs += ffa_abi.o +optee-$(CONFIG_HAVE_ARM_SMCCC) += smc_abi.o +optee-$(CONFIG_ARM_FFA_TRANSPORT) += ffa_abi.o # for tracing framework to find optee_trace.h CFLAGS_smc_abi.o := -I$(src) diff --git a/drivers/tee/optee/ffa_abi.c b/drivers/tee/optee/ffa_abi.c index d3cc7c5fc1e994..723b929d02bd3d 100644 --- a/drivers/tee/optee/ffa_abi.c +++ b/drivers/tee/optee/ffa_abi.c @@ -1212,14 +1212,10 @@ static struct ffa_driver optee_ffa_driver = { int optee_ffa_abi_register(void) { - if (IS_REACHABLE(CONFIG_ARM_FFA_TRANSPORT)) - return ffa_register(&optee_ffa_driver); - else - return -EOPNOTSUPP; + return ffa_register(&optee_ffa_driver); } void optee_ffa_abi_unregister(void) { - if (IS_REACHABLE(CONFIG_ARM_FFA_TRANSPORT)) - ffa_unregister(&optee_ffa_driver); + ffa_unregister(&optee_ffa_driver); } diff --git a/drivers/tee/optee/notif.c b/drivers/tee/optee/notif.c index 6e85f2f5c516f8..68014222d7be72 100644 --- a/drivers/tee/optee/notif.c +++ b/drivers/tee/optee/notif.c @@ -5,7 +5,6 @@ #define pr_fmt(fmt) KBUILD_MODNAME ": " fmt -#include #include #include #include diff --git a/drivers/tee/optee/optee_private.h b/drivers/tee/optee/optee_private.h index aefe1e6f568915..02d6f79df407bc 100644 --- a/drivers/tee/optee/optee_private.h +++ b/drivers/tee/optee/optee_private.h @@ -6,7 +6,6 @@ #ifndef OPTEE_PRIVATE_H #define OPTEE_PRIVATE_H -#include #include #include #include @@ -15,6 +14,10 @@ #include #include "optee_msg.h" +#ifdef CONFIG_HAVE_ARM_SMCCC +#include +#endif + #define DRIVER_NAME "optee" #define OPTEE_MAX_ARG_SIZE 1024 @@ -42,10 +45,12 @@ */ #define OPTEE_DEFAULT_MAX_NOTIF_VALUE 255 +#ifdef CONFIG_HAVE_ARM_SMCCC typedef void (optee_invoke_fn)(unsigned long, unsigned long, unsigned long, unsigned long, unsigned long, unsigned long, unsigned long, unsigned long, struct arm_smccc_res *); +#endif /** * struct optee_call_waiter - TEE entry may need to wait for a free TEE thread @@ -119,6 +124,7 @@ struct optee_supp { struct completion reqs_c; }; +#ifdef CONFIG_HAVE_ARM_SMCCC /** * struct optee_pcpu - per cpu notif private struct passed to work functions * @optee: optee device reference @@ -149,7 +155,9 @@ struct optee_smc { struct work_struct notif_pcpu_work; unsigned int notif_cpuhp_state; }; +#endif +#if IS_REACHABLE(CONFIG_ARM_FFA_TRANSPORT) /** * struct optee_ffa - FFA communication struct * @ffa_dev: FFA device, contains the destination id, the id of @@ -170,6 +178,7 @@ struct optee_ffa { struct workqueue_struct *notif_wq; struct work_struct notif_work; }; +#endif struct optee; @@ -257,8 +266,12 @@ struct optee { const struct optee_ops *ops; struct tee_context *ctx; union { +#ifdef CONFIG_HAVE_ARM_SMCCC struct optee_smc smc; +#endif +#if IS_REACHABLE(CONFIG_ARM_FFA_TRANSPORT) struct optee_ffa ffa; +#endif }; struct optee_shm_arg_cache shm_arg_cache; struct optee_call_queue call_queue; @@ -290,6 +303,7 @@ struct optee_context_data { struct list_head sess_list; }; +#ifdef CONFIG_HAVE_ARM_SMCCC struct optee_rpc_param { u32 a0; u32 a1; @@ -300,6 +314,7 @@ struct optee_rpc_param { u32 a6; u32 a7; }; +#endif /* Holds context that is preserved during one STD call */ struct optee_call_ctx { @@ -422,9 +437,31 @@ static inline void reg_pair_from_64(u32 *reg0, u32 *reg1, u64 val) } /* Registration of the ABIs */ +#ifdef CONFIG_HAVE_ARM_SMCCC int optee_smc_abi_register(void); void optee_smc_abi_unregister(void); +#else +static inline int optee_smc_abi_register(void) +{ + return -EOPNOTSUPP; +} + +static inline void optee_smc_abi_unregister(void) +{ +} +#endif +#if IS_REACHABLE(CONFIG_ARM_FFA_TRANSPORT) int optee_ffa_abi_register(void); void optee_ffa_abi_unregister(void); +#else +static inline int optee_ffa_abi_register(void) +{ + return -EOPNOTSUPP; +} + +static inline void optee_ffa_abi_unregister(void) +{ +} +#endif #endif /*OPTEE_PRIVATE_H*/ From 25e222e57c666c596afee29d9a924094efc84a13 Mon Sep 17 00:00:00 2001 From: Eric Biggers Date: Thu, 24 Sep 2026 22:25:02 -0700 Subject: [PATCH 0141/1012] Documentation: libcrypto: Add additional testing information Link to the KUnit documentation, mention the existence of the benchmarks and how to enable them, explicitly mention that the tests can be run on real hardware, and mention that it's often possible to clear CPU features via the kernel command line. Acked-by: Ard Biesheuvel Link: https://patch.msgid.link/20260925052502.188024-1-ebiggers@kernel.org Signed-off-by: Eric Biggers --- Documentation/crypto/libcrypto.rst | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/Documentation/crypto/libcrypto.rst b/Documentation/crypto/libcrypto.rst index 9533c12caa79dc..494541c8a8e01b 100644 --- a/Documentation/crypto/libcrypto.rst +++ b/Documentation/crypto/libcrypto.rst @@ -127,6 +127,8 @@ The crypto library uses standard KUnit tests. Like many of the kernel's other KUnit tests, they are included in the set of tests that is run by ``tools/testing/kunit/kunit.py run --alltests``. +For more information about KUnit, see Documentation/dev-tools/kunit/start.rst. + A ``.kunitconfig`` file is also provided to run just the crypto library tests. For example, here's how to run them in user-mode Linux: @@ -148,6 +150,17 @@ emulate the correct type of hardware for the code to be reached. Since correctness is essential in cryptographic code, new architecture-optimized code is accepted only if it can be tested in QEMU. +Most of the crypto KUnit tests also include benchmarks. To enable these, enable +``CONFIG_CRYPTO_LIB_BENCHMARK=y`` (in addition to the tests themselves). The +benchmark results are printed to the kernel log when the test runs. + +Of course, the crypto KUnit tests can also be run on real hardware. Note that +it is generally still possible to test and benchmark non-default code paths in +this case (for example, the software implementation of AES when the CPU has +hardware-accelerated AES), since on many architectures the kernel supports +disabling CPU features via the kernel command line. For example, on x86, +the ``clearcpuid=aes`` kernel command line option disables AES acceleration. + Note: the crypto library also includes FIPS 140 self-tests. These are lightweight, are designed specifically to meet FIPS 140 requirements, and exist *only* to meet those requirements. Normal testing done by kernel developers and From b63b9b3d5ddaf0f1767f31aeef7aef08bb36225e Mon Sep 17 00:00:00 2001 From: Eric Biggers Date: Thu, 24 Sep 2026 22:32:40 -0700 Subject: [PATCH 0142/1012] MAINTAINERS: Place libcrypto docs under CRYPTO LIBRARY Place Documentation/crypto/libcrypto* under the appropriate entry in MAINTAINERS. Acked-by: Ard Biesheuvel Link: https://patch.msgid.link/20260925053240.189395-1-ebiggers@kernel.org Signed-off-by: Eric Biggers --- MAINTAINERS | 1 + 1 file changed, 1 insertion(+) diff --git a/MAINTAINERS b/MAINTAINERS index cc3cae2e378b34..9dc69113f47a46 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -6979,6 +6979,7 @@ L: linux-crypto@vger.kernel.org S: Maintained T: git https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git libcrypto-next T: git https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git libcrypto-fixes +F: Documentation/crypto/libcrypto* F: lib/crypto/ F: scripts/crypto/ From a06776565b9e73512e880a95b66c198ae8e7e24a Mon Sep 17 00:00:00 2001 From: Thorsten Blum Date: Thu, 24 Sep 2026 23:32:55 +0200 Subject: [PATCH 0143/1012] riscv: process: Use str_supported_unsupported() in compat_mode_detect() Replace hard-coded strings with the str_supported_unsupported() helper. This unifies the output and helps the linker with deduplication, which can result in a smaller binary. Signed-off-by: Thorsten Blum Link: https://patch.msgid.link/20260924213256.122530-2-blum@kernel.org Signed-off-by: Paul Walmsley --- arch/riscv/kernel/process.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/arch/riscv/kernel/process.c b/arch/riscv/kernel/process.c index 7cc5a6a5c02062..55294ee6896bf2 100644 --- a/arch/riscv/kernel/process.c +++ b/arch/riscv/kernel/process.c @@ -13,6 +13,7 @@ #include #include #include +#include #include #include #include @@ -134,7 +135,7 @@ static int __init compat_mode_detect(void) csr_write(CSR_STATUS, tmp); pr_info("riscv: ELF compat mode %s", - compat_mode_supported ? "supported" : "unsupported"); + str_supported_unsupported(compat_mode_supported)); return 0; } From 0755b7d25bec4618b6cfc5a7d66ae077e8b19a61 Mon Sep 17 00:00:00 2001 From: Violet Monti Date: Tue, 22 Sep 2026 09:56:08 -0700 Subject: [PATCH 0144/1012] drm/xe/ptl: Add a new PTL PCI ID A new PCI ID for PTL. Bspec: 72574 Signed-off-by: Matt Atwood Signed-off-by: Violet Monti Reviewed-by: Rodrigo Vivi Link: https://patch.msgid.link/20260922165607.129560-2-violet.monti@intel.com Signed-off-by: Rodrigo Vivi --- include/drm/intel/pciids.h | 1 + 1 file changed, 1 insertion(+) diff --git a/include/drm/intel/pciids.h b/include/drm/intel/pciids.h index dff389b56eb3bf..32c88a690ad822 100644 --- a/include/drm/intel/pciids.h +++ b/include/drm/intel/pciids.h @@ -880,6 +880,7 @@ MACRO__(0xB08F, ## __VA_ARGS__), \ MACRO__(0xB090, ## __VA_ARGS__), \ MACRO__(0xB0A0, ## __VA_ARGS__), \ + MACRO__(0xB0A1, ## __VA_ARGS__), \ MACRO__(0xB0B0, ## __VA_ARGS__) /* WCL */ From 9b2f8146fcef9faa73f9dc1bad9a8d6c585cfec8 Mon Sep 17 00:00:00 2001 From: Paolo Bonzini Date: Sat, 26 Sep 2026 02:04:03 -0400 Subject: [PATCH 0145/1012] KVM: selftests: Extend nested x2APIC test to validate disabling x2APIC virt Signed-off-by: Sean Christopherson Message-ID: <20260813223610.2043560-6-seanjc@google.com> Signed-off-by: Paolo Bonzini --- .../selftests/kvm/x86/nested_x2apic_test.c | 88 ++++++++++++++----- 1 file changed, 68 insertions(+), 20 deletions(-) diff --git a/tools/testing/selftests/kvm/x86/nested_x2apic_test.c b/tools/testing/selftests/kvm/x86/nested_x2apic_test.c index 3b59ba3e334206..964937aefe39f1 100644 --- a/tools/testing/selftests/kvm/x86/nested_x2apic_test.c +++ b/tools/testing/selftests/kvm/x86/nested_x2apic_test.c @@ -29,10 +29,12 @@ static void l2_guest_code(void) if (inhibit_apicv) wrmsr(MSR_IA32_APICBASE, rdmsr(MSR_IA32_APICBASE) & GENMASK_ULL(11, 0)); - x2apic_write_reg(APIC_TASKPRI, 0xf0); - GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0xf0); + for (;;) { + x2apic_write_reg(APIC_TASKPRI, 0xf0); + GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0xf0); - asm volatile("cpuid" ::: "eax", "ebx", "ecx", "edx"); + asm volatile("cpuid" ::: "eax", "ebx", "ecx", "edx"); + } } static void l1_svm_code(struct svm_test_data *svm) @@ -46,6 +48,7 @@ static void l1_svm_code(struct svm_test_data *svm) GUEST_ASSERT_EQ(ctrl->exit_code, SVM_EXIT_CPUID); stgi(); + x2apic_write_reg(APIC_TASKPRI, 0); } static void l1_vmx_code(struct vmx_pages *vmx) @@ -58,22 +61,50 @@ static void l1_vmx_code(struct vmx_pages *vmx) prepare_vmcs(vmx, NULL); GUEST_ASSERT_EQ(vmwrite(GUEST_RIP, (unsigned long)l2_guest_code), 0); + control = vmreadz(PIN_BASED_VM_EXEC_CONTROL); + control |= PIN_BASED_EXT_INTR_MASK; + vmwrite(PIN_BASED_VM_EXEC_CONTROL, control); + control = vmreadz(CPU_BASED_VM_EXEC_CONTROL); - control |= CPU_BASED_USE_MSR_BITMAPS; + control |= CPU_BASED_USE_MSR_BITMAPS | CPU_BASED_TPR_SHADOW; GUEST_ASSERT_EQ(vmwrite(CPU_BASED_VM_EXEC_CONTROL, control), 0); + control = vmreadz(SECONDARY_VM_EXEC_CONTROL); + control |= SECONDARY_EXEC_VIRTUALIZE_X2APIC_MODE | + SECONDARY_EXEC_APIC_REGISTER_VIRT | + SECONDARY_EXEC_VIRTUAL_INTR_DELIVERY; + control &= (rdmsr(MSR_IA32_VMX_PROCBASED_CTLS2) >> 32); + GUEST_ASSERT_EQ(vmwrite(SECONDARY_VM_EXEC_CONTROL, control), 0); + GUEST_ASSERT(!vmlaunch()); GUEST_ASSERT_EQ(vmreadz(VM_EXIT_REASON), EXIT_REASON_CPUID); + GUEST_ASSERT_EQ(vmwrite(GUEST_RIP, + vmreadz(GUEST_RIP) + vmreadz(VM_EXIT_INSTRUCTION_LEN)), 0); } -static void l1_guest_code(void *test_data) +static void l1_vmx_code_part2(void) { - x2apic_enable(); + u64 control; - if (this_cpu_has(X86_FEATURE_SVM)) - l1_svm_code(test_data); - else - l1_vmx_code(test_data); + control = vmreadz(CPU_BASED_VM_EXEC_CONTROL); + control &= ~CPU_BASED_TPR_SHADOW; + GUEST_ASSERT_EQ(vmwrite(CPU_BASED_VM_EXEC_CONTROL, control), 0); + + control = vmreadz(SECONDARY_VM_EXEC_CONTROL); + control &= ~(SECONDARY_EXEC_VIRTUALIZE_X2APIC_MODE | + SECONDARY_EXEC_APIC_REGISTER_VIRT | + SECONDARY_EXEC_VIRTUAL_INTR_DELIVERY); + GUEST_ASSERT_EQ(vmwrite(SECONDARY_VM_EXEC_CONTROL, control), 0); + + GUEST_ASSERT(!vmresume()); + GUEST_ASSERT_EQ(vmreadz(VM_EXIT_REASON), EXIT_REASON_CPUID); + GUEST_ASSERT_EQ(vmwrite(GUEST_RIP, + vmreadz(GUEST_RIP) + vmreadz(VM_EXIT_INSTRUCTION_LEN)), 0); +} + +static void l1_test_x2apic_intercepts(void) +{ + GUEST_ASSERT_EQ(nr_irqs, 0); sti_nop(); @@ -93,16 +124,40 @@ static void l1_guest_code(void *test_data) x2apic_write_reg(APIC_ICR, APIC_DEST_SELF | APIC_INT_ASSERT | POSTED_INTR_NESTED_VECTOR); GUEST_ASSERT_EQ(nr_irqs, 3); + nr_irqs = 0; +} + +static void l1_guest_code(void *test_data) +{ + x2apic_enable(); + + if (this_cpu_has(X86_FEATURE_SVM)) + l1_svm_code(test_data); + else + l1_vmx_code(test_data); + + GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0); + x2apic_write_reg(APIC_TASKPRI, 0xf0); + + l1_test_x2apic_intercepts(); + + if (this_cpu_has(X86_FEATURE_VMX)) { + l1_vmx_code_part2(); + l1_test_x2apic_intercepts(); + } + GUEST_DONE(); } -static void __test_x2apic_intercepts(void) +static void __test_x2apic_intercepts(bool with_inhibit_apicv) { gva_t nested_test_data_gva; struct kvm_vcpu *vcpu; struct kvm_vm *vm; struct ucall uc; + inhibit_apicv = with_inhibit_apicv; + vm = vm_create_with_one_vcpu(&vcpu, l1_guest_code); vm_install_exception_handler(vm, POSTED_INTR_VECTOR, guest_irq_handler); vm_install_exception_handler(vm, POSTED_INTR_WAKEUP_VECTOR, guest_irq_handler); @@ -134,17 +189,10 @@ static void __test_x2apic_intercepts(void) kvm_vm_free(vm); } -#define test_x2apic_intercepts(inhibit_apic_setting) \ -do { \ - inhibit_apic_setting; \ - \ - __test_x2apic_intercepts(); \ -} while (0) - int main(int argc, char *argv[]) { TEST_REQUIRE(kvm_cpu_has(X86_FEATURE_SVM) || kvm_cpu_has(X86_FEATURE_VMX)); - test_x2apic_intercepts(inhibit_apicv = true); - test_x2apic_intercepts(inhibit_apicv = false); + test_x2apic_intercepts(true); + test_x2apic_intercepts(false); } From a4bab1c12fd528ccaafee00360e07463873404a9 Mon Sep 17 00:00:00 2001 From: Sean Christopherson Date: Thu, 13 Aug 2026 15:36:10 -0700 Subject: [PATCH 0146/1012] KVM: selftests: Extend nested x2APIC test to validate using eVMCS for vmcs12 Signed-off-by: Sean Christopherson Message-ID: <20260813223610.2043560-7-seanjc@google.com> Signed-off-by: Paolo Bonzini --- .../selftests/kvm/x86/nested_x2apic_test.c | 42 +++++++++++++++---- 1 file changed, 33 insertions(+), 9 deletions(-) diff --git a/tools/testing/selftests/kvm/x86/nested_x2apic_test.c b/tools/testing/selftests/kvm/x86/nested_x2apic_test.c index 964937aefe39f1..ce204ce29a9c0b 100644 --- a/tools/testing/selftests/kvm/x86/nested_x2apic_test.c +++ b/tools/testing/selftests/kvm/x86/nested_x2apic_test.c @@ -51,12 +51,24 @@ static void l1_svm_code(struct svm_test_data *svm) x2apic_write_reg(APIC_TASKPRI, 0); } -static void l1_vmx_code(struct vmx_pages *vmx) +static void l1_vmx_code(struct vmx_pages *vmx, struct hyperv_test_pages *hv_pages) { u64 control; + if (hv_pages) { + wrmsr(HV_X64_MSR_GUEST_OS_ID, HYPERV_LINUX_OS_ID); + enable_vp_assist(hv_pages->vp_assist_gpa, hv_pages->vp_assist); + evmcs_enable(); + } + GUEST_ASSERT_EQ(prepare_for_vmx_operation(vmx), true); - GUEST_ASSERT_EQ(load_vmcs(vmx), true); + + if (hv_pages) { + GUEST_ASSERT(load_evmcs(hv_pages)); + current_evmcs->hv_enlightenments_control.msr_bitmap = 1; + } else { + GUEST_ASSERT(load_vmcs(vmx)); + } prepare_vmcs(vmx, NULL); GUEST_ASSERT_EQ(vmwrite(GUEST_RIP, (unsigned long)l2_guest_code), 0); @@ -127,14 +139,14 @@ static void l1_test_x2apic_intercepts(void) nr_irqs = 0; } -static void l1_guest_code(void *test_data) +static void l1_guest_code(void *test_data, void *hv_pages) { x2apic_enable(); if (this_cpu_has(X86_FEATURE_SVM)) l1_svm_code(test_data); else - l1_vmx_code(test_data); + l1_vmx_code(test_data, hv_pages); GUEST_ASSERT_EQ(x2apic_read_reg(APIC_TASKPRI), 0); x2apic_write_reg(APIC_TASKPRI, 0xf0); @@ -149,9 +161,9 @@ static void l1_guest_code(void *test_data) GUEST_DONE(); } -static void __test_x2apic_intercepts(bool with_inhibit_apicv) +static void test_x2apic_intercepts(bool with_inhibit_apicv, bool use_evmcs) { - gva_t nested_test_data_gva; + gva_t nested_test_data_gva, hv_pages_gva = 0; struct kvm_vcpu *vcpu; struct kvm_vm *vm; struct ucall uc; @@ -170,7 +182,14 @@ static void __test_x2apic_intercepts(bool with_inhibit_apicv) else vcpu_alloc_vmx(vm, &nested_test_data_gva); - vcpu_args_set(vcpu, 1, nested_test_data_gva); + if (use_evmcs) { + vcpu_set_hv_cpuid(vcpu); + vcpu_enable_evmcs(vcpu); + + vcpu_alloc_hyperv_test_pages(vm, &hv_pages_gva); + } + + vcpu_args_set(vcpu, 2, nested_test_data_gva, hv_pages_gva); vcpu_run(vcpu); @@ -193,6 +212,11 @@ int main(int argc, char *argv[]) { TEST_REQUIRE(kvm_cpu_has(X86_FEATURE_SVM) || kvm_cpu_has(X86_FEATURE_VMX)); - test_x2apic_intercepts(true); - test_x2apic_intercepts(false); + test_x2apic_intercepts(true, false); + test_x2apic_intercepts(false, false); + + if (kvm_has_cap(KVM_CAP_HYPERV_ENLIGHTENED_VMCS)) { + test_x2apic_intercepts(true, true); + test_x2apic_intercepts(false, true); + } } From 0db8ffc8f15b781c506c5d98088bee02c3dacf54 Mon Sep 17 00:00:00 2001 From: Jiale Yao Date: Fri, 25 Sep 2026 22:09:43 +0800 Subject: [PATCH 0147/1012] iio: accel: kxcjk-1013: reject duplicate event disable The IIO core does not filter duplicate writes to the event enable attribute. kxcjk1013_write_event_config() already ignores repeated enable requests, but a repeated disable request still calls kxcjk1013_set_power_state(data, false), dropping a runtime PM reference that was not acquired for this request. This can underflow the runtime PM usage count and trigger a "Runtime PM usage count underflow" warning. Return early when the requested state already matches ev_enable_state. Fixes: b4b491c0832e ("iio: accel: kxcjk-1013: Support thresholds") Signed-off-by: Jiale Yao Cc: stable@vger.kernel.org Signed-off-by: Jonathan Cameron --- drivers/iio/accel/kxcjk-1013.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/iio/accel/kxcjk-1013.c b/drivers/iio/accel/kxcjk-1013.c index 166fb786425f8d..8994c9e8e048d0 100644 --- a/drivers/iio/accel/kxcjk-1013.c +++ b/drivers/iio/accel/kxcjk-1013.c @@ -1030,7 +1030,7 @@ static int kxcjk1013_write_event_config(struct iio_dev *indio_dev, struct kxcjk1013_data *data = iio_priv(indio_dev); int ret; - if (state && data->ev_enable_state) + if (state == data->ev_enable_state) return 0; mutex_lock(&data->mutex); From d2f2868d2f487b418f68db2ef5bd7cf41722ee68 Mon Sep 17 00:00:00 2001 From: Fan Wu Date: Wed, 23 Sep 2026 09:48:07 +0000 Subject: [PATCH 0148/1012] iio: adc: ad_sigma_delta: fix use-after-free on unbind MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ad_sd_buffer_postenable() allocates sigma_delta->samples_buf with devm_krealloc() at runtime, so its devres entry sits after all probe-time entries of the driver. devm resources are released in reverse allocation order, which means unbind frees samples_buf before iio_device_unregister() disables the buffers and detaches the trigger pollfunc. The data ready IRQ is still enabled at that point, so ad_sd_trigger_handler() can still run and memcpy() incoming samples into the freed samples_buf. Fix this by preallocating the buffer in devm_ad_sd_setup_buffer_and_trigger(), before the triggered buffer and the IRQ are set up, so it is freed only after iio_device_unregister() has drained the trigger handler via free_irq(). Size it for the worst case of all sequencer slots being active; ad_sd_validate_scan_mask() already caps the number of active channels at num_slots. This issue was found by an in-house static analysis tool. Fixes: 8bea9af887de ("iio: adc: ad_sigma_delta: Add sequencer support") Cc: stable@vger.kernel.org Co-developed-by: Song Li Signed-off-by: Song Li Signed-off-by: Fan Wu Reviewed-by: Nuno Sá Signed-off-by: Jonathan Cameron --- drivers/iio/adc/ad_sigma_delta.c | 31 ++++++++++++++++++------------- 1 file changed, 18 insertions(+), 13 deletions(-) diff --git a/drivers/iio/adc/ad_sigma_delta.c b/drivers/iio/adc/ad_sigma_delta.c index 1b410291da5378..119ddc5a13e27c 100644 --- a/drivers/iio/adc/ad_sigma_delta.c +++ b/drivers/iio/adc/ad_sigma_delta.c @@ -498,7 +498,6 @@ static int ad_sd_buffer_postenable(struct iio_dev *indio_dev) const struct iio_scan_type *scan_type = &indio_dev->channels[0].scan_type; struct spi_transfer *xfer = sigma_delta->sample_xfer; unsigned int i, slot, channel; - u8 *samples_buf; int ret; if (sigma_delta->num_slots == 1) { @@ -530,7 +529,7 @@ static int ad_sd_buffer_postenable(struct iio_dev *indio_dev) xfer[1].bits_per_word = scan_type->realbits; xfer[1].len = spi_bpw_to_bytes(scan_type->realbits); } else { - unsigned int samples_buf_size, scan_size; + unsigned int scan_size; if (sigma_delta->active_slots > 1) { ret = ad_sigma_delta_append_status(sigma_delta, true); @@ -538,17 +537,6 @@ static int ad_sd_buffer_postenable(struct iio_dev *indio_dev) return ret; } - samples_buf_size = - ALIGN(slot * BITS_TO_BYTES(scan_type->storagebits), - sizeof(s64)); - samples_buf_size += sizeof(s64); - samples_buf = devm_krealloc(&sigma_delta->spi->dev, - sigma_delta->samples_buf, - samples_buf_size, GFP_KERNEL); - if (!samples_buf) - return -ENOMEM; - - sigma_delta->samples_buf = samples_buf; scan_size = BITS_TO_BYTES(scan_type->realbits + scan_type->shift); /* For 24-bit data, there is an extra byte of padding. */ xfer[1].rx_buf = &sigma_delta->rx_buf[scan_size == 3 ? 1 : 0]; @@ -855,6 +843,23 @@ int devm_ad_sd_setup_buffer_and_trigger(struct device *dev, struct iio_dev *indi indio_dev->setup_ops = &ad_sd_buffer_setup_ops; } else { + const struct iio_scan_type *scan_type = + &indio_dev->channels[0].scan_type; + unsigned int samples_buf_size; + + /* + * Worst-case size: all sequencer slots can be active, capped + * at num_slots by ad_sd_validate_scan_mask(). + */ + samples_buf_size = + ALIGN(sigma_delta->num_slots * + BITS_TO_BYTES(scan_type->storagebits), + sizeof(s64)); + samples_buf_size += sizeof(s64); + sigma_delta->samples_buf = devm_kzalloc(dev, samples_buf_size, GFP_KERNEL); + if (!sigma_delta->samples_buf) + return -ENOMEM; + ret = devm_iio_triggered_buffer_setup(dev, indio_dev, &iio_pollfunc_store_time, &ad_sd_trigger_handler, From e2fce4620933428a91e8c05e44c3f6b35ea1ac07 Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Sat, 11 Jul 2026 22:59:37 +0200 Subject: [PATCH 0149/1012] arm64: dts: st: Add pinmux nodes for DH electronics STM32MP23xx/STM32MP25xx DHCOS SoM and Breakout Board Add new pinmux nodes for DH electronics STM32MP2 DHCOS SoM and BB board. The following pinmux nodes are added: - ETH2 pins - I2C8 pins - MCO1 pins - SDMMC1,2,3 pins - SPI1,8 pins - UART8,9 pins - USART1,2,6 pins Signed-off-by: Marek Vasut Link: https://lore.kernel.org/r/20260711210131.236025-9-marex@nabladev.com Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi | 504 +++++++++++++++++- 1 file changed, 494 insertions(+), 10 deletions(-) diff --git a/arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi b/arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi index 68cc89cbe754ec..1089c36012cab6 100644 --- a/arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi +++ b/arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi @@ -184,6 +184,30 @@ }; }; + /omit-if-no-ref/ + eth2_mdio_pins_a: eth2-mdio-0 { + pins1 { + pinmux = ; /* ETH_MDC */ + bias-disable; + drive-push-pull; + slew-rate = <3>; + }; + pins2 { + pinmux = ; /* ETH_MDIO */ + bias-disable; + drive-push-pull; + slew-rate = <0>; + }; + }; + + /omit-if-no-ref/ + eth2_mdio_sleep_pins_a: eth2-mdio-sleep-0 { + pins { + pinmux = , /* ETH_MDC */ + ; /* ETH_MDIO */ + }; + }; + /omit-if-no-ref/ eth2_rgmii_sleep_pins_a: eth2-rgmii-sleep-0 { pins { @@ -205,6 +229,58 @@ }; }; + /omit-if-no-ref/ + eth2_rgmii_pins_b: eth2-rgmii-1 { + pins1 { + pinmux = , /* ETH_RGMII_TXD0 */ + , /* ETH_RGMII_TXD1 */ + , /* ETH_RGMII_TXD2 */ + , /* ETH_RGMII_TXD3 */ + ; /* ETH_RGMII_TX_CTL */ + bias-disable; + drive-push-pull; + slew-rate = <3>; + }; + pins2 { + pinmux = , /* ETH_RGMII_CLK125 */ + ; /* ETH_MDC */ + bias-disable; + drive-push-pull; + slew-rate = <3>; + }; + pins3 { + pinmux = , /* ETH_RGMII_RXD0 */ + , /* ETH_RGMII_RXD1 */ + , /* ETH_RGMII_RXD2 */ + , /* ETH_RGMII_RXD3 */ + ; /* ETH_RGMII_RX_CTL */ + bias-disable; + }; + pins4 { + pinmux = ; /* ETH_RGMII_RX_CLK */ + bias-disable; + }; + }; + + /omit-if-no-ref/ + eth2_rgmii_sleep_pins_b: eth2-rgmii-sleep-1 { + pins { + pinmux = , /* ETH_RGMII_TXD0 */ + , /* ETH_RGMII_TXD1 */ + , /* ETH_RGMII_TXD2 */ + , /* ETH_RGMII_TXD3 */ + , /* ETH_RGMII_TX_CTL */ + , /* ETH_RGMII_CLK125 */ + , /* ETH_RGMII_GTX_CLK */ + , /* ETH_RGMII_RXD0 */ + , /* ETH_RGMII_RXD1 */ + , /* ETH_RGMII_RXD2 */ + , /* ETH_RGMII_RXD3 */ + , /* ETH_RGMII_RX_CTL */ + ; /* ETH_RGMII_RX_CLK */ + }; + }; + /omit-if-no-ref/ i2c1_pins_a: i2c1-0 { pins { @@ -355,6 +431,16 @@ }; }; + /omit-if-no-ref/ + mco1_pins_a: mco1-0 { + pins { + pinmux = ; /* MCO1 */ + bias-disable; + drive-push-pull; + slew-rate = <2>; + }; + }; + /omit-if-no-ref/ ospi_port1_clk_pins_a: ospi-port1-clk-0 { pins { @@ -587,6 +673,26 @@ }; }; + /omit-if-no-ref/ + sdmmc1_b4_pins_b: sdmmc1-b4-1 { + pins1 { + pinmux = , /* SDMMC1_D0 */ + , /* SDMMC1_D1 */ + , /* SDMMC1_D2 */ + , /* SDMMC1_D3 */ + ; /* SDMMC1_CMD */ + slew-rate = <1>; + drive-push-pull; + bias-disable; + }; + pins2 { + pinmux = ; /* SDMMC1_CK */ + slew-rate = <2>; + drive-push-pull; + bias-disable; + }; + }; + /omit-if-no-ref/ sdmmc1_b4_od_pins_a: sdmmc1-b4-od-0 { pins1 { @@ -612,6 +718,31 @@ }; }; + /omit-if-no-ref/ + sdmmc1_b4_od_pins_b: sdmmc1-b4-od-1 { + pins1 { + pinmux = , /* SDMMC1_D0 */ + , /* SDMMC1_D1 */ + , /* SDMMC1_D2 */ + ; /* SDMMC1_D3 */ + slew-rate = <1>; + drive-push-pull; + bias-disable; + }; + pins2 { + pinmux = ; /* SDMMC1_CK */ + slew-rate = <2>; + drive-push-pull; + bias-disable; + }; + pins3 { + pinmux = ; /* SDMMC1_CMD */ + slew-rate = <1>; + drive-open-drain; + bias-disable; + }; + }; + /omit-if-no-ref/ sdmmc1_b4_sleep_pins_a: sdmmc1-b4-sleep-0 { pins { @@ -669,6 +800,19 @@ }; }; + /omit-if-no-ref/ + sdmmc2_d47_pins_a: sdmmc2-d47-0 { + pins { + pinmux = , /* SDMMC2_D4 */ + , /* SDMMC2_D5 */ + , /* SDMMC2_D6 */ + ; /* SDMMC2_D7 */ + slew-rate = <1>; + drive-push-pull; + bias-pull-up; + }; + }; + /omit-if-no-ref/ sdmmc2_b4_sleep_pins_a: sdmmc2-b4-sleep-0 { pins { @@ -682,12 +826,29 @@ }; /omit-if-no-ref/ - sdmmc2_d47_pins_a: sdmmc2-d47-0 { + sdmmc2_d47_sleep_pins_a: sdmmc2-d47-sleep-0 { pins { - pinmux = , /* SDMMC2_D4 */ - , /* SDMMC2_D5 */ - , /* SDMMC2_D6 */ - ; /* SDMMC2_D7 */ + pinmux = , /* SDMMC2_D4 */ + , /* SDMMC2_D5 */ + , /* SDMMC2_D6 */ + ; /* SDMMC2_D7 */ + }; + }; + + /omit-if-no-ref/ + sdmmc3_b4_pins_a: sdmmc3-b4-0 { + pins1 { + pinmux = , /* SDMMC3_D0 */ + , /* SDMMC3_D1 */ + , /* SDMMC3_D2 */ + , /* SDMMC3_D3 */ + ; /* SDMMC3_CMD */ + slew-rate = <0>; + drive-push-pull; + bias-pull-up; + }; + pins2 { + pinmux = ; /* SDMMC3_CK */ slew-rate = <1>; drive-push-pull; bias-pull-up; @@ -695,12 +856,59 @@ }; /omit-if-no-ref/ - sdmmc2_d47_sleep_pins_a: sdmmc2-d47-sleep-0 { + sdmmc3_b4_pins_b: sdmmc3-b4-1 { + pins1 { + pinmux = , /* SDMMC3_D0 */ + , /* SDMMC3_D1 */ + , /* SDMMC3_D2 */ + , /* SDMMC3_D3 */ + ; /* SDMMC3_CMD */ + slew-rate = <0>; + drive-push-pull; + bias-pull-up; + }; + pins2 { + pinmux = ; /* SDMMC3_CK */ + slew-rate = <1>; + drive-push-pull; + bias-pull-up; + }; + }; + + /omit-if-no-ref/ + sdmmc3_b4_od_pins_b: sdmmc3-b4-od-1 { + pins1 { + pinmux = , /* SDMMC3_D0 */ + , /* SDMMC3_D1 */ + , /* SDMMC3_D2 */ + ; /* SDMMC3_D3 */ + slew-rate = <2>; + drive-push-pull; + bias-disable; + }; + pins2 { + pinmux = ; /* SDMMC3_CK */ + slew-rate = <3>; + drive-push-pull; + bias-disable; + }; + pins3 { + pinmux = ; /* SDMMC3_CMD */ + slew-rate = <2>; + drive-open-drain; + bias-disable; + }; + }; + + /omit-if-no-ref/ + sdmmc3_b4_sleep_pins_b: sdmmc3-b4-sleep-1 { pins { - pinmux = , /* SDMMC2_D4 */ - , /* SDMMC2_D5 */ - , /* SDMMC2_D6 */ - ; /* SDMMC2_D7 */ + pinmux = , /* SDMMC3_D0 */ + , /* SDMMC3_D1 */ + , /* SDMMC3_D2 */ + , /* SDMMC3_D3 */ + , /* SDMMC3_CK */ + ; /* SDMMC3_CMD */ }; }; @@ -728,6 +936,30 @@ }; }; + /omit-if-no-ref/ + spi1_pins_b: spi1-1 { + pins1 { + pinmux = , /* SPI1_SCK */ + ; /* SPI1_MOSI */ + drive-push-pull; + bias-disable; + slew-rate = <1>; + }; + pins2 { + pinmux = ; /* SPI1_MISO */ + bias-disable; + }; + }; + + /omit-if-no-ref/ + spi1_sleep_pins_b: spi1-sleep-1 { + pins1 { + pinmux = , /* SPI1_SCK */ + , /* SPI1_MOSI */ + ; /* SPI1_MISO */ + }; + }; + /omit-if-no-ref/ spi3_pins_a: spi3-0 { pins1 { @@ -801,6 +1033,50 @@ }; }; + /omit-if-no-ref/ + usart1_pins_a: usart1-0 { + pins1 { + pinmux = , /* USART1_TX */ + ; /* USART1_RTS */ + bias-disable; + drive-push-pull; + slew-rate = <0>; + }; + pins2 { + pinmux = , /* USART1_RX */ + ; /* USART1_CTS_NSS */ + bias-pull-up; + }; + }; + + /omit-if-no-ref/ + usart1_idle_pins_a: usart1-idle-0 { + pins1 { + pinmux = , /* USART1_TX */ + ; /* USART1_CTS_NSS */ + }; + pins2 { + pinmux = ; /* USART1_RTS */ + bias-disable; + drive-push-pull; + slew-rate = <0>; + }; + pins3 { + pinmux = ; /* USART1_RX */ + bias-pull-up; + }; + }; + + /omit-if-no-ref/ + usart1_sleep_pins_a: usart1-sleep-0 { + pins { + pinmux = , /* USART1_TX */ + , /* USART1_RTS */ + , /* USART1_CTS_NSS */ + ; /* USART1_RX */ + }; + }; + /omit-if-no-ref/ usart2_pins_a: usart2-0 { pins1 { @@ -834,6 +1110,50 @@ }; }; + /omit-if-no-ref/ + usart2_pins_b: usart2-1 { + pins1 { + pinmux = , /* USART2_TX */ + ; /* USART2_RTS */ + bias-disable; + drive-push-pull; + slew-rate = <0>; + }; + pins2 { + pinmux = , /* USART2_RX */ + ; /* USART2_CTS_NSS */ + bias-disable; + }; + }; + + /omit-if-no-ref/ + usart2_idle_pins_b: usart2-idle-1 { + pins1 { + pinmux = , /* USART2_TX */ + ; /* USART2_CTS_NSS */ + }; + pins2 { + pinmux = ; /* USART2_RTS */ + bias-disable; + drive-push-pull; + slew-rate = <0>; + }; + pins3 { + pinmux = ; /* USART2_RX */ + bias-pull-up; + }; + }; + + /omit-if-no-ref/ + usart2_sleep_pins_b: usart2-sleep-1 { + pins { + pinmux = , /* USART2_TX */ + , /* USART2_RTS */ + , /* USART2_CTS_NSS */ + ; /* USART2_RX */ + }; + }; + /omit-if-no-ref/ usart6_pins_a: usart6-0 { pins1 { @@ -877,6 +1197,127 @@ ; /* USART6_RX */ }; }; + + /omit-if-no-ref/ + usart6_pins_b: usart6-1 { + pins1 { + pinmux = ; /* USART6_TX */ + bias-disable; + drive-push-pull; + slew-rate = <0>; + }; + pins2 { + pinmux = ; /* USART6_RX */ + bias-disable; + }; + }; + + /omit-if-no-ref/ + usart6_idle_pins_b: usart6-idle-1 { + pins1 { + pinmux = ; /* USART6_TX */ + }; + pins2 { + pinmux = ; /* USART6_RX */ + bias-disable; + }; + }; + + /omit-if-no-ref/ + usart6_sleep_pins_b: usart6-sleep-1 { + pins { + pinmux = , /* USART6_TX */ + ; /* USART6_RX */ + }; + }; + + /omit-if-no-ref/ + uart8_pins_a: uart8-0 { + pins1 { + pinmux = , /* UART8_TX */ + ; /* UART8_RTS */ + bias-disable; + drive-push-pull; + slew-rate = <0>; + }; + pins2 { + pinmux = , /* UART8_RX */ + ; /* UART8_CTS_NSS */ + bias-pull-up; + }; + }; + + /omit-if-no-ref/ + uart8_idle_pins_a: uart8-idle-0 { + pins1 { + pinmux = , /* UART8_TX */ + ; /* UART8_CTS_NSS */ + }; + pins2 { + pinmux = ; /* UART8_RTS */ + bias-disable; + drive-push-pull; + slew-rate = <0>; + }; + pins3 { + pinmux = ; /* UART8_RX */ + bias-pull-up; + }; + }; + + /omit-if-no-ref/ + uart8_sleep_pins_a: uart8-sleep-0 { + pins { + pinmux = , /* UART8_TX */ + , /* UART8_RTS */ + , /* UART8_CTS_NSS */ + ; /* UART8_RX */ + }; + }; + + /omit-if-no-ref/ + uart9_pins_a: uart9-0 { + pins1 { + pinmux = , /* UART9_TX */ + ; /* UART9_RTS */ + bias-disable; + drive-push-pull; + slew-rate = <0>; + }; + pins2 { + pinmux = , /* UART9_RX */ + ; /* UART9_CTS_NSS */ + bias-pull-up; + }; + }; + + /omit-if-no-ref/ + uart9_idle_pins_a: uart9-idle-0 { + pins1 { + pinmux = , /* UART9_TX */ + ; /* UART9_CTS_NSS */ + }; + pins2 { + pinmux = ; /* UART9_RTS */ + bias-disable; + drive-push-pull; + slew-rate = <0>; + }; + pins3 { + pinmux = ; /* UART9_RX */ + bias-pull-up; + }; + }; + + /omit-if-no-ref/ + uart9_sleep_pins_a: uart9-sleep-0 { + pins { + pinmux = , /* UART9_TX */ + , /* UART9_RTS */ + , /* UART9_CTS_NSS */ + ; /* UART9_RX */ + }; + }; }; &pinctrl_z { @@ -899,6 +1340,25 @@ }; }; + /omit-if-no-ref/ + i2c8_pins_b: i2c8-1 { + pins { + pinmux = , /* I2C1_SCL */ + ; /* I2C1_SDA */ + bias-disable; + drive-open-drain; + slew-rate = <0>; + }; + }; + + /omit-if-no-ref/ + i2c8_sleep_pins_b: i2c8-sleep-1 { + pins { + pinmux = , /* I2C1_SCL */ + ; /* I2C1_SDA */ + }; + }; + /omit-if-no-ref/ spi8_pins_a: spi8-0 { pins1 { @@ -922,4 +1382,28 @@ ; /* SPI8_MISO */ }; }; + + /omit-if-no-ref/ + spi8_pins_b: spi8-1 { + pins1 { + pinmux = , /* SPI8_SCK */ + ; /* SPI8_MOSI */ + drive-push-pull; + bias-disable; + slew-rate = <1>; + }; + pins2 { + pinmux = ; /* SPI8_MISO */ + bias-disable; + }; + }; + + /omit-if-no-ref/ + spi8_sleep_pins_b: spi8-sleep-1 { + pins1 { + pinmux = , /* SPI8_SCK */ + , /* SPI8_MOSI */ + ; /* SPI8_MISO */ + }; + }; }; From 77613eff2d02ed6cb611d5b7e01d6a2a94a4a6fa Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Tue, 15 Sep 2026 13:15:32 +0200 Subject: [PATCH 0150/1012] fuse: use "vdax" naming for virtiofs DAX We are introducing DAX functionality largely unrelated to the virtiofs code. Rename dax -> vdax for the virtiofs case for clarity. Also rename CONFIG_FUSE_DAX -> CONFIG_FUSE_VDAX. Reviewed-by: Amir Goldstein Signed-off-by: Miklos Szeredi --- fs/fuse/Kconfig | 8 +- fs/fuse/Makefile | 2 +- fs/fuse/dax.c | 214 ++++++++++++++++++++++---------------------- fs/fuse/dir.c | 4 +- fs/fuse/file.c | 40 ++++----- fs/fuse/fuse_i.h | 70 +++++++-------- fs/fuse/inode.c | 54 +++++------ fs/fuse/iomode.c | 4 +- fs/fuse/virtio_fs.c | 22 ++--- 9 files changed, 211 insertions(+), 207 deletions(-) diff --git a/fs/fuse/Kconfig b/fs/fuse/Kconfig index 3a4ae632c94aa8..8c41838b1c3d9a 100644 --- a/fs/fuse/Kconfig +++ b/fs/fuse/Kconfig @@ -40,9 +40,9 @@ config VIRTIO_FS If you want to share files between guests or with the host, answer Y or M. -config FUSE_DAX +config FUSE_VDAX bool "Virtio Filesystem Direct Host Memory Access support" - default y + default FUSE_DAX select INTERVAL_TREE depends on VIRTIO_FS depends on FS_DAX @@ -54,6 +54,10 @@ config FUSE_DAX If you want to allow mounting a Virtio Filesystem with the "dax" option, answer Y. +config FUSE_DAX + bool + transitional + config FUSE_PASSTHROUGH bool "FUSE passthrough operations support" default y diff --git a/fs/fuse/Makefile b/fs/fuse/Makefile index 245e67852b03e8..5858feafa91647 100644 --- a/fs/fuse/Makefile +++ b/fs/fuse/Makefile @@ -14,7 +14,7 @@ fuse-y := trace.o # put trace.o first so we see ftrace errors sooner fuse-y += dev.o dir.o file.o inode.o control.o xattr.o acl.o readdir.o ioctl.o req_timeout.o req.o fuse-y += poll.o notify.o fuse-y += iomode.o -fuse-$(CONFIG_FUSE_DAX) += dax.o +fuse-$(CONFIG_FUSE_VDAX) += dax.o fuse-$(CONFIG_FUSE_PASSTHROUGH) += passthrough.o backing.o fuse-$(CONFIG_SYSCTL) += sysctl.o fuse-$(CONFIG_FUSE_IO_URING) += dev_uring.o diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index 32c88ef814340b..7905e7b3664448 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0 /* - * dax: direct host memory access + * dax: direct host memory access for virtiofs * Copyright (C) 2020 Red Hat, Inc. */ @@ -59,7 +59,7 @@ struct fuse_dax_mapping { }; /* Per-inode dax map */ -struct fuse_inode_dax { +struct fuse_inode_vdax { /* Semaphore to protect modifications to the dmap tree */ struct rw_semaphore sem; @@ -68,7 +68,7 @@ struct fuse_inode_dax { unsigned long nr; }; -struct fuse_conn_dax { +struct fuse_conn_vdax { /* DAX device */ struct dax_device *dev; @@ -102,10 +102,10 @@ node_to_dmap(struct interval_tree_node *node) } static struct fuse_dax_mapping * -alloc_dax_mapping_reclaim(struct fuse_conn_dax *fcd, struct inode *inode); +alloc_dax_mapping_reclaim(struct fuse_conn_vdax *fcd, struct inode *inode); static void -__kick_dmap_free_worker(struct fuse_conn_dax *fcd, unsigned long delay_ms) +__kick_dmap_free_worker(struct fuse_conn_vdax *fcd, unsigned long delay_ms) { unsigned long free_threshold; @@ -117,7 +117,7 @@ __kick_dmap_free_worker(struct fuse_conn_dax *fcd, unsigned long delay_ms) msecs_to_jiffies(delay_ms)); } -static void kick_dmap_free_worker(struct fuse_conn_dax *fcd, +static void kick_dmap_free_worker(struct fuse_conn_vdax *fcd, unsigned long delay_ms) { spin_lock(&fcd->lock); @@ -125,7 +125,7 @@ static void kick_dmap_free_worker(struct fuse_conn_dax *fcd, spin_unlock(&fcd->lock); } -static struct fuse_dax_mapping *alloc_dax_mapping(struct fuse_conn_dax *fcd) +static struct fuse_dax_mapping *alloc_dax_mapping(struct fuse_conn_vdax *fcd) { struct fuse_dax_mapping *dmap; @@ -144,7 +144,7 @@ static struct fuse_dax_mapping *alloc_dax_mapping(struct fuse_conn_dax *fcd) } /* This assumes fcd->lock is held */ -static void __dmap_remove_busy_list(struct fuse_conn_dax *fcd, +static void __dmap_remove_busy_list(struct fuse_conn_vdax *fcd, struct fuse_dax_mapping *dmap) { list_del_init(&dmap->busy_list); @@ -152,7 +152,7 @@ static void __dmap_remove_busy_list(struct fuse_conn_dax *fcd, fcd->nr_busy_ranges--; } -static void dmap_remove_busy_list(struct fuse_conn_dax *fcd, +static void dmap_remove_busy_list(struct fuse_conn_vdax *fcd, struct fuse_dax_mapping *dmap) { spin_lock(&fcd->lock); @@ -161,7 +161,7 @@ static void dmap_remove_busy_list(struct fuse_conn_dax *fcd, } /* This assumes fcd->lock is held */ -static void __dmap_add_to_free_pool(struct fuse_conn_dax *fcd, +static void __dmap_add_to_free_pool(struct fuse_conn_vdax *fcd, struct fuse_dax_mapping *dmap) { list_add_tail(&dmap->list, &fcd->free_ranges); @@ -169,7 +169,7 @@ static void __dmap_add_to_free_pool(struct fuse_conn_dax *fcd, wake_up(&fcd->range_waitq); } -static void dmap_add_to_free_pool(struct fuse_conn_dax *fcd, +static void dmap_add_to_free_pool(struct fuse_conn_vdax *fcd, struct fuse_dax_mapping *dmap) { /* Return fuse_dax_mapping to free list */ @@ -183,7 +183,7 @@ static int fuse_setup_one_mapping(struct inode *inode, unsigned long start_idx, bool upgrade) { struct fuse_mount *fm = get_fuse_mount(inode); - struct fuse_conn_dax *fcd = fm->fc->dax; + struct fuse_conn_vdax *fcd = fm->fc->vdax; struct fuse_inode *fi = get_fuse_inode(inode); struct fuse_setupmapping_in inarg; loff_t offset = start_idx << FUSE_DAX_SHIFT; @@ -218,9 +218,9 @@ static int fuse_setup_one_mapping(struct inode *inode, unsigned long start_idx, */ dmap->inode = inode; dmap->itn.start = dmap->itn.last = start_idx; - /* Protected by fi->dax->sem */ - interval_tree_insert(&dmap->itn, &fi->dax->tree); - fi->dax->nr++; + /* Protected by fi->vdax->sem */ + interval_tree_insert(&dmap->itn, &fi->vdax->tree); + fi->vdax->nr++; spin_lock(&fcd->lock); list_add_tail(&dmap->busy_list, &fcd->busy_ranges); fcd->nr_busy_ranges++; @@ -288,7 +288,7 @@ static int dmap_removemapping_list(struct inode *inode, unsigned int num, * Cleanup dmap entry and add back to free list. This should be called with * fcd->lock held. */ -static void dmap_reinit_add_to_free_pool(struct fuse_conn_dax *fcd, +static void dmap_reinit_add_to_free_pool(struct fuse_conn_vdax *fcd, struct fuse_dax_mapping *dmap) { pr_debug("fuse: freeing memory range start_idx=0x%lx end_idx=0x%lx window_offset=0x%llx length=0x%llx\n", @@ -306,7 +306,7 @@ static void dmap_reinit_add_to_free_pool(struct fuse_conn_dax *fcd, * called from evict_inode() path where we know all dmap entries can be * reclaimed. */ -static void inode_reclaim_dmap_range(struct fuse_conn_dax *fcd, +static void inode_reclaim_dmap_range(struct fuse_conn_vdax *fcd, struct inode *inode, loff_t start, loff_t end) { @@ -319,14 +319,14 @@ static void inode_reclaim_dmap_range(struct fuse_conn_dax *fcd, struct interval_tree_node *node; while (1) { - node = interval_tree_iter_first(&fi->dax->tree, start_idx, + node = interval_tree_iter_first(&fi->vdax->tree, start_idx, end_idx); if (!node) break; dmap = node_to_dmap(node); /* inode is going away. There should not be any users of dmap */ WARN_ON(refcount_read(&dmap->refcnt) > 1); - interval_tree_remove(&dmap->itn, &fi->dax->tree); + interval_tree_remove(&dmap->itn, &fi->vdax->tree); num++; list_add(&dmap->list, &to_remove); } @@ -335,8 +335,8 @@ static void inode_reclaim_dmap_range(struct fuse_conn_dax *fcd, if (list_empty(&to_remove)) return; - WARN_ON(fi->dax->nr < num); - fi->dax->nr -= num; + WARN_ON(fi->vdax->nr < num); + fi->vdax->nr -= num; err = dmap_removemapping_list(inode, num, &to_remove); if (err && err != -ENOTCONN) { pr_warn("Failed to removemappings. start=0x%llx end=0x%llx\n", @@ -367,11 +367,11 @@ static int dmap_removemapping_one(struct inode *inode, /* * It is called from evict_inode() and by that time inode is going away. So - * this function does not take any locks like fi->dax->sem for traversing + * this function does not take any locks like fi->vdax->sem for traversing * that fuse inode interval tree. If that lock is taken then lock validator * complains of deadlock situation w.r.t fs_reclaim lock. */ -void fuse_dax_inode_cleanup(struct inode *inode) +void fuse_vdax_inode_cleanup(struct inode *inode) { struct fuse_conn *fc = get_fuse_conn(inode); struct fuse_inode *fi = get_fuse_inode(inode); @@ -381,8 +381,8 @@ void fuse_dax_inode_cleanup(struct inode *inode) * before we arrive here. So we should not have to worry about any * pages/exception entries still associated with inode. */ - inode_reclaim_dmap_range(fc->dax, inode, 0, -1); - WARN_ON(fi->dax->nr); + inode_reclaim_dmap_range(fc->vdax, inode, 0, -1); + WARN_ON(fi->vdax->nr); } static void fuse_fill_iomap_hole(struct iomap *iomap, loff_t length) @@ -414,7 +414,7 @@ static void fuse_fill_iomap(struct inode *inode, loff_t pos, loff_t length, iomap->type = IOMAP_MAPPED; /* * increace refcnt so that reclaim code knows this dmap is in - * use. This assumes fi->dax->sem mutex is held either + * use. This assumes fi->vdax->sem mutex is held either * shared/exclusive. */ refcount_inc(&dmap->refcnt); @@ -434,7 +434,7 @@ static int fuse_setup_new_dax_mapping(struct inode *inode, loff_t pos, { struct fuse_inode *fi = get_fuse_inode(inode); struct fuse_conn *fc = get_fuse_conn(inode); - struct fuse_conn_dax *fcd = fc->dax; + struct fuse_conn_vdax *fcd = fc->vdax; struct fuse_dax_mapping *dmap, *alloc_dmap = NULL; int ret; bool writable = flags & IOMAP_WRITE; @@ -469,17 +469,17 @@ static int fuse_setup_new_dax_mapping(struct inode *inode, loff_t pos, * Take write lock so that only one caller can try to setup mapping * and other waits. */ - down_write(&fi->dax->sem); + down_write(&fi->vdax->sem); /* * We dropped lock. Check again if somebody else setup * mapping already. */ - node = interval_tree_iter_first(&fi->dax->tree, start_idx, start_idx); + node = interval_tree_iter_first(&fi->vdax->tree, start_idx, start_idx); if (node) { dmap = node_to_dmap(node); fuse_fill_iomap(inode, pos, length, iomap, dmap, flags); dmap_add_to_free_pool(fcd, alloc_dmap); - up_write(&fi->dax->sem); + up_write(&fi->vdax->sem); return 0; } @@ -488,11 +488,11 @@ static int fuse_setup_new_dax_mapping(struct inode *inode, loff_t pos, writable, false); if (ret < 0) { dmap_add_to_free_pool(fcd, alloc_dmap); - up_write(&fi->dax->sem); + up_write(&fi->vdax->sem); return ret; } fuse_fill_iomap(inode, pos, length, iomap, alloc_dmap, flags); - up_write(&fi->dax->sem); + up_write(&fi->vdax->sem); return 0; } @@ -510,14 +510,14 @@ static int fuse_upgrade_dax_mapping(struct inode *inode, loff_t pos, * Take exclusive lock so that only one caller can try to setup * mapping and others wait. */ - down_write(&fi->dax->sem); - node = interval_tree_iter_first(&fi->dax->tree, idx, idx); + down_write(&fi->vdax->sem); + node = interval_tree_iter_first(&fi->vdax->tree, idx, idx); /* We are holding either inode lock or invalidate_lock, and that should * ensure that dmap can't be truncated. We are holding a reference * on dmap and that should make sure it can't be reclaimed. So dmap * should still be there in tree despite the fact we dropped and - * re-acquired the fi->dax->sem lock. + * re-acquired the fi->vdax->sem lock. */ ret = -EIO; if (WARN_ON(!node)) @@ -526,7 +526,7 @@ static int fuse_upgrade_dax_mapping(struct inode *inode, loff_t pos, dmap = node_to_dmap(node); /* We took an extra reference on dmap to make sure its not reclaimd. - * Now we hold fi->dax->sem lock and that reference is not needed + * Now we hold fi->vdax->sem lock and that reference is not needed * anymore. Drop it. */ if (refcount_dec_and_test(&dmap->refcnt)) { @@ -551,7 +551,7 @@ static int fuse_upgrade_dax_mapping(struct inode *inode, loff_t pos, out_fill_iomap: fuse_fill_iomap(inode, pos, length, iomap, dmap, flags); out_err: - up_write(&fi->dax->sem); + up_write(&fi->vdax->sem); return ret; } @@ -576,7 +576,7 @@ static int fuse_iomap_begin(struct inode *inode, loff_t pos, loff_t length, iomap->offset = pos; iomap->flags = 0; iomap->bdev = NULL; - iomap->dax_dev = fc->dax->dev; + iomap->dax_dev = fc->vdax->dev; /* * Both read/write and mmap path can race here. So we need something @@ -585,33 +585,33 @@ static int fuse_iomap_begin(struct inode *inode, loff_t pos, loff_t length, * For now, use a semaphore for this. It probably needs to be * optimized later. */ - down_read(&fi->dax->sem); - node = interval_tree_iter_first(&fi->dax->tree, start_idx, start_idx); + down_read(&fi->vdax->sem); + node = interval_tree_iter_first(&fi->vdax->tree, start_idx, start_idx); if (node) { dmap = node_to_dmap(node); if (writable && !dmap->writable) { /* Upgrade read-only mapping to read-write. This will - * require exclusive fi->dax->sem lock as we don't want + * require exclusive fi->vdax->sem lock as we don't want * two threads to be trying to this simultaneously * for same dmap. So drop shared lock and acquire * exclusive lock. * - * Before dropping fi->dax->sem lock, take reference + * Before dropping fi->vdax->sem lock, take reference * on dmap so that its not freed by range reclaim. */ refcount_inc(&dmap->refcnt); - up_read(&fi->dax->sem); + up_read(&fi->vdax->sem); pr_debug("%s: Upgrading mapping at offset 0x%llx length 0x%llx\n", __func__, pos, length); return fuse_upgrade_dax_mapping(inode, pos, length, flags, iomap); } else { fuse_fill_iomap(inode, pos, length, iomap, dmap, flags); - up_read(&fi->dax->sem); + up_read(&fi->vdax->sem); return 0; } } else { - up_read(&fi->dax->sem); + up_read(&fi->vdax->sem); pr_debug("%s: no mapping at offset 0x%llx length 0x%llx\n", __func__, pos, length); if (pos >= i_size_read(inode)) @@ -668,14 +668,14 @@ static void fuse_wait_dax_page(struct inode *inode) } /* Should be called with mapping->invalidate_lock held exclusively. */ -int fuse_dax_break_layouts(struct inode *inode, u64 dmap_start, +int fuse_vdax_break_layouts(struct inode *inode, u64 dmap_start, u64 dmap_end) { return dax_break_layout(inode, dmap_start, dmap_end, fuse_wait_dax_page); } -ssize_t fuse_dax_read_iter(struct kiocb *iocb, struct iov_iter *to) +ssize_t fuse_vdax_read_iter(struct kiocb *iocb, struct iov_iter *to) { struct inode *inode = file_inode(iocb->ki_filp); ssize_t ret; @@ -715,7 +715,7 @@ static ssize_t fuse_dax_direct_write(struct kiocb *iocb, struct iov_iter *from) return ret; } -ssize_t fuse_dax_write_iter(struct kiocb *iocb, struct iov_iter *from) +ssize_t fuse_vdax_write_iter(struct kiocb *iocb, struct iov_iter *from) { struct inode *inode = file_inode(iocb->ki_filp); ssize_t ret; @@ -761,7 +761,7 @@ static vm_fault_t __fuse_dax_fault(struct vm_fault *vmf, unsigned int order, unsigned long pfn; int error = 0; struct fuse_conn *fc = get_fuse_conn(inode); - struct fuse_conn_dax *fcd = fc->dax; + struct fuse_conn_vdax *fcd = fc->vdax; bool retry = false; if (write) @@ -822,7 +822,7 @@ static const struct vm_operations_struct fuse_dax_vm_ops = { .pfn_mkwrite = fuse_dax_pfn_mkwrite, }; -int fuse_dax_mmap(struct file *file, struct vm_area_struct *vma) +int fuse_vdax_mmap(struct file *file, struct vm_area_struct *vma) { file_accessed(file); vma->vm_ops = &fuse_dax_vm_ops; @@ -870,8 +870,8 @@ static int reclaim_one_dmap_locked(struct inode *inode, return ret; /* Remove dax mapping from inode interval tree now */ - interval_tree_remove(&dmap->itn, &fi->dax->tree); - fi->dax->nr--; + interval_tree_remove(&dmap->itn, &fi->vdax->tree); + fi->vdax->nr--; /* It is possible that umount/shutdown has killed the fuse connection * and worker thread is trying to reclaim memory in parallel. Don't @@ -886,7 +886,7 @@ static int reclaim_one_dmap_locked(struct inode *inode, } /* Find first mapped dmap for an inode and return file offset. Caller needs - * to hold fi->dax->sem lock either shared or exclusive. + * to hold fi->vdax->sem lock either shared or exclusive. */ static struct fuse_dax_mapping *inode_lookup_first_dmap(struct inode *inode) { @@ -894,7 +894,7 @@ static struct fuse_dax_mapping *inode_lookup_first_dmap(struct inode *inode) struct fuse_dax_mapping *dmap; struct interval_tree_node *node; - for (node = interval_tree_iter_first(&fi->dax->tree, 0, -1); node; + for (node = interval_tree_iter_first(&fi->vdax->tree, 0, -1); node; node = interval_tree_iter_next(node, 0, -1)) { dmap = node_to_dmap(node); /* still in use. */ @@ -912,7 +912,7 @@ static struct fuse_dax_mapping *inode_lookup_first_dmap(struct inode *inode) * it back to free pool. */ static struct fuse_dax_mapping * -inode_inline_reclaim_one_dmap(struct fuse_conn_dax *fcd, struct inode *inode, +inode_inline_reclaim_one_dmap(struct fuse_conn_vdax *fcd, struct inode *inode, bool *retry) { struct fuse_inode *fi = get_fuse_inode(inode); @@ -925,14 +925,14 @@ inode_inline_reclaim_one_dmap(struct fuse_conn_dax *fcd, struct inode *inode, filemap_invalidate_lock(inode->i_mapping); /* Lookup a dmap and corresponding file offset to reclaim. */ - down_read(&fi->dax->sem); + down_read(&fi->vdax->sem); dmap = inode_lookup_first_dmap(inode); if (dmap) { start_idx = dmap->itn.start; dmap_start = start_idx << FUSE_DAX_SHIFT; dmap_end = dmap_start + FUSE_DAX_SZ - 1; } - up_read(&fi->dax->sem); + up_read(&fi->vdax->sem); if (!dmap) goto out_mmap_sem; @@ -940,16 +940,16 @@ inode_inline_reclaim_one_dmap(struct fuse_conn_dax *fcd, struct inode *inode, * Make sure there are no references to inode pages using * get_user_pages() */ - ret = fuse_dax_break_layouts(inode, dmap_start, dmap_end); + ret = fuse_vdax_break_layouts(inode, dmap_start, dmap_end); if (ret) { - pr_debug("fuse: fuse_dax_break_layouts() failed. err=%d\n", + pr_debug("fuse: fuse_vdax_break_layouts() failed. err=%d\n", ret); dmap = ERR_PTR(ret); goto out_mmap_sem; } - down_write(&fi->dax->sem); - node = interval_tree_iter_first(&fi->dax->tree, start_idx, start_idx); + down_write(&fi->vdax->sem); + node = interval_tree_iter_first(&fi->vdax->tree, start_idx, start_idx); /* Range already got reclaimed by somebody else */ if (!node) { if (retry) @@ -981,14 +981,14 @@ inode_inline_reclaim_one_dmap(struct fuse_conn_dax *fcd, struct inode *inode, __func__, inode, dmap->window_offset, dmap->length); out_write_dmap_sem: - up_write(&fi->dax->sem); + up_write(&fi->vdax->sem); out_mmap_sem: filemap_invalidate_unlock(inode->i_mapping); return dmap; } static struct fuse_dax_mapping * -alloc_dax_mapping_reclaim(struct fuse_conn_dax *fcd, struct inode *inode) +alloc_dax_mapping_reclaim(struct fuse_conn_vdax *fcd, struct inode *inode) { struct fuse_dax_mapping *dmap; struct fuse_inode *fi = get_fuse_inode(inode); @@ -1015,18 +1015,18 @@ alloc_dax_mapping_reclaim(struct fuse_conn_dax *fcd, struct inode *inode) * if a deadlock is possible if we sleep with * mapping->invalidate_lock held and worker to free memory * can't make progress due to unavailability of - * mapping->invalidate_lock. So sleep only if fi->dax->nr=0 + * mapping->invalidate_lock. So sleep only if fi->vdax->nr=0 */ if (retry) continue; /* * There are no mappings which can be reclaimed. Wait for one. - * We are not holding fi->dax->sem. So it is possible + * We are not holding fi->vdax->sem. So it is possible * that range gets added now. But as we are not holding * mapping->invalidate_lock, worker should still be able to * free up a range and wake us up. */ - if (!fi->dax->nr && !(fcd->nr_free_ranges > 0)) { + if (!fi->vdax->nr && !(fcd->nr_free_ranges > 0)) { if (wait_event_killable_exclusive(fcd->range_waitq, (fcd->nr_free_ranges > 0))) { return ERR_PTR(-EINTR); @@ -1035,7 +1035,7 @@ alloc_dax_mapping_reclaim(struct fuse_conn_dax *fcd, struct inode *inode) } } -static int lookup_and_reclaim_dmap_locked(struct fuse_conn_dax *fcd, +static int lookup_and_reclaim_dmap_locked(struct fuse_conn_vdax *fcd, struct inode *inode, unsigned long start_idx) { @@ -1045,7 +1045,7 @@ static int lookup_and_reclaim_dmap_locked(struct fuse_conn_dax *fcd, struct interval_tree_node *node; /* Find fuse dax mapping at file offset inode. */ - node = interval_tree_iter_first(&fi->dax->tree, start_idx, start_idx); + node = interval_tree_iter_first(&fi->vdax->tree, start_idx, start_idx); /* Range already got cleaned up by somebody else */ if (!node) @@ -1071,10 +1071,10 @@ static int lookup_and_reclaim_dmap_locked(struct fuse_conn_dax *fcd, * Free a range of memory. * Locking: * 1. Take mapping->invalidate_lock to block dax faults. - * 2. Take fi->dax->sem to protect interval tree and also to make sure + * 2. Take fi->vdax->sem to protect interval tree and also to make sure * read/write can not reuse a dmap which we might be freeing. */ -static int lookup_and_reclaim_dmap(struct fuse_conn_dax *fcd, +static int lookup_and_reclaim_dmap(struct fuse_conn_vdax *fcd, struct inode *inode, unsigned long start_idx, unsigned long end_idx) @@ -1085,22 +1085,22 @@ static int lookup_and_reclaim_dmap(struct fuse_conn_dax *fcd, loff_t dmap_end = (dmap_start + FUSE_DAX_SZ) - 1; filemap_invalidate_lock(inode->i_mapping); - ret = fuse_dax_break_layouts(inode, dmap_start, dmap_end); + ret = fuse_vdax_break_layouts(inode, dmap_start, dmap_end); if (ret) { - pr_debug("virtio_fs: fuse_dax_break_layouts() failed. err=%d\n", + pr_debug("virtio_fs: fuse_vdax_break_layouts() failed. err=%d\n", ret); goto out_mmap_sem; } - down_write(&fi->dax->sem); + down_write(&fi->vdax->sem); ret = lookup_and_reclaim_dmap_locked(fcd, inode, start_idx); - up_write(&fi->dax->sem); + up_write(&fi->vdax->sem); out_mmap_sem: filemap_invalidate_unlock(inode->i_mapping); return ret; } -static int try_to_free_dmap_chunks(struct fuse_conn_dax *fcd, +static int try_to_free_dmap_chunks(struct fuse_conn_vdax *fcd, unsigned long nr_to_free) { struct fuse_dax_mapping *dmap, *pos, *temp; @@ -1161,7 +1161,7 @@ static int try_to_free_dmap_chunks(struct fuse_conn_dax *fcd, static void fuse_dax_free_mem_worker(struct work_struct *work) { int ret; - struct fuse_conn_dax *fcd = container_of(work, struct fuse_conn_dax, + struct fuse_conn_vdax *fcd = container_of(work, struct fuse_conn_vdax, free_work.work); ret = try_to_free_dmap_chunks(fcd, FUSE_DAX_RECLAIM_CHUNK); if (ret) { @@ -1186,16 +1186,16 @@ static void fuse_free_dax_mem_ranges(struct list_head *mem_list) } } -void fuse_dax_conn_free(struct fuse_conn *fc) +void fuse_vdax_conn_free(struct fuse_conn *fc) { - if (fc->dax) { - fuse_free_dax_mem_ranges(&fc->dax->free_ranges); - kfree(fc->dax); - fc->dax = NULL; + if (fc->vdax) { + fuse_free_dax_mem_ranges(&fc->vdax->free_ranges); + kfree(fc->vdax); + fc->vdax = NULL; } } -static int fuse_dax_mem_range_init(struct fuse_conn_dax *fcd) +static int fuse_dax_mem_range_init(struct fuse_conn_vdax *fcd) { long nr_pages, nr_ranges; struct fuse_dax_mapping *range; @@ -1247,13 +1247,13 @@ static int fuse_dax_mem_range_init(struct fuse_conn_dax *fcd) return ret; } -int fuse_dax_conn_alloc(struct fuse_conn *fc, enum fuse_dax_mode dax_mode, +int fuse_vdax_conn_alloc(struct fuse_conn *fc, enum fuse_vdax_mode dax_mode, struct dax_device *dax_dev) { - struct fuse_conn_dax *fcd; + struct fuse_conn_vdax *fcd; int err; - fc->dax_mode = dax_mode; + fc->vdax_mode = dax_mode; if (!dax_dev) return 0; @@ -1270,22 +1270,22 @@ int fuse_dax_conn_alloc(struct fuse_conn *fc, enum fuse_dax_mode dax_mode, return err; } - fc->dax = fcd; + fc->vdax = fcd; return 0; } -bool fuse_dax_inode_alloc(struct super_block *sb, struct fuse_inode *fi) +bool fuse_vdax_inode_alloc(struct super_block *sb, struct fuse_inode *fi) { struct fuse_conn *fc = get_fuse_conn_super(sb); - fi->dax = NULL; - if (fc->dax) { - fi->dax = kzalloc_obj(*fi->dax, GFP_KERNEL_ACCOUNT); - if (!fi->dax) + fi->vdax = NULL; + if (fc->vdax) { + fi->vdax = kzalloc_obj(*fi->vdax, GFP_KERNEL_ACCOUNT); + if (!fi->vdax) return false; - init_rwsem(&fi->dax->sem); - fi->dax->tree = RB_ROOT_CACHED; + init_rwsem(&fi->vdax->sem); + fi->vdax->tree = RB_ROOT_CACHED; } return true; @@ -1299,26 +1299,26 @@ static const struct address_space_operations fuse_dax_file_aops = { static bool fuse_should_enable_dax(struct inode *inode, unsigned int flags) { struct fuse_conn *fc = get_fuse_conn(inode); - enum fuse_dax_mode dax_mode = fc->dax_mode; + enum fuse_vdax_mode dax_mode = fc->vdax_mode; - if (dax_mode == FUSE_DAX_NEVER) + if (dax_mode == FUSE_VDAX_NEVER) return false; /* - * fc->dax may be NULL in 'inode' mode when filesystem device doesn't + * fc->vdax may be NULL in 'inode' mode when filesystem device doesn't * support DAX, in which case it will silently fallback to 'never' mode. */ - if (!fc->dax) + if (!fc->vdax) return false; - if (dax_mode == FUSE_DAX_ALWAYS) + if (dax_mode == FUSE_VDAX_ALWAYS) return true; /* dax_mode is FUSE_DAX_INODE* */ - return fc->inode_dax && (flags & FUSE_ATTR_DAX); + return fc->inode_vdax && (flags & FUSE_ATTR_DAX); } -void fuse_dax_inode_init(struct inode *inode, unsigned int flags) +void fuse_vdax_inode_init(struct inode *inode, unsigned int flags) { if (!fuse_should_enable_dax(inode, flags)) return; @@ -1327,18 +1327,18 @@ void fuse_dax_inode_init(struct inode *inode, unsigned int flags) inode->i_data.a_ops = &fuse_dax_file_aops; } -void fuse_dax_dontcache(struct inode *inode, unsigned int flags) +void fuse_vdax_dontcache(struct inode *inode, unsigned int flags) { struct fuse_conn *fc = get_fuse_conn(inode); - if (fuse_is_inode_dax_mode(fc->dax_mode) && + if (fuse_is_inode_vdax_mode(fc->vdax_mode) && ((bool) IS_DAX(inode) != (bool) (flags & FUSE_ATTR_DAX))) d_mark_dontcache(inode); } -bool fuse_dax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment) +bool fuse_vdax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment) { - if (fc->dax && (map_alignment > FUSE_DAX_SHIFT)) { + if (fc->vdax && (map_alignment > FUSE_DAX_SHIFT)) { pr_warn("FUSE: map_alignment %u incompatible with dax mem range size %u\n", map_alignment, FUSE_DAX_SZ); return false; @@ -1346,12 +1346,12 @@ bool fuse_dax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment) return true; } -void fuse_dax_cancel_work(struct fuse_conn *fc) +void fuse_vdax_cancel_work(struct fuse_conn *fc) { - struct fuse_conn_dax *fcd = fc->dax; + struct fuse_conn_vdax *fcd = fc->vdax; if (fcd) cancel_delayed_work_sync(&fcd->free_work); } -EXPORT_SYMBOL_GPL(fuse_dax_cancel_work); +EXPORT_SYMBOL_GPL(fuse_vdax_cancel_work); diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c index e49b4e874b15f8..f48fafccce4b53 100644 --- a/fs/fuse/dir.c +++ b/fs/fuse/dir.c @@ -2174,10 +2174,10 @@ int fuse_do_setattr(struct mnt_idmap *idmap, struct dentry *dentry, is_truncate = true; } - if (FUSE_IS_DAX(inode) && is_truncate) { + if (FUSE_IS_VDAX(inode) && is_truncate) { filemap_invalidate_lock(mapping); fault_blocked = true; - err = fuse_dax_break_layouts(inode, 0, -1); + err = fuse_vdax_break_layouts(inode, 0, -1); if (err) goto unlock; } diff --git a/fs/fuse/file.c b/fs/fuse/file.c index c5e77e13bfdc30..eeda31cd8afdaf 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -119,7 +119,7 @@ static void fuse_file_put(struct fuse_file *ff, bool sync) * DAX inodes may need to issue a number of synchronous * request for clearing the mappings. */ - if (ra && ra->inode && FUSE_IS_DAX(ra->inode)) + if (ra && ra->inode && FUSE_IS_VDAX(ra->inode)) args->may_block = true; args->end = fuse_release_end; if (fuse_simple_background(ff->fm, args, @@ -256,7 +256,7 @@ static int fuse_open(struct inode *inode, struct file *file) int err; bool is_truncate = (file->f_flags & O_TRUNC) && fc->atomic_o_trunc; bool is_wb_truncate = is_truncate && fc->writeback_cache; - bool dax_truncate = is_truncate && FUSE_IS_DAX(inode); + bool vdax_truncate = is_truncate && FUSE_IS_VDAX(inode); if (fuse_is_bad(inode)) return -EIO; @@ -265,17 +265,17 @@ static int fuse_open(struct inode *inode, struct file *file) if (err) return err; - if (is_wb_truncate || dax_truncate) + if (is_wb_truncate || vdax_truncate) inode_lock(inode); - if (dax_truncate) { + if (vdax_truncate) { filemap_invalidate_lock(inode->i_mapping); - err = fuse_dax_break_layouts(inode, 0, -1); + err = fuse_vdax_break_layouts(inode, 0, -1); if (err) goto out_unlock; } - if (is_wb_truncate || dax_truncate) + if (is_wb_truncate || vdax_truncate) fuse_set_nowrite(inode); err = fuse_do_open(fm, get_node_id(inode), file, false); @@ -288,7 +288,7 @@ static int fuse_open(struct inode *inode, struct file *file) fuse_truncate_update_attr(inode, file); } - if (is_wb_truncate || dax_truncate) + if (is_wb_truncate || vdax_truncate) fuse_release_nowrite(inode); if (!err) { if (is_truncate) @@ -297,9 +297,9 @@ static int fuse_open(struct inode *inode, struct file *file) invalidate_inode_pages2(inode->i_mapping); } out_unlock: - if (dax_truncate) + if (vdax_truncate) filemap_invalidate_unlock(inode->i_mapping); - if (is_wb_truncate || dax_truncate) + if (is_wb_truncate || vdax_truncate) inode_unlock(inode); return err; @@ -1856,8 +1856,8 @@ static ssize_t fuse_file_read_iter(struct kiocb *iocb, struct iov_iter *to) if (fuse_is_bad(inode)) return -EIO; - if (FUSE_IS_DAX(inode)) - return fuse_dax_read_iter(iocb, to); + if (FUSE_IS_VDAX(inode)) + return fuse_vdax_read_iter(iocb, to); /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ if (ff->open_flags & FOPEN_DIRECT_IO) @@ -1877,8 +1877,8 @@ static ssize_t fuse_file_write_iter(struct kiocb *iocb, struct iov_iter *from) if (fuse_is_bad(inode)) return -EIO; - if (FUSE_IS_DAX(inode)) - return fuse_dax_write_iter(iocb, from); + if (FUSE_IS_VDAX(inode)) + return fuse_vdax_write_iter(iocb, from); /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ if (ff->open_flags & FOPEN_DIRECT_IO) @@ -2418,8 +2418,8 @@ static int fuse_file_mmap(struct file *file, struct vm_area_struct *vma) int rc; /* DAX mmap is superior to direct_io mmap */ - if (FUSE_IS_DAX(inode)) - return fuse_dax_mmap(file, vma); + if (FUSE_IS_VDAX(inode)) + return fuse_vdax_mmap(file, vma); /* * If inode is in passthrough io mode, because it has some file open @@ -2868,7 +2868,7 @@ static long fuse_file_fallocate(struct file *file, int mode, loff_t offset, .mode = mode }; int err; - bool block_faults = FUSE_IS_DAX(inode) && + bool block_faults = FUSE_IS_VDAX(inode) && (!(mode & FALLOC_FL_KEEP_SIZE) || (mode & (FALLOC_FL_PUNCH_HOLE | FALLOC_FL_ZERO_RANGE))); @@ -2882,7 +2882,7 @@ static long fuse_file_fallocate(struct file *file, int mode, loff_t offset, inode_lock(inode); if (block_faults) { filemap_invalidate_lock(inode->i_mapping); - err = fuse_dax_break_layouts(inode, 0, -1); + err = fuse_vdax_break_layouts(inode, 0, -1); if (err) goto out; } @@ -3154,10 +3154,10 @@ void fuse_init_file_inode(struct inode *inode, unsigned int flags) init_waitqueue_head(&fi->page_waitq); init_waitqueue_head(&fi->direct_io_waitq); - if (IS_ENABLED(CONFIG_FUSE_DAX)) - fuse_dax_inode_init(inode, flags); + if (IS_ENABLED(CONFIG_FUSE_VDAX)) + fuse_vdax_inode_init(inode, flags); - if (FUSE_IS_DAX(inode)) + if (FUSE_IS_VDAX(inode)) return; /* diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 97d356588d0019..5a2e31b8a7b6b7 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -218,11 +218,11 @@ struct fuse_inode { /** @lock: Lock to protect write-related fields */ spinlock_t lock; -#ifdef CONFIG_FUSE_DAX +#ifdef CONFIG_FUSE_VDAX /** - * @dax: Dax specific inode data + * @vdax: Virtiofs DAX specific inode data */ - struct fuse_inode_dax *dax; + struct fuse_inode_vdax *vdax; #endif /** @submount_lookup: Submount specific lookup tracking */ struct fuse_submount_lookup *submount_lookup; @@ -364,16 +364,16 @@ struct fuse_io_priv { .iocb = i, \ } -enum fuse_dax_mode { - FUSE_DAX_INODE_DEFAULT, /* default */ - FUSE_DAX_ALWAYS, /* "-o dax=always" */ - FUSE_DAX_NEVER, /* "-o dax=never" */ - FUSE_DAX_INODE_USER, /* "-o dax=inode" */ +enum fuse_vdax_mode { + FUSE_VDAX_INODE_DEFAULT, /* default */ + FUSE_VDAX_ALWAYS, /* "-o dax=always" */ + FUSE_VDAX_NEVER, /* "-o dax=never" */ + FUSE_VDAX_INODE_USER, /* "-o dax=inode" */ }; -static inline bool fuse_is_inode_dax_mode(enum fuse_dax_mode mode) +static inline bool fuse_is_inode_vdax_mode(enum fuse_vdax_mode mode) { - return mode == FUSE_DAX_INODE_DEFAULT || mode == FUSE_DAX_INODE_USER; + return mode == FUSE_VDAX_INODE_DEFAULT || mode == FUSE_VDAX_INODE_USER; } struct fuse_fs_context { @@ -392,13 +392,13 @@ struct fuse_fs_context { bool no_force_umount:1; bool legacy_opts_show:1; bool syncfs_capable:1; - enum fuse_dax_mode dax_mode; + enum fuse_vdax_mode vdax_mode; unsigned int max_read; unsigned int blksize; const char *subtype; - /* DAX device, may be NULL */ - struct dax_device *dax_dev; + /* Virtiofs DAX device, may be NULL */ + struct dax_device *vdax_dev; }; struct fuse_sync_bucket { @@ -693,8 +693,8 @@ struct fuse_conn { */ unsigned int create_supp_group:1; - /** @inode_dax: Does the filesystem support per inode DAX? */ - unsigned int inode_dax:1; + /** @inode_vdax: Does the filesystem support per inode virtiofs DAX? */ + unsigned int inode_vdax:1; /** @no_tmpfile: Is tmpfile not implemented by fs? */ unsigned int no_tmpfile:1; @@ -753,12 +753,12 @@ struct fuse_conn { */ struct rw_semaphore killsb; -#ifdef CONFIG_FUSE_DAX - /** @dax_mode: Dax mode */ - enum fuse_dax_mode dax_mode; +#ifdef CONFIG_FUSE_VDAX + /** @vdax_mode: Virtiofs DAX mode */ + enum fuse_vdax_mode vdax_mode; - /** @dax: Dax specific conn data, non-NULL if DAX is enabled */ - struct fuse_conn_dax *dax; + /** @dax: Dax specific conn data, non-NULL if virtiofs DAX is enabled */ + struct fuse_conn_vdax *vdax; #endif /** @mounts: List of filesystems using this connection */ @@ -1226,21 +1226,21 @@ void fuse_free_conn(struct fuse_conn *fc); /* dax.c */ -#define FUSE_IS_DAX(inode) (IS_ENABLED(CONFIG_FUSE_DAX) && IS_DAX(inode)) - -ssize_t fuse_dax_read_iter(struct kiocb *iocb, struct iov_iter *to); -ssize_t fuse_dax_write_iter(struct kiocb *iocb, struct iov_iter *from); -int fuse_dax_mmap(struct file *file, struct vm_area_struct *vma); -int fuse_dax_break_layouts(struct inode *inode, u64 dmap_start, u64 dmap_end); -int fuse_dax_conn_alloc(struct fuse_conn *fc, enum fuse_dax_mode mode, - struct dax_device *dax_dev); -void fuse_dax_conn_free(struct fuse_conn *fc); -bool fuse_dax_inode_alloc(struct super_block *sb, struct fuse_inode *fi); -void fuse_dax_inode_init(struct inode *inode, unsigned int flags); -void fuse_dax_inode_cleanup(struct inode *inode); -void fuse_dax_dontcache(struct inode *inode, unsigned int flags); -bool fuse_dax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment); -void fuse_dax_cancel_work(struct fuse_conn *fc); +#define FUSE_IS_VDAX(inode) (IS_ENABLED(CONFIG_FUSE_VDAX) && IS_DAX(inode)) + +ssize_t fuse_vdax_read_iter(struct kiocb *iocb, struct iov_iter *to); +ssize_t fuse_vdax_write_iter(struct kiocb *iocb, struct iov_iter *from); +int fuse_vdax_mmap(struct file *file, struct vm_area_struct *vma); +int fuse_vdax_break_layouts(struct inode *inode, u64 dmap_start, u64 dmap_end); +int fuse_vdax_conn_alloc(struct fuse_conn *fc, enum fuse_vdax_mode mode, + struct dax_device *vdax_dev); +void fuse_vdax_conn_free(struct fuse_conn *fc); +bool fuse_vdax_inode_alloc(struct super_block *sb, struct fuse_inode *fi); +void fuse_vdax_inode_init(struct inode *inode, unsigned int flags); +void fuse_vdax_inode_cleanup(struct inode *inode); +void fuse_vdax_dontcache(struct inode *inode, unsigned int flags); +bool fuse_vdax_check_alignment(struct fuse_conn *fc, unsigned int map_alignment); +void fuse_vdax_cancel_work(struct fuse_conn *fc); /* ioctl.c */ long fuse_file_ioctl(struct file *file, unsigned int cmd, unsigned long arg); diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c index c93d744b661e36..cbb10e19e7e862 100644 --- a/fs/fuse/inode.c +++ b/fs/fuse/inode.c @@ -100,7 +100,7 @@ static struct inode *fuse_alloc_inode(struct super_block *sb) if (!fi->forget) goto out_free; - if (IS_ENABLED(CONFIG_FUSE_DAX) && !fuse_dax_inode_alloc(sb, fi)) + if (IS_ENABLED(CONFIG_FUSE_VDAX) && !fuse_vdax_inode_alloc(sb, fi)) goto out_free_forget; if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH)) @@ -121,8 +121,8 @@ static void fuse_free_inode(struct inode *inode) mutex_destroy(&fi->mutex); kfree(fi->forget); -#ifdef CONFIG_FUSE_DAX - kfree(fi->dax); +#ifdef CONFIG_FUSE_VDAX + kfree(fi->vdax); #endif if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH)) fuse_backing_put(fuse_inode_backing(fi)); @@ -148,7 +148,7 @@ static void fuse_evict_inode(struct inode *inode) /* Will write inode on close/munmap and in all other dirtiers */ WARN_ON(inode_state_read_once(inode) & I_DIRTY_INODE); - if (FUSE_IS_DAX(inode)) + if (FUSE_IS_VDAX(inode)) dax_break_layout_final(inode); truncate_inode_pages_final(&inode->i_data); @@ -156,8 +156,8 @@ static void fuse_evict_inode(struct inode *inode) if (inode->i_sb->s_flags & SB_ACTIVE) { struct fuse_conn *fc = get_fuse_conn(inode); - if (FUSE_IS_DAX(inode)) - fuse_dax_inode_cleanup(inode); + if (FUSE_IS_VDAX(inode)) + fuse_vdax_inode_cleanup(inode); if (fi->nlookup) { fuse_chan_queue_forget(fc->chan, fi->forget, fi->nodeid, fi->nlookup); @@ -385,8 +385,8 @@ static void fuse_change_attributes_i(struct inode *inode, struct fuse_attr *attr invalidate_inode_pages2(inode->i_mapping); } - if (IS_ENABLED(CONFIG_FUSE_DAX)) - fuse_dax_dontcache(inode, attr->flags); + if (IS_ENABLED(CONFIG_FUSE_VDAX)) + fuse_vdax_dontcache(inode, attr->flags); } void fuse_change_attributes(struct inode *inode, struct fuse_attr *attr, @@ -953,12 +953,12 @@ static int fuse_show_options(struct seq_file *m, struct dentry *root) if (sb->s_bdev && sb->s_blocksize != FUSE_DEFAULT_BLKSIZE) seq_printf(m, ",blksize=%lu", sb->s_blocksize); } -#ifdef CONFIG_FUSE_DAX - if (fc->dax_mode == FUSE_DAX_ALWAYS) +#ifdef CONFIG_FUSE_VDAX + if (fc->vdax_mode == FUSE_VDAX_ALWAYS) seq_puts(m, ",dax=always"); - else if (fc->dax_mode == FUSE_DAX_NEVER) + else if (fc->vdax_mode == FUSE_VDAX_NEVER) seq_puts(m, ",dax=never"); - else if (fc->dax_mode == FUSE_DAX_INODE_USER) + else if (fc->vdax_mode == FUSE_VDAX_INODE_USER) seq_puts(m, ",dax=inode"); #endif @@ -1016,8 +1016,8 @@ void fuse_conn_put(struct fuse_conn *fc) if (!refcount_dec_and_test(&fc->count)) return; - if (IS_ENABLED(CONFIG_FUSE_DAX)) - fuse_dax_conn_free(fc); + if (IS_ENABLED(CONFIG_FUSE_VDAX)) + fuse_vdax_conn_free(fc); cancel_work_sync(&fc->epoch_work); fuse_chan_release(fc->chan); put_pid_ns(fc->pid_ns); @@ -1373,13 +1373,13 @@ static void process_init_reply(struct fuse_args *args, int error) if (fc->max_pages > 1) fc->name_max = FUSE_NAME_MAX; } - if (IS_ENABLED(CONFIG_FUSE_DAX)) { + if (IS_ENABLED(CONFIG_FUSE_VDAX)) { if (flags & FUSE_MAP_ALIGNMENT && - !fuse_dax_check_alignment(fc, arg->map_alignment)) { + !fuse_vdax_check_alignment(fc, arg->map_alignment)) { ok = false; } if (flags & FUSE_HAS_INODE_DAX) - fc->inode_dax = 1; + fc->inode_vdax = 1; } if (flags & FUSE_HANDLE_KILLPRIV_V2) { fc->handle_killpriv_v2 = 1; @@ -1491,10 +1491,10 @@ static struct fuse_init_args *fuse_new_init(struct fuse_mount *fm) FUSE_HAS_EXPIRE_ONLY | FUSE_DIRECT_IO_ALLOW_MMAP | FUSE_NO_EXPORT_SUPPORT | FUSE_HAS_RESEND | FUSE_ALLOW_IDMAP | FUSE_REQUEST_TIMEOUT; -#ifdef CONFIG_FUSE_DAX - if (fm->fc->dax) +#ifdef CONFIG_FUSE_VDAX + if (fm->fc->vdax) flags |= FUSE_MAP_ALIGNMENT; - if (fuse_is_inode_dax_mode(fm->fc->dax_mode)) + if (fuse_is_inode_vdax_mode(fm->fc->vdax_mode)) flags |= FUSE_HAS_INODE_DAX; #endif if (fm->fc->auto_submounts) @@ -1778,8 +1778,8 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) sb->s_subtype = ctx->subtype; ctx->subtype = NULL; - if (IS_ENABLED(CONFIG_FUSE_DAX)) { - err = fuse_dax_conn_alloc(fc, ctx->dax_mode, ctx->dax_dev); + if (IS_ENABLED(CONFIG_FUSE_VDAX)) { + err = fuse_vdax_conn_alloc(fc, ctx->vdax_mode, ctx->vdax_dev); if (err) goto err; } @@ -1788,7 +1788,7 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) fm->sb = sb; err = fuse_bdi_init(fc, sb); if (err) - goto err_free_dax; + goto err_free_vdax; /* Handle umasking inside the fuse code */ if (sb->s_flags & SB_POSIXACL) @@ -1811,7 +1811,7 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) set_default_d_op(sb, &fuse_dentry_operations); root_dentry = d_make_root(root); if (!root_dentry) - goto err_free_dax; + goto err_free_vdax; mutex_lock(&fuse_mutex); err = -EINVAL; @@ -1837,9 +1837,9 @@ int fuse_fill_super_common(struct super_block *sb, struct fuse_fs_context *ctx) err_unlock: mutex_unlock(&fuse_mutex); dput(root_dentry); - err_free_dax: - if (IS_ENABLED(CONFIG_FUSE_DAX)) - fuse_dax_conn_free(fc); + err_free_vdax: + if (IS_ENABLED(CONFIG_FUSE_VDAX)) + fuse_vdax_conn_free(fc); err: return err; } diff --git a/fs/fuse/iomode.c b/fs/fuse/iomode.c index 3728933188f307..79637c09e88397 100644 --- a/fs/fuse/iomode.c +++ b/fs/fuse/iomode.c @@ -200,10 +200,10 @@ int fuse_file_io_open(struct file *file, struct inode *inode) int err; /* - * io modes are not relevant with DAX and with server that does not + * io modes are not relevant with virtiofs DAX and with server that does not * implement open. */ - if (FUSE_IS_DAX(inode) || !ff->args) + if (FUSE_IS_VDAX(inode) || !ff->args) return 0; /* diff --git a/fs/fuse/virtio_fs.c b/fs/fuse/virtio_fs.c index 61ab7cb302c189..4f334766b8c307 100644 --- a/fs/fuse/virtio_fs.c +++ b/fs/fuse/virtio_fs.c @@ -102,9 +102,9 @@ static int virtio_fs_enqueue_req(struct virtio_fs_vq *fsvq, gfp_t gfp); static const struct constant_table dax_param_enums[] = { - {"always", FUSE_DAX_ALWAYS }, - {"never", FUSE_DAX_NEVER }, - {"inode", FUSE_DAX_INODE_USER }, + {"always", FUSE_VDAX_ALWAYS }, + {"never", FUSE_VDAX_NEVER }, + {"inode", FUSE_VDAX_INODE_USER }, {} }; @@ -132,10 +132,10 @@ static int virtio_fs_parse_param(struct fs_context *fsc, switch (opt) { case OPT_DAX: - ctx->dax_mode = FUSE_DAX_ALWAYS; + ctx->vdax_mode = FUSE_VDAX_ALWAYS; break; case OPT_DAX_ENUM: - ctx->dax_mode = result.uint_32; + ctx->vdax_mode = result.uint_32; break; default: return -EINVAL; @@ -1081,7 +1081,7 @@ static int virtio_fs_setup_dax(struct virtio_device *vdev, struct virtio_fs *fs) struct dev_pagemap *pgmap; bool have_cache; - if (!IS_ENABLED(CONFIG_FUSE_DAX)) + if (!IS_ENABLED(CONFIG_FUSE_VDAX)) return 0; dax_dev = alloc_dax(fs, &virtio_fs_dax_ops); @@ -1594,14 +1594,14 @@ static int virtio_fs_fill_super(struct super_block *sb, struct fs_context *fsc) goto err_free_fuse_devs; } - if (ctx->dax_mode != FUSE_DAX_NEVER) { - if (ctx->dax_mode == FUSE_DAX_ALWAYS && !fs->dax_dev) { + if (ctx->vdax_mode != FUSE_VDAX_NEVER) { + if (ctx->vdax_mode == FUSE_VDAX_ALWAYS && !fs->dax_dev) { err = -EINVAL; pr_err("virtio-fs: dax can't be enabled as filesystem" " device does not support it.\n"); goto err_free_fuse_devs; } - ctx->dax_dev = fs->dax_dev; + ctx->vdax_dev = fs->dax_dev; } err = fuse_fill_super_common(sb, ctx); if (err < 0) @@ -1635,8 +1635,8 @@ static void virtio_fs_conn_destroy(struct fuse_mount *fm) /* Stop dax worker. Soon evict_inodes() will be called which * will free all memory ranges belonging to all inodes. */ - if (IS_ENABLED(CONFIG_FUSE_DAX)) - fuse_dax_cancel_work(fc); + if (IS_ENABLED(CONFIG_FUSE_VDAX)) + fuse_vdax_cancel_work(fc); /* Stop forget queue. Soon destroy will be sent */ spin_lock(&fsvq->lock); From af5891a4fb9ae9f78e695cf4a1809f363f76c6e0 Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Mon, 21 Sep 2026 14:05:45 +0200 Subject: [PATCH 0151/1012] fuse: don't assume ff->passthrough is set for FOPEN_PASSTHROUGH Following patch will introduce extent map passthrough mode, where ff->passthrough is not set. Check if in passthrough mode via FOPEN_PASSTHROUGH instead. No functional change. Reviewed-by: Amir Goldstein Signed-off-by: Miklos Szeredi --- fs/fuse/file.c | 12 ++++++------ fs/fuse/fuse_i.h | 8 ++------ fs/fuse/passthrough.c | 13 +++++++++++-- 3 files changed, 19 insertions(+), 14 deletions(-) diff --git a/fs/fuse/file.c b/fs/fuse/file.c index eeda31cd8afdaf..6d707f2b3bff80 100644 --- a/fs/fuse/file.c +++ b/fs/fuse/file.c @@ -311,7 +311,7 @@ static void fuse_prepare_release(struct fuse_inode *fi, struct fuse_file *ff, struct fuse_conn *fc = ff->fm->fc; struct fuse_release_args *ra = &ff->args->release_args; - if (fuse_file_passthrough(ff)) + if (fuse_is_passthrough(ff)) fuse_passthrough_release(ff, fuse_inode_backing(fi)); /* Inode is NULL on error path of fuse_create_open() */ @@ -1862,7 +1862,7 @@ static ssize_t fuse_file_read_iter(struct kiocb *iocb, struct iov_iter *to) /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ if (ff->open_flags & FOPEN_DIRECT_IO) return fuse_direct_read_iter(iocb, to); - else if (fuse_file_passthrough(ff)) + else if (fuse_is_passthrough(ff)) return fuse_passthrough_read_iter(iocb, to); else return fuse_cache_read_iter(iocb, to); @@ -1883,7 +1883,7 @@ static ssize_t fuse_file_write_iter(struct kiocb *iocb, struct iov_iter *from) /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ if (ff->open_flags & FOPEN_DIRECT_IO) return fuse_direct_write_iter(iocb, from); - else if (fuse_file_passthrough(ff)) + else if (fuse_is_passthrough(ff)) return fuse_passthrough_write_iter(iocb, from); else return fuse_cache_write_iter(iocb, from); @@ -1899,7 +1899,7 @@ static ssize_t fuse_splice_read(struct file *in, loff_t *ppos, if (ff->open_flags & FOPEN_DIRECT_IO) return copy_splice_read(in, ppos, pipe, len, flags); - else if (fuse_file_passthrough(ff)) + else if (fuse_is_passthrough(ff)) return fuse_passthrough_splice_read(in, ppos, pipe, len, flags); else return filemap_splice_read(in, ppos, pipe, len, flags); @@ -1911,7 +1911,7 @@ static ssize_t fuse_splice_write(struct pipe_inode_info *pipe, struct file *out, struct fuse_file *ff = out->private_data; /* FOPEN_DIRECT_IO overrides FOPEN_PASSTHROUGH */ - if (fuse_file_passthrough(ff) && !(ff->open_flags & FOPEN_DIRECT_IO)) + if (fuse_is_passthrough(ff) && !(ff->open_flags & FOPEN_DIRECT_IO)) return fuse_passthrough_splice_write(pipe, out, ppos, len, flags); else return iter_file_splice_write(pipe, out, ppos, len, flags); @@ -2426,7 +2426,7 @@ static int fuse_file_mmap(struct file *file, struct vm_area_struct *vma) * in passthrough mode, either mmap to backing file or fail mmap, * because mixing cached mmap and passthrough io mode is not allowed. */ - if (fuse_file_passthrough(ff)) + if (fuse_is_passthrough(ff)) return fuse_passthrough_mmap(file, vma); else if (fuse_inode_backing(get_fuse_inode(inode))) return -ENODEV; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 5a2e31b8a7b6b7..6af2603b522ad4 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1313,13 +1313,9 @@ static inline struct fuse_backing *fuse_inode_backing_set(struct fuse_inode *fi, struct fuse_backing *fuse_passthrough_open(struct file *file, int backing_id); void fuse_passthrough_release(struct fuse_file *ff, struct fuse_backing *fb); -static inline struct file *fuse_file_passthrough(struct fuse_file *ff) +static inline bool fuse_is_passthrough(struct fuse_file *ff) { -#ifdef CONFIG_FUSE_PASSTHROUGH - return ff->passthrough; -#else - return NULL; -#endif + return IS_ENABLED(CONFIG_FUSE_PASSTHROUGH) && (ff->open_flags & FOPEN_PASSTHROUGH); } ssize_t fuse_passthrough_read_iter(struct kiocb *iocb, struct iov_iter *iter); diff --git a/fs/fuse/passthrough.c b/fs/fuse/passthrough.c index f2d08ac2459b7e..b43d3e0f70817d 100644 --- a/fs/fuse/passthrough.c +++ b/fs/fuse/passthrough.c @@ -11,6 +11,10 @@ #include #include +static inline struct file *fuse_file_passthrough(struct fuse_file *ff) +{ + return ff->passthrough; +} static void fuse_file_accessed(struct file *file) { struct inode *inode = file_inode(file); @@ -187,10 +191,15 @@ struct fuse_backing *fuse_passthrough_open(struct file *file, int backing_id) void fuse_passthrough_release(struct fuse_file *ff, struct fuse_backing *fb) { + struct file *backing_file = fuse_file_passthrough(ff); + pr_debug("%s: fb=0x%p, backing_file=0x%p\n", __func__, - fb, ff->passthrough); + fb, backing_file); + + if (!backing_file) + return; - fput(ff->passthrough); + fput(backing_file); ff->passthrough = NULL; put_cred(ff->cred); ff->cred = NULL; From 29b60c6c561f71784c43d4c5d65565ce46e223c6 Mon Sep 17 00:00:00 2001 From: Miklos Szeredi Date: Mon, 21 Sep 2026 14:13:11 +0200 Subject: [PATCH 0152/1012] fuse: make fuse_backing_get() static Since it's not used outside of backing.c. Also remove the fuse_backing_lookup() stub from fuse_i.h, since it's called only from passthrough.c. Reviewed-by: Amir Goldstein Signed-off-by: Miklos Szeredi --- fs/fuse/backing.c | 2 +- fs/fuse/fuse_i.h | 11 ----------- 2 files changed, 1 insertion(+), 12 deletions(-) diff --git a/fs/fuse/backing.c b/fs/fuse/backing.c index 472b6afa7dfff1..433fa3098d71fa 100644 --- a/fs/fuse/backing.c +++ b/fs/fuse/backing.c @@ -10,7 +10,7 @@ #include -struct fuse_backing *fuse_backing_get(struct fuse_backing *fb) +static struct fuse_backing *fuse_backing_get(struct fuse_backing *fb) { if (fb && refcount_inc_not_zero(&fb->count)) return fb; diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h index 6af2603b522ad4..8546855386b5a7 100644 --- a/fs/fuse/fuse_i.h +++ b/fs/fuse/fuse_i.h @@ -1267,24 +1267,13 @@ void fuse_file_release(struct inode *inode, struct fuse_file *ff, /* backing.c */ #ifdef CONFIG_FUSE_PASSTHROUGH -struct fuse_backing *fuse_backing_get(struct fuse_backing *fb); void fuse_backing_put(struct fuse_backing *fb); struct fuse_backing *fuse_backing_lookup(struct fuse_conn *fc, int backing_id); #else -static inline struct fuse_backing *fuse_backing_get(struct fuse_backing *fb) -{ - return NULL; -} - static inline void fuse_backing_put(struct fuse_backing *fb) { } -static inline struct fuse_backing *fuse_backing_lookup(struct fuse_conn *fc, - int backing_id) -{ - return NULL; -} #endif void fuse_backing_files_init(struct fuse_conn *fc); From 9b584c92f94f3a3d75831906b5e2f51344a17b84 Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Sat, 11 Jul 2026 22:59:38 +0200 Subject: [PATCH 0153/1012] arm64: dts: st: Add support for DH electronics STM32MP23xx/STM32MP25xx DHCOS SoM and Breakout Board and DHSBC This stm32mp25xx-dhcos-bb board is a stack of DHCOS SoM based on STM32MP25xx SoC (1200MHz / crypto capabilities) populated on SoM Breakout Board, the stm32mp255c-dhcos-dhsbc is the SoM populated on DHSBC carrier board. The stm32mp23xx-dhcos-bb is a stack with STM32MP23xx SoC. The SoM contains the following peripherals: - STPMIC (power delivery) - 4GiB LPDDR4 memory - eMMC and SDIO WiFi module The Breakout Board carrier board contains the following peripherals: - USB-C peripheral port, power supply plug The DHSBC carrier board contains the following peripherals: - Two RGMII Ethernet ports - MicroSD slot - LVDS connector - MIPI CSI2 connector - USB-A Host port, USB-C power supply plug - USB-C / DP port - Expansion connector Signed-off-by: Marek Vasut Link: https://lore.kernel.org/r/20260711210131.236025-10-marex@nabladev.com Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/Makefile | 10 + arch/arm64/boot/dts/st/stm32mp23xc.dtsi | 7 + .../boot/dts/st/stm32mp23xx-dhcos-bb.dts | 15 + .../boot/dts/st/stm32mp23xx-dhcos-som.dtsi | 51 ++ ...mp255c-dhcos-dhsbc-overlay-imx219-x10.dtso | 111 +++++ .../boot/dts/st/stm32mp255c-dhcos-dhsbc.dts | 189 ++++++++ arch/arm64/boot/dts/st/stm32mp25xc.dtsi | 7 + .../boot/dts/st/stm32mp25xx-dhcos-bb.dts | 15 + .../boot/dts/st/stm32mp25xx-dhcos-som.dtsi | 51 ++ .../boot/dts/st/stm32mp2xxx-dhcos-som.dtsi | 452 ++++++++++++++++++ 10 files changed, 908 insertions(+) create mode 100644 arch/arm64/boot/dts/st/stm32mp23xc.dtsi create mode 100644 arch/arm64/boot/dts/st/stm32mp23xx-dhcos-bb.dts create mode 100644 arch/arm64/boot/dts/st/stm32mp23xx-dhcos-som.dtsi create mode 100644 arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtso create mode 100644 arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc.dts create mode 100644 arch/arm64/boot/dts/st/stm32mp25xc.dtsi create mode 100644 arch/arm64/boot/dts/st/stm32mp25xx-dhcos-bb.dts create mode 100644 arch/arm64/boot/dts/st/stm32mp25xx-dhcos-som.dtsi create mode 100644 arch/arm64/boot/dts/st/stm32mp2xxx-dhcos-som.dtsi diff --git a/arch/arm64/boot/dts/st/Makefile b/arch/arm64/boot/dts/st/Makefile index 6cbccfd50993b4..931a64f87d8882 100644 --- a/arch/arm64/boot/dts/st/Makefile +++ b/arch/arm64/boot/dts/st/Makefile @@ -1,7 +1,17 @@ # SPDX-License-Identifier: GPL-2.0-only + +stm32mp255c-dhcos-dhsbc-overlay-imx219-x10-dtbs := \ + stm32mp255c-dhcos-dhsbc.dtb \ + stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtbo + dtb-$(CONFIG_ARCH_STM32) += \ stm32mp215f-dk.dtb \ stm32mp235f-dk.dtb \ + stm32mp23xx-dhcos-bb.dtb \ + stm32mp25xx-dhcos-bb.dtb \ + stm32mp255c-dhcos-dhsbc.dtb \ + stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtb \ + stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtbo \ stm32mp257d-engicam-microgea-rmm.dtb \ stm32mp257f-dk.dtb \ stm32mp257f-ev1.dtb diff --git a/arch/arm64/boot/dts/st/stm32mp23xc.dtsi b/arch/arm64/boot/dts/st/stm32mp23xc.dtsi new file mode 100644 index 00000000000000..56872fb1deeb0f --- /dev/null +++ b/arch/arm64/boot/dts/st/stm32mp23xc.dtsi @@ -0,0 +1,7 @@ +// SPDX-License-Identifier: (GPL-2.0-or-later OR BSD-3-Clause) +/* + * Copyright (C) 2026 Marek Vasut + */ + +/ { +}; diff --git a/arch/arm64/boot/dts/st/stm32mp23xx-dhcos-bb.dts b/arch/arm64/boot/dts/st/stm32mp23xx-dhcos-bb.dts new file mode 100644 index 00000000000000..125c76fe3e7bed --- /dev/null +++ b/arch/arm64/boot/dts/st/stm32mp23xx-dhcos-bb.dts @@ -0,0 +1,15 @@ +// SPDX-License-Identifier: (GPL-2.0-or-later OR BSD-3-Clause) +/* + * Copyright (C) 2026 Marek Vasut + */ + +/dts-v1/; + +#include "stm32mp235.dtsi" +#include "stm32mp23xc.dtsi" +#include "stm32mp23xx-dhcos-som.dtsi" + +/ { + model = "DH electronics STM32MP23xx DHCOS Breakout Board"; + compatible = "dh,stm32mp231a-dhcos-bb", "dh,stm32mp231a-dhcos-som", "st,stm32mp231"; +}; diff --git a/arch/arm64/boot/dts/st/stm32mp23xx-dhcos-som.dtsi b/arch/arm64/boot/dts/st/stm32mp23xx-dhcos-som.dtsi new file mode 100644 index 00000000000000..ffdcceb2fa237b --- /dev/null +++ b/arch/arm64/boot/dts/st/stm32mp23xx-dhcos-som.dtsi @@ -0,0 +1,51 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-3-Clause) +/* + * Copyright (C) 2025-2026 Marek Vasut + */ + +#include +#include +#include +#include "stm32mp25-pinctrl.dtsi" +#include "stm32mp25xxak-pinctrl.dtsi" +#include "stm32mp2xxx-dhcos-som.dtsi" + +/ { + model = "DH electronics STM32MP23xx DHCOS SoM"; + compatible = "dh,stm32mp231a-dhcos-som", "st,stm32mp231"; + + aliases { + serial1 = &usart1; + serial2 = &usart2; + }; +}; + +&rv3032 { + interrupts-extended = <&gpiod 10 IRQ_TYPE_EDGE_FALLING>; +}; + +/* Bluetooth */ +&usart1 { + pinctrl-names = "default", "sleep", "idle"; + pinctrl-0 = <&usart1_pins_a>; + pinctrl-1 = <&usart1_sleep_pins_a>; + pinctrl-2 = <&usart1_idle_pins_a>; + uart-has-rtscts; + status = "okay"; + + bluetooth { + compatible = "infineon,cyw55572-bt"; + brcm,requires-autobaud-mode; + max-speed = <3000000>; + shutdown-gpios = <&ioexp 2 GPIO_ACTIVE_HIGH>; + }; +}; + +&usart2 { + pinctrl-names = "default", "sleep", "idle"; + pinctrl-0 = <&usart2_pins_b>; + pinctrl-1 = <&usart2_sleep_pins_b>; + pinctrl-2 = <&usart2_idle_pins_b>; + uart-has-rtscts; + status = "okay"; +}; diff --git a/arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtso b/arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtso new file mode 100644 index 00000000000000..fbec84eb7720f1 --- /dev/null +++ b/arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtso @@ -0,0 +1,111 @@ +// SPDX-License-Identifier: (GPL-2.0-or-later OR BSD-3-Clause) +/* + * Copyright (C) 2026 Marek Vasut + * + * Set up pipeline for raw bayer capture: + * $ media-ctl -d platform:48030000.dcmipp -r + * $ media-ctl -d platform:48030000.dcmipp -l '"48020000.csi":1->"dcmipp_input":0[1]' + * $ media-ctl -d platform:48030000.dcmipp -l "'dcmipp_input':1->'dcmipp_dump_postproc':0[1]" + * $ media-ctl -d platform:48030000.dcmipp --set-v4l2 "'imx219 0-0010':0[fmt:SRGGB8_1X8/1920x1080]" + * $ media-ctl -d platform:48030000.dcmipp --set-v4l2 "'48020000.csi':1[fmt:SRGGB8_1X8/1920x1080]" + * $ media-ctl -d platform:48030000.dcmipp --set-v4l2 "'dcmipp_input':1[fmt:SRGGB8_1X8/1920x1080 field:none]" + * $ media-ctl -d platform:48030000.dcmipp --set-v4l2 "'dcmipp_dump_postproc':0[compose:(0,0)/1920x1080]" + * $ media-ctl -d platform:48030000.dcmipp --set-v4l2 "'dcmipp_dump_postproc':1[fmt:SRGGB8_1X8/1920x1080]" + * $ v4l2-ctl -d /dev/video0 --set-fmt-video=width=1920,height=1080,pixelformat=RGGB + * + * Capture frame using v4l2-ctl: + * $ v4l2-ctl -d /dev/video0 --stream-mmap --stream-count=1 --stream-to=/frame.raw + * + * Capture frame using gstreamer: + * $ gst-launch-1.0 v4l2src device=/dev/video0 num-buffers=1 ! \ + * video/x-bayer,width=1920,height=1080,format=rggb ! \ + * bayer2rgb ! jpegenc ! filesink location=/test.jpg + */ + +/dts-v1/; +/plugin/; + +#include + +&{/} { + clk_cam_x10: clk-cam-j1 { + compatible = "fixed-clock"; + #clock-cells = <0>; + clock-frequency = <24000000>; + }; + + /* Page 29 / CSI_IF_CN / J1 */ + reg_cam_x10: reg-cam-j1 { + compatible = "regulator-fixed"; + regulator-name = "cam-X10"; + enable-active-high; + gpios = <&ioexp 13 GPIO_ACTIVE_HIGH>; + }; +}; + +&csi { + vdd-supply = <&scmi_vddcore>; + vdda18-supply = <&scmi_v1v8>; + status = "okay"; + + ports { + #address-cells = <1>; + #size-cells = <0>; + + port@0 { + reg = <0>; + + csi_sink: endpoint { + remote-endpoint = <&imx219_x10_out>; + data-lanes = <1 2>; + bus-type = <4>; + }; + }; + port@1 { + reg = <1>; + + csi_source: endpoint { + remote-endpoint = <&dcmipp_0>; + }; + }; + }; +}; + +&dcmipp { + status = "okay"; + + port { + dcmipp_0: endpoint { + remote-endpoint = <&csi_source>; + bus-type = <4>; + }; + }; +}; + +&i2c2 { + #address-cells = <1>; + #size-cells = <0>; + + cam@10 { + compatible = "sony,imx219"; + reg = <0x10>; + + clocks = <&clk_cam_x10>; + + VANA-supply = <®_cam_x10>; + VDIG-supply = <®_cam_x10>; + VDDL-supply = <®_cam_x10>; + + orientation = <2>; + rotation = <0>; + + port { + imx219_x10_out: endpoint { + clock-noncontinuous; + link-frequencies = /bits/ 64 <456000000>; + data-lanes = <1 2>; + remote-endpoint = <&csi_sink>; + }; + }; + }; +}; diff --git a/arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc.dts b/arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc.dts new file mode 100644 index 00000000000000..332a8034cc3724 --- /dev/null +++ b/arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc.dts @@ -0,0 +1,189 @@ +// SPDX-License-Identifier: (GPL-2.0-or-later OR BSD-3-Clause) +/* + * Copyright (C) 2025-2026 Marek Vasut + */ + +/dts-v1/; + +#include "stm32mp255.dtsi" +#include "stm32mp25xc.dtsi" +#include "stm32mp25xx-dhcos-som.dtsi" + +/ { + model = "DH electronics STM32MP255C DHCOS DHSBC"; + compatible = "dh,stm32mp255c-dhcos-dhsbc", "dh,stm32mp255c-dhcos-som", "st,stm32mp255"; + + aliases { + ethernet0 = ðernet1; + ethernet1 = ðernet2; + }; +}; + +ðernet1 { + phy-handle = <ðphy1>; + phy-mode = "rgmii-id"; + pinctrl-0 = <ð1_mdio_pins_a ð1_rgmii_pins_a>; + pinctrl-1 = <ð1_mdio_sleep_pins_a ð1_rgmii_sleep_pins_a>; + pinctrl-names = "default", "sleep"; + st,ext-phyclk; + status = "okay"; + + mdio { + #address-cells = <1>; + #size-cells = <0>; + compatible = "snps,dwmac-mdio"; + + ethphy1: ethernet-phy@1 { + /* RTL8211F */ + compatible = "ethernet-phy-id001c.c916"; + interrupt-parent = <&gpioc>; + interrupts = <6 IRQ_TYPE_LEVEL_LOW>; + reg = <1>; + realtek,clkout-disable; + realtek,clkout-ssc-enable; + realtek,rxc-ssc-enable; + realtek,sysclk-ssc-enable; + reset-assert-us = <15000>; + reset-deassert-us = <55000>; + reset-gpios = <&ioexp 1 GPIO_ACTIVE_LOW>; + + leds { + #address-cells = <1>; + #size-cells = <0>; + + led@0 { + reg = <0>; + color = ; + function = LED_FUNCTION_WAN; + linux,default-trigger = "netdev"; + }; + + led@1 { + reg = <1>; + color = ; + function = LED_FUNCTION_WAN; + linux,default-trigger = "netdev"; + }; + }; + }; + }; +}; + +ðernet2 { + phy-handle = <ðphy2>; + phy-mode = "rgmii-id"; + pinctrl-0 = <ð2_mdio_pins_a ð2_rgmii_pins_b>; + pinctrl-1 = <ð2_mdio_sleep_pins_a ð2_rgmii_sleep_pins_b>; + pinctrl-names = "default", "sleep"; + st,ext-phyclk; + status = "okay"; + + mdio { + #address-cells = <1>; + #size-cells = <0>; + compatible = "snps,dwmac-mdio"; + + ethphy2: ethernet-phy@1 { + /* RTL8211F */ + compatible = "ethernet-phy-id001c.c916"; + interrupt-parent = <&gpiog>; + interrupts = <3 IRQ_TYPE_LEVEL_LOW>; + reg = <1>; + realtek,clkout-disable; + realtek,clkout-ssc-enable; + realtek,rxc-ssc-enable; + realtek,sysclk-ssc-enable; + reset-assert-us = <15000>; + reset-deassert-us = <55000>; + reset-gpios = <&ioexp 0 GPIO_ACTIVE_LOW>; + + leds { + #address-cells = <1>; + #size-cells = <0>; + + led@0 { + reg = <0>; + color = ; + function = LED_FUNCTION_LAN; + linux,default-trigger = "netdev"; + }; + + led@1 { + reg = <1>; + color = ; + function = LED_FUNCTION_LAN; + linux,default-trigger = "netdev"; + }; + }; + }; + }; +}; + +&gpioa { + gpio-line-names = "DHSBC_HW-CODE_0", "DHSBC_HW-CODE_1", "DHSBC_HW-CODE_2", "", + "DHCOS-E", "DHCOS-J", "", "", + "DHCOS-D", "", "", "", + "", "", "", ""; +}; + +&i2c2 { + #address-cells = <1>; + #size-cells = <0>; + pinctrl-names = "default", "sleep"; + pinctrl-0 = <&i2c2_pins_a>; + pinctrl-1 = <&i2c2_sleep_pins_a>; + i2c-scl-rising-time-ns = <57>; + i2c-scl-falling-time-ns = <7>; + clock-frequency = <400000>; + status = "okay"; +}; + +&scmi_vddio1 { + regulator-min-microvolt = <1800000>; + regulator-max-microvolt = <3300000>; + regulator-settling-time-up-us = <1000>; + regulator-settling-time-down-us = <100000>; +}; + +&sdmmc1 { + pinctrl-names = "default", "opendrain", "sleep"; + pinctrl-0 = <&sdmmc1_b4_pins_b>; + pinctrl-1 = <&sdmmc1_b4_od_pins_b>; + pinctrl-2 = <&sdmmc1_b4_sleep_pins_a>; + cd-gpios = <&ioexp 8 GPIO_ACTIVE_HIGH>; + disable-wp; + st,neg-edge; + bus-width = <4>; + vmmc-supply = <&scmi_v3v3>; + vqmmc-supply = <&scmi_vddio1>; + sd-uhs-sdr12; + sd-uhs-sdr25; + sd-uhs-sdr50; + sd-uhs-ddr50; + sd-uhs-sdr104; + status = "okay"; +}; + +&spi1 { + pinctrl-names = "default", "sleep"; + pinctrl-0 = <&spi1_pins_b>; + pinctrl-1 = <&spi1_sleep_pins_b>; + cs-gpios = <&gpioh 3 0>; + status = "okay"; + + st33htph: tpm@0 { + compatible = "st,st33htpm-spi", "tcg,tpm_tis-spi"; + reg = <0>; + interrupt-parent = <&gpioa>; + interrupts = <5 IRQ_TYPE_LEVEL_LOW>; + reset-gpios = <&gpioh 2 GPIO_ACTIVE_LOW>; + spi-max-frequency = <24000000>; + }; +}; + +&spi8 { + pinctrl-names = "default", "sleep"; + pinctrl-0 = <&spi8_pins_b>; + pinctrl-1 = <&spi8_sleep_pins_b>; + cs-gpios = <&gpioz 6 0>; +}; diff --git a/arch/arm64/boot/dts/st/stm32mp25xc.dtsi b/arch/arm64/boot/dts/st/stm32mp25xc.dtsi new file mode 100644 index 00000000000000..56872fb1deeb0f --- /dev/null +++ b/arch/arm64/boot/dts/st/stm32mp25xc.dtsi @@ -0,0 +1,7 @@ +// SPDX-License-Identifier: (GPL-2.0-or-later OR BSD-3-Clause) +/* + * Copyright (C) 2026 Marek Vasut + */ + +/ { +}; diff --git a/arch/arm64/boot/dts/st/stm32mp25xx-dhcos-bb.dts b/arch/arm64/boot/dts/st/stm32mp25xx-dhcos-bb.dts new file mode 100644 index 00000000000000..cf66e8e48c99a3 --- /dev/null +++ b/arch/arm64/boot/dts/st/stm32mp25xx-dhcos-bb.dts @@ -0,0 +1,15 @@ +// SPDX-License-Identifier: (GPL-2.0-or-later OR BSD-3-Clause) +/* + * Copyright (C) 2025-2026 Marek Vasut + */ + +/dts-v1/; + +#include "stm32mp255.dtsi" +#include "stm32mp25xc.dtsi" +#include "stm32mp25xx-dhcos-som.dtsi" + +/ { + model = "DH electronics STM32MP25xx DHCOS Breakout Board"; + compatible = "dh,stm32mp251a-dhcos-bb", "dh,stm32mp251a-dhcos-som", "st,stm32mp251"; +}; diff --git a/arch/arm64/boot/dts/st/stm32mp25xx-dhcos-som.dtsi b/arch/arm64/boot/dts/st/stm32mp25xx-dhcos-som.dtsi new file mode 100644 index 00000000000000..23c25ea086445a --- /dev/null +++ b/arch/arm64/boot/dts/st/stm32mp25xx-dhcos-som.dtsi @@ -0,0 +1,51 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-3-Clause) +/* + * Copyright (C) 2025-2026 Marek Vasut + */ + +#include +#include +#include +#include "stm32mp25-pinctrl.dtsi" +#include "stm32mp25xxak-pinctrl.dtsi" +#include "stm32mp2xxx-dhcos-som.dtsi" + +/ { + model = "DH electronics STM32MP25xx DHCOS SoM"; + compatible = "dh,stm32mp251a-dhcos-som", "st,stm32mp251"; + + aliases { + serial1 = &uart8; + serial2 = &uart9; + }; +}; + +&rv3032 { + interrupts-extended = <&gpioi 10 IRQ_TYPE_EDGE_FALLING>; +}; + +/* Bluetooth */ +&uart8 { + pinctrl-names = "default", "sleep", "idle"; + pinctrl-0 = <&uart8_pins_a>; + pinctrl-1 = <&uart8_sleep_pins_a>; + pinctrl-2 = <&uart8_idle_pins_a>; + uart-has-rtscts; + status = "okay"; + + bluetooth { + compatible = "infineon,cyw55572-bt"; + brcm,requires-autobaud-mode; + max-speed = <3000000>; + shutdown-gpios = <&ioexp 2 GPIO_ACTIVE_HIGH>; + }; +}; + +&uart9 { + pinctrl-names = "default", "sleep", "idle"; + pinctrl-0 = <&uart9_pins_a>; + pinctrl-1 = <&uart9_sleep_pins_a>; + pinctrl-2 = <&uart9_idle_pins_a>; + uart-has-rtscts; + status = "okay"; +}; diff --git a/arch/arm64/boot/dts/st/stm32mp2xxx-dhcos-som.dtsi b/arch/arm64/boot/dts/st/stm32mp2xxx-dhcos-som.dtsi new file mode 100644 index 00000000000000..80d829c55d9d87 --- /dev/null +++ b/arch/arm64/boot/dts/st/stm32mp2xxx-dhcos-som.dtsi @@ -0,0 +1,452 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-3-Clause) +/* + * Copyright (C) 2025-2026 Marek Vasut + */ + +/ { + aliases { + mmc0 = &sdmmc2; + mmc1 = &sdmmc3; + serial0 = &usart6; + serial1 = &uart8; + eeprom0 = &eeprom0; + eeprom0wl = &eeprom0wl; + rtc0 = &rv3032; + rtc1 = &rtc; + }; + + chosen { + stdout-path = "serial0:115200n8"; + }; + + memory@80000000 { + device_type = "memory"; + reg = <0x0 0x80000000 0x1 0x0>; + }; + + reserved-memory { + #address-cells = <2>; + #size-cells = <2>; + ranges; + + /* Internal RAM reserved memory declaration */ + tfa_bl31: tfa-bl31@a000000 { + reg = <0x0 0xa000000 0x0 0x20000>; + no-map; + }; + + hpdma1_lli: hpdma1-lli@a020000 { + reg = <0x0 0xa020000 0x0 0xf0f0>; + no-map; + }; + + hpdma2_lli: hpdma2-lli@a02f0f0 { + reg = <0x0 0xa02f0f0 0x0 0xf0f0>; + no-map; + }; + + hpdma3_lli: hpdma3-lli@a03e1e0 { + reg = <0x0 0xa03e1e0 0x0 0x1e20>; + no-map; + }; + + bsec_mirror: bsec-mirror@a040000 { + reg = <0x0 0xa040000 0x0 0x1000>; + no-map; + }; + + scmi_cid2_s: scmi-cid2-s@a041000 { + reg = <0x0 0xa041000 0x0 0x1000>; + no-map; + }; + + scmi_cid2_ns: scmi-cid2-ns@a042000 { + reg = <0x0 0xa042000 0x0 0x1000>; + no-map; + }; + + cm33_sram1: cm33-sram1@a043000 { + reg = <0x0 0xa043000 0x0 0x1d000>; + no-map; + }; + + cm33_sram2: cm33-sram2@a060000 { + reg = <0x0 0xa060000 0x0 0x20000>; + no-map; + }; + + cm33_retram: cm33-retram@a080000 { + reg = <0x0 0xa080000 0x0 0x1f000>; + no-map; + }; + + ddr_param: ddr-param@a09f000 { + reg = <0x0 0xa09f000 0x0 0x1000>; + no-map; + }; + + cm0_cube_fw: cm0-cube-fw@200C0000 { + compatible = "shared-dma-pool"; + reg = <0x0 0x200C0000 0x0 0x4000>; + no-map; + }; + + cm0_cube_data: cm0-cube-data@200C4000 { + compatible = "shared-dma-pool"; + reg = <0x0 0x200C4000 0x0 0x2000>; + no-map; + }; + + ipc_shmem_2: ipc-shmem-2@200C6000{ + compatible = "shared-dma-pool"; + reg = <0x0 0x200C6000 0x0 0x2000>; + no-map; + }; + + /* Backup RAM reserved memory declaration */ + bl31_lowpower: bl31-lowpower@42000000 { + reg = <0x0 0x42000000 0x0 0x1000>; + no-map; + }; + + tfm_its: tfm-its@42001000 { + reg = <0x0 0x42001000 0x0 0x1000>; + no-map; + }; + + /* Octo Memory Manager reserved memory declaration */ + mm_ospi1: mm-ospi@60000000 { + reg = <0x0 0x60000000 0x0 0x10000000>; + no-map; + }; + + /* DDR reserved memory declaration */ + tfm_code: tfm-code@80000000 { + reg = <0x0 0x80000000 0x0 0x100000>; + no-map; + }; + + cm33_cube_fw: cm33-cube-fw@80100000 { + reg = <0x0 0x80100000 0x0 0x800000>; + no-map; + }; + + tfm_data: tfm-data@80900000 { + reg = <0x0 0x80900000 0x0 0x100000>; + no-map; + }; + + cm33_cube_data: cm33-cube-data@80a00000 { + reg = <0x0 0x80a00000 0x0 0x800000>; + no-map; + }; + + ipc_shmem_1: ipc-shmem-1@81200000 { + compatible = "shared-dma-pool"; + reg = <0x0 0x81200000 0x0 0xf8000>; + no-map; + }; + + vdev0vring0: vdev0vring0@812f8000 { + compatible = "shared-dma-pool"; + reg = <0x0 0x812f8000 0x0 0x1000>; + no-map; + }; + + vdev0vring1: vdev0vring1@812f9000 { + compatible = "shared-dma-pool"; + reg = <0x0 0x812f9000 0x0 0x1000>; + no-map; + }; + + vdev0buffer: vdev0buffer@812fa000 { + compatible = "shared-dma-pool"; + reg = <0x0 0x812fa000 0x0 0x6000>; + no-map; + }; + + spare1: spare1@81300000 { + reg = <0x0 0x81300000 0x0 0xcc0000>; + no-map; + }; + + bl31_context: bl31-context@81fc0000 { + reg = <0x0 0x81fc0000 0x0 0x40000>; + no-map; + }; + + op_tee: op-tee@82000000 { + reg = <0x0 0x82000000 0x0 0x2000000>; + no-map; + }; + + gpu_reserved: gpu-reserved@fa800000 { + reg = <0x0 0xfa800000 0x0 0x4000000>; + no-map; + }; + + ltdc_sec_layer: ltdc-sec-layer@fe800000 { + reg = <0x0 0xfe800000 0x0 0x800000>; + no-map; + }; + + ltdc_sec_rotation: ltdc-sec-rotation@ff000000 { + reg = <0x0 0xff000000 0x0 0x1000000>; + no-map; + }; + + /* global autoconfigured region for contiguous allocations */ + linux,cma { + compatible = "shared-dma-pool"; + reusable; + alloc-ranges = <0 0x80000000 0 0x80000000>; + size = <0x0 0x8000000>; + alignment = <0x0 0x2000>; + linux,cma-default; + }; + }; + + sdio_pwrseq: sdio-pwrseq { + compatible = "mmc-pwrseq-simple"; + post-power-on-delay-ms = <50>; + reset-gpios = <&ioexp 3 GPIO_ACTIVE_LOW>; + }; +}; + +&arm_wdt { + timeout-sec = <32>; + status = "okay"; +}; + +&gpioa { + gpio-line-names = "", "DHCOS-M", "DHCOS-K", "DHCOS-L", + "DHCOS-E", "DHCOS-J", "", "", + "DHCOS-D", "", "", "", + "", "", "", ""; +}; + +&gpiof { + gpio-line-names = "", "", "", "", + "", "", "", "", + "", "", "", "", + "", "", "", "DHCOS_RAM-CODE_2"; +}; + +&gpiog { + gpio-line-names = "", "", "", "", + "", "", "", "", + "", "DHCOS_RAM-CODE_0", "DHCOS_RAM-CODE_1", "", + "", "", "DHCOS-F", "DHCOS_HW-CODE_0"; +}; + +&gpioh { + gpio-line-names = "", "", "DHCOS-I", "DHCOS-N", + "", "", "DHCOS-O", "DHCOS-H", + "DHCOS-G", "", "", "", + "", "", "", ""; +}; + +&gpioi { + gpio-line-names = "DHCOS_HW-CODE_1", "DHCOS_HW-CODE_2", "", "", + "", "", "", "", + "", "DHCOS-C", "", "", + "", "", "", ""; +}; + +&gpioz { + gpio-line-names = "", "", "DHCOS-A", "DHCOS-B", + "", "", "", "", + "", "", "", "", + "", "", "", ""; +}; + +&i2c8 { + pinctrl-names = "default", "sleep"; + pinctrl-0 = <&i2c8_pins_b>; + pinctrl-1 = <&i2c8_sleep_pins_b>; + i2c-scl-rising-time-ns = <57>; + i2c-scl-falling-time-ns = <7>; + clock-frequency = <400000>; + status = "okay"; + + ioexp: gpio@20 { + compatible = "kinetic,kts1622", "ti,tcal6416"; + reg = <0x20>; + gpio-controller; + #gpio-cells = <2>; + interrupts-extended = <&gpiod 11 IRQ_TYPE_LEVEL_LOW>; + interrupt-controller; + #interrupt-cells = <2>; + wakeup-source; + + gpio-line-names = + "#ETH1_RST_P0_0", "#ETH0_RST_P0_1", + "BT_REG_ON_P0_2", "WL_REG_ON_P0_3", + "USB1_PWR_EN_P0_4", "USB1_PWR_STAT_P0_5", + "USB0_PWR_EN_P0_6", "USB0_PWR_STAT_P0_7", + "SDIO0_CD_P1_0", "USB0_SS_SEL_P1_1", + "#PCIe_RST_1_2", "#PCIe_Wake_P1_3", + "#QSPI_RST_P1_4", "#CSI0_PWDN_P1_5", + "#CSI0_RST_P1_6", "#ETH3_RST_P1_7"; + }; + + eeprom0: eeprom@50 { + compatible = "atmel,24c256"; /* ST M24256 */ + reg = <0x50>; + pagesize = <64>; + }; + + rv3032: rtc@51 { + compatible = "microcrystal,rv3032"; + reg = <0x51>; + wakeup-source; + }; + + eeprom0wl: eeprom@58 { + compatible = "st,24256e-wl"; /* ST M24256E WL page of 0x50 */ + pagesize = <64>; + reg = <0x58>; + }; +}; + +&ommanager { + memory-region = <&mm_ospi1>; + pinctrl-names = "default", "sleep"; + pinctrl-0 = <&ospi_port1_clk_pins_a + &ospi_port1_io03_pins_a + &ospi_port1_cs0_pins_a>; + pinctrl-1 = <&ospi_port1_clk_sleep_pins_a + &ospi_port1_io03_sleep_pins_a + &ospi_port1_cs0_sleep_pins_a>; + pinctrl-names = "default", "sleep"; + status = "okay"; +}; + +&ospi1 { + #address-cells = <1>; + #size-cells = <0>; + memory-region = <&mm_ospi1>; + status = "okay"; + + flash0: flash@0 { + compatible = "jedec,spi-nor"; + reg = <0>; + spi-rx-bus-width = <4>; + spi-tx-bus-width = <4>; + spi-max-frequency = <100000000>; + }; +}; + +&rtc { + status = "okay"; +}; + +&scmi_regu { + scmi_vddcore: regulator@b { + reg = ; + regulator-name = "vddcore"; + }; + + regulator@c { + reg = ; + regulator-name = "vddgpu"; + regulator-always-on; + }; + + scmi_v1v8: regulator@e { + reg = ; + regulator-name = "v1v8"; + regulator-always-on; + }; + + scmi_v3v3: regulator@10 { + reg = ; + regulator-name = "v3v3"; + regulator-always-on; + }; + + scmi_vdd_emmc: regulator@12 { + reg = ; + regulator-name = "vdd_emmc"; + }; + + scmi_vdd3v3_usb: regulator@14 { + reg = ; + regulator-name = "vdd3v3_usb"; + }; + + scmi_vdd_sdcard: regulator@17 { + reg = ; + regulator-name = "vdd_sdcard"; + }; +}; + +&scmi_vddio3 { + regulator-always-on; +}; + +&scmi_vddio4 { + regulator-always-on; +}; + +&{sdmmc2_b4_pins_a/pins2} { + /delete-property/ bias-disable; + bias-pull-up; +}; + +&{sdmmc2_b4_od_pins_a/pins2} { + /delete-property/ bias-disable; + bias-pull-up; +}; + +&sdmmc2 { + pinctrl-names = "default", "opendrain", "sleep"; + pinctrl-0 = <&sdmmc2_b4_pins_a &sdmmc2_d47_pins_a>; + pinctrl-1 = <&sdmmc2_b4_od_pins_a &sdmmc2_d47_pins_a>; + pinctrl-2 = <&sdmmc2_b4_sleep_pins_a &sdmmc2_d47_sleep_pins_a>; + non-removable; + no-sd; + no-sdio; + st,neg-edge; + bus-width = <8>; + vmmc-supply = <&scmi_vdd_emmc>; + vqmmc-supply = <&scmi_vddio2>; + mmc-ddr-1_8v; + mmc-hs200-1_8v; + status = "okay"; +}; + +&sdmmc3 { + pinctrl-names = "default", "opendrain", "sleep"; + pinctrl-0 = <&sdmmc3_b4_pins_b>; + pinctrl-1 = <&sdmmc3_b4_od_pins_b>; + pinctrl-2 = <&sdmmc3_b4_sleep_pins_b>; + bus-width = <4>; + keep-power-in-suspend; + non-removable; + no-sd; + no-mmc; + st,neg-edge; + vmmc-supply = <&scmi_v3v3>; + vqmmc-supply = <&scmi_v1v8>; + mmc-pwrseq = <&sdio_pwrseq>; + status = "okay"; + + #address-cells = <1>; + #size-cells = <0>; + + brcmf: wifi@1 { /* muRata 2FY */ + reg = <1>; + compatible = "brcm,bcm4329-fmac"; + }; +}; + +&usart6 { + pinctrl-names = "default", "idle", "sleep"; + pinctrl-0 = <&usart6_pins_b>; + pinctrl-1 = <&usart6_idle_pins_b>; + pinctrl-2 = <&usart6_sleep_pins_b>; + /delete-property/dmas; + /delete-property/dma-names; + status = "okay"; +}; From 9558b1a76c7f7142881fe721dc774f0f3c89530f Mon Sep 17 00:00:00 2001 From: Marek Vasut Date: Sat, 11 Jul 2026 22:59:39 +0200 Subject: [PATCH 0154/1012] MAINTAINERS: Add DH electronics DHCOS SoM entry and fix email address Add another SoM type N: match and update email address to an up to date one in the process. Signed-off-by: Marek Vasut Link: https://lore.kernel.org/r/20260711210131.236025-11-marex@nabladev.com Signed-off-by: Alexandre Torgue --- MAINTAINERS | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/MAINTAINERS b/MAINTAINERS index 3a19da74d00c9d..e2aaf437a652bc 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -7557,11 +7557,12 @@ F: drivers/iio/chemical/sen0322.c DH ELECTRONICS DHSOM SOM AND BOARD SUPPORT M: Christoph Niedermaier -M: Marek Vasut +M: Marek Vasut L: kernel@dh-electronics.com S: Maintained N: dhcom N: dhcor +N: dhcos N: dhsom DIALOG SEMICONDUCTOR DRIVERS From 36676c897db593aee7875b67dd06d1e8654b00c4 Mon Sep 17 00:00:00 2001 From: Alain Volmat Date: Mon, 14 Sep 2026 08:54:17 +0200 Subject: [PATCH 0155/1012] arm64: dts: st: add imx335/csi/dcmipp nodes on stm32mp257f-dk Add all nodes necessary for the imx335 camera capture via csi / dcmipp on stm32mp257f discovery board. Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260914-stm32mp2-dk-camera-v1-1-2f2bd719bb63@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/stm32mp257f-dk.dts | 74 +++++++++++++++++++++++ 1 file changed, 74 insertions(+) diff --git a/arch/arm64/boot/dts/st/stm32mp257f-dk.dts b/arch/arm64/boot/dts/st/stm32mp257f-dk.dts index 8daf3dfd513393..39ad23d95069c9 100644 --- a/arch/arm64/boot/dts/st/stm32mp257f-dk.dts +++ b/arch/arm64/boot/dts/st/stm32mp257f-dk.dts @@ -9,6 +9,7 @@ #include #include #include +#include #include "stm32mp257.dtsi" #include "stm32mp25xf.dtsi" #include "stm32mp25-pinctrl.dtsi" @@ -27,6 +28,14 @@ stdout-path = "serial0:115200n8"; }; + clocks { + clk_ext_camera: clk-ext-camera { + #clock-cells = <0>; + compatible = "fixed-clock"; + clock-frequency = <24000000>; + }; + }; + gpio-keys { compatible = "gpio-keys"; @@ -138,6 +147,42 @@ status = "okay"; }; +&csi { + vdd-supply = <&scmi_vddcore>; + vdda18-supply = <&scmi_v1v8>; + status = "okay"; + + ports { + #address-cells = <1>; + #size-cells = <0>; + port@0 { + reg = <0>; + csi_sink: endpoint { + remote-endpoint = <&imx335_ep>; + data-lanes = <1 2>; + bus-type = ; + }; + }; + port@1 { + reg = <1>; + csi_source: endpoint { + remote-endpoint = <&dcmipp_0>; + }; + }; + }; +}; + +&dcmipp { + status = "okay"; + + port { + dcmipp_0: endpoint { + remote-endpoint = <&csi_source>; + bus-type = ; + }; + }; +}; + ðernet1 { pinctrl-0 = <ð1_rgmii_pins_b>; pinctrl-1 = <ð1_rgmii_sleep_pins_b>; @@ -160,6 +205,16 @@ }; }; +&gpiob { + /* Enable the IMX335 power line */ + imx335-en-hog { + gpio-hog; + gpios = <11 GPIO_ACTIVE_HIGH>; + output-high; + line-name = "imx335_en"; + }; +}; + &i2c2 { pinctrl-names = "default", "sleep"; pinctrl-0 = <&i2c2_pins_b>; @@ -172,6 +227,25 @@ /delete-property/dmas; /delete-property/dma-names; + imx335: camera@1a { + compatible = "sony,imx335"; + reg = <0x1a>; + clocks = <&clk_ext_camera>; + avdd-supply = <&scmi_v3v3>; + ovdd-supply = <&scmi_v3v3>; + dvdd-supply = <&scmi_v3v3>; + reset-gpios = <&gpiob 1 (GPIO_ACTIVE_LOW | GPIO_PUSH_PULL)>; + + port { + imx335_ep: endpoint { + remote-endpoint = <&csi_sink>; + clock-lanes = <0>; + data-lanes = <1 2>; + link-frequencies = /bits/ 64 <594000000>; + }; + }; + }; + ili2511: ili2511@41 { compatible = "ilitek,ili251x"; reg = <0x41>; From 2b30170bba8d04c7f508b3966fea0dcb68d1b794 Mon Sep 17 00:00:00 2001 From: Alain Volmat Date: Mon, 14 Sep 2026 08:54:18 +0200 Subject: [PATCH 0156/1012] arm64: dts: st: ensure IMX335 is enabled on stm32mp257f-ev1 Set the GPIO enable line for the IMX335 camera via a GPIO hog in order to ensure it is well enabled. Use scmi_v3v3 for imx335 regulators. The 3 imx335 supplies are generated within the MB1854 and all come from the v3v3 coming from the board. Ensure that this regulator is enabled by using the scmi_v3v3 as supplier of the imx335. Signed-off-by: Hugues Fruchet Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260914-stm32mp2-dk-camera-v1-2-2f2bd719bb63@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/stm32mp257f-ev1.dts | 40 +++++++--------------- 1 file changed, 13 insertions(+), 27 deletions(-) diff --git a/arch/arm64/boot/dts/st/stm32mp257f-ev1.dts b/arch/arm64/boot/dts/st/stm32mp257f-ev1.dts index 12b4018edeb0bb..acc730146287af 100644 --- a/arch/arm64/boot/dts/st/stm32mp257f-ev1.dts +++ b/arch/arm64/boot/dts/st/stm32mp257f-ev1.dts @@ -72,30 +72,6 @@ io-width = <32>; }; - imx335_2v9: regulator-2v9 { - compatible = "regulator-fixed"; - regulator-name = "imx335-avdd"; - regulator-min-microvolt = <2900000>; - regulator-max-microvolt = <2900000>; - regulator-always-on; - }; - - imx335_1v8: regulator-1v8 { - compatible = "regulator-fixed"; - regulator-name = "imx335-ovdd"; - regulator-min-microvolt = <1800000>; - regulator-max-microvolt = <1800000>; - regulator-always-on; - }; - - imx335_1v2: regulator-1v2 { - compatible = "regulator-fixed"; - regulator-name = "imx335-dvdd"; - regulator-min-microvolt = <1200000>; - regulator-max-microvolt = <1200000>; - regulator-always-on; - }; - memory@80000000 { device_type = "memory"; reg = <0x0 0x80000000 0x1 0x0>; @@ -253,6 +229,16 @@ }; }; +&gpioi { + /* Enable the IMX335 power line */ + imx335-en-hog { + gpio-hog; + gpios = <0 GPIO_ACTIVE_HIGH>; + output-high; + line-name = "imx335_en"; + }; +}; + &i2c2 { pinctrl-names = "default", "sleep"; pinctrl-0 = <&i2c2_pins_a>; @@ -269,9 +255,9 @@ compatible = "sony,imx335"; reg = <0x1a>; clocks = <&clk_ext_camera>; - avdd-supply = <&imx335_2v9>; - ovdd-supply = <&imx335_1v8>; - dvdd-supply = <&imx335_1v2>; + avdd-supply = <&scmi_v3v3>; + ovdd-supply = <&scmi_v3v3>; + dvdd-supply = <&scmi_v3v3>; reset-gpios = <&gpioi 7 (GPIO_ACTIVE_LOW | GPIO_PUSH_PULL)>; port { From 41dac3256307093e3f11197f86e5e399ed0aa0b4 Mon Sep 17 00:00:00 2001 From: Alain Volmat Date: Mon, 14 Sep 2026 08:54:19 +0200 Subject: [PATCH 0157/1012] arm64: dts: st: use video-interfaces media bus type in stm32mp257f-ev1 Use MEDIA_BUS_TYPE macro in stm32mp257f-ev1 for csi/dcmipp bus-type. Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260914-stm32mp2-dk-camera-v1-3-2f2bd719bb63@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/stm32mp257f-ev1.dts | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/arch/arm64/boot/dts/st/stm32mp257f-ev1.dts b/arch/arm64/boot/dts/st/stm32mp257f-ev1.dts index acc730146287af..782b4bb843eeb1 100644 --- a/arch/arm64/boot/dts/st/stm32mp257f-ev1.dts +++ b/arch/arm64/boot/dts/st/stm32mp257f-ev1.dts @@ -8,6 +8,7 @@ #include #include +#include #include #include "stm32mp257.dtsi" #include "stm32mp25xf.dtsi" @@ -161,7 +162,7 @@ csi_sink: endpoint { remote-endpoint = <&imx335_ep>; data-lanes = <1 2>; - bus-type = <4>; + bus-type = ; }; }; port@1 { @@ -178,7 +179,7 @@ port { dcmipp_0: endpoint { remote-endpoint = <&csi_source>; - bus-type = <4>; + bus-type = ; }; }; }; From 1cd13a9faf9951e5544bf8f0bd6212b2c9a4c57f Mon Sep 17 00:00:00 2001 From: Alain Volmat Date: Mon, 14 Sep 2026 08:54:20 +0200 Subject: [PATCH 0158/1012] arm64: dts: st: add imx335/csi/dcmipp nodes on stm32mp235f-dk Add all nodes necessary for the imx335 camera capture via csi / dcmipp on stm32mp235f discovery board. Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260914-stm32mp2-dk-camera-v1-4-2f2bd719bb63@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/stm32mp235f-dk.dts | 75 +++++++++++++++++++++++ 1 file changed, 75 insertions(+) diff --git a/arch/arm64/boot/dts/st/stm32mp235f-dk.dts b/arch/arm64/boot/dts/st/stm32mp235f-dk.dts index dd4efbe5a46e86..f06b334dc9ce1d 100644 --- a/arch/arm64/boot/dts/st/stm32mp235f-dk.dts +++ b/arch/arm64/boot/dts/st/stm32mp235f-dk.dts @@ -9,6 +9,7 @@ #include #include #include +#include #include "stm32mp235.dtsi" #include "stm32mp23xf.dtsi" #include "stm32mp25-pinctrl.dtsi" @@ -27,6 +28,14 @@ stdout-path = "serial0:115200n8"; }; + clocks { + clk_ext_camera: clk-ext-camera { + #clock-cells = <0>; + compatible = "fixed-clock"; + clock-frequency = <24000000>; + }; + }; + gpio-keys { compatible = "gpio-keys"; @@ -131,6 +140,42 @@ status = "okay"; }; +&csi { + vdd-supply = <&scmi_vddcore>; + vdda18-supply = <&scmi_v1v8>; + status = "okay"; + + ports { + #address-cells = <1>; + #size-cells = <0>; + port@0 { + reg = <0>; + csi_sink: endpoint { + remote-endpoint = <&imx335_ep>; + data-lanes = <1 2>; + bus-type = ; + }; + }; + port@1 { + reg = <1>; + csi_source: endpoint { + remote-endpoint = <&dcmipp_0>; + }; + }; + }; +}; + +&dcmipp { + status = "okay"; + + port { + dcmipp_0: endpoint { + remote-endpoint = <&csi_source>; + bus-type = ; + }; + }; +}; + ðernet1 { pinctrl-0 = <ð1_rgmii_pins_b>; pinctrl-1 = <ð1_rgmii_sleep_pins_b>; @@ -153,6 +198,16 @@ }; }; +&gpiob { + /* Enable the IMX335 power line */ + imx335-en-hog { + gpio-hog; + gpios = <11 GPIO_ACTIVE_HIGH>; + output-high; + line-name = "imx335_en"; + }; +}; + &i2c2 { pinctrl-names = "default", "sleep"; pinctrl-0 = <&i2c2_pins_b>; @@ -165,6 +220,26 @@ /delete-property/dmas; /delete-property/dma-names; + imx335: camera@1a { + compatible = "sony,imx335"; + reg = <0x1a>; + clocks = <&clk_ext_camera>; + avdd-supply = <&scmi_v3v3>; + ovdd-supply = <&scmi_v3v3>; + dvdd-supply = <&scmi_v3v3>; + reset-gpios = <&gpiob 1 (GPIO_ACTIVE_LOW | GPIO_PUSH_PULL)>; + status = "okay"; + + port { + imx335_ep: endpoint { + remote-endpoint = <&csi_sink>; + clock-lanes = <0>; + data-lanes = <1 2>; + link-frequencies = /bits/ 64 <594000000>; + }; + }; + }; + ili2511: ili2511@41 { compatible = "ilitek,ili251x"; reg = <0x41>; From 61e7fce4274f96edb47363567d27c602569d0077 Mon Sep 17 00:00:00 2001 From: Alain Volmat Date: Mon, 14 Sep 2026 09:09:30 +0200 Subject: [PATCH 0159/1012] arm64: dts: st: add dcmi node on stm32mp25x Add the node for the DCMI controller in stm32mp251.dtsi Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260914-stm32mp2x-dcmi-v1-1-c594e081d21b@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/stm32mp251.dtsi | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/arch/arm64/boot/dts/st/stm32mp251.dtsi b/arch/arm64/boot/dts/st/stm32mp251.dtsi index 978d6102524311..992d77e14e74d1 100644 --- a/arch/arm64/boot/dts/st/stm32mp251.dtsi +++ b/arch/arm64/boot/dts/st/stm32mp251.dtsi @@ -1450,6 +1450,20 @@ status = "disabled"; }; + dcmi: dcmi@404a0000 { + compatible = "st,stm32-dcmi"; + reg = <0x404a0000 0x400>; + interrupts = ; + resets = <&rcc CCI_R>; + clocks = <&rcc CK_BUS_CCI>; + clock-names = "mclk"; + dmas = <&hpdma 137 0x60 0x00003012>; + dma-names = "tx"; + access-controllers = <&rifsc 88>; + power-domains = <&CLUSTER_PD>; + status = "disabled"; + }; + rng: rng@42020000 { compatible = "st,stm32mp25-rng"; reg = <0x42020000 0x400>; From 34bf9ed9b5ac28e26741eb2f83fe4ed1e4d0c74f Mon Sep 17 00:00:00 2001 From: Alain Volmat Date: Mon, 14 Sep 2026 09:09:31 +0200 Subject: [PATCH 0160/1012] arm64: dts: st: add dcmi node on stm32mp23x Add the node for the DCMI controller in stm32mp231.dtsi Signed-off-by: Alain Volmat Link: https://lore.kernel.org/r/20260914-stm32mp2x-dcmi-v1-2-c594e081d21b@foss.st.com Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/stm32mp231.dtsi | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/arch/arm64/boot/dts/st/stm32mp231.dtsi b/arch/arm64/boot/dts/st/stm32mp231.dtsi index 49493f9849705c..f72ee55ea2c3ca 100644 --- a/arch/arm64/boot/dts/st/stm32mp231.dtsi +++ b/arch/arm64/boot/dts/st/stm32mp231.dtsi @@ -700,6 +700,20 @@ status = "disabled"; }; + dcmi: dcmi@404a0000 { + compatible = "st,stm32-dcmi"; + reg = <0x404a0000 0x400>; + interrupts = ; + resets = <&rcc CCI_R>; + clocks = <&rcc CK_BUS_CCI>; + clock-names = "mclk"; + dmas = <&hpdma 137 0x60 0x00003012>; + dma-names = "tx"; + access-controllers = <&rifsc 88>; + power-domains = <&cluster_pd>; + status = "disabled"; + }; + rng: rng@42020000 { compatible = "st,stm32mp25-rng"; reg = <0x42020000 0x400>; From 9a7f28a06dcf8a2519f247dc1c199ce7bf23eed6 Mon Sep 17 00:00:00 2001 From: Mateusz Nowicki Date: Wed, 23 Sep 2026 11:47:39 +0000 Subject: [PATCH 0161/1012] arm64: dts: st: add sdmmc2 pins_b for stm32mp25 Add a second variant of the SDMMC2 default and open-drain pin groups, with the same pins and bias as the _a variant but faster output speeds: very high on CK, high on CMD and D0-D7. These are the OSPEEDR values given in the STM32MP25 datasheet for 120 MHz and 166 MHz operation. On the STM32MP257F-DK, the eMMC returns corrupted data in HS200 mode at 100 MHz with the _a groups, and no CRC error is reported. With the _b groups, four consecutive 1 GiB reads return the same data as a read at 50 MHz. The sleep groups are shared with the _a variant. Signed-off-by: Mateusz Nowicki Link: https://lore.kernel.org/r/20260923114714.140842-2-mateusz.nowicki@posteo.net --- arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi | 58 +++++++++++++++++++ 1 file changed, 58 insertions(+) diff --git a/arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi b/arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi index 1089c36012cab6..757cdd24153947 100644 --- a/arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi +++ b/arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi @@ -800,6 +800,51 @@ }; }; + /omit-if-no-ref/ + sdmmc2_b4_pins_b: sdmmc2-b4-1 { + pins1 { + pinmux = , /* SDMMC2_D0 */ + , /* SDMMC2_D1 */ + , /* SDMMC2_D2 */ + , /* SDMMC2_D3 */ + ; /* SDMMC2_CMD */ + slew-rate = <2>; + drive-push-pull; + bias-pull-up; + }; + pins2 { + pinmux = ; /* SDMMC2_CK */ + slew-rate = <3>; + drive-push-pull; + bias-disable; + }; + }; + + /omit-if-no-ref/ + sdmmc2_b4_od_pins_b: sdmmc2-b4-od-1 { + pins1 { + pinmux = , /* SDMMC2_D0 */ + , /* SDMMC2_D1 */ + , /* SDMMC2_D2 */ + ; /* SDMMC2_D3 */ + slew-rate = <2>; + drive-push-pull; + bias-pull-up; + }; + pins2 { + pinmux = ; /* SDMMC2_CK */ + slew-rate = <3>; + drive-push-pull; + bias-disable; + }; + pins3 { + pinmux = ; /* SDMMC2_CMD */ + slew-rate = <2>; + drive-open-drain; + bias-pull-up; + }; + }; + /omit-if-no-ref/ sdmmc2_d47_pins_a: sdmmc2-d47-0 { pins { @@ -912,6 +957,19 @@ }; }; + /omit-if-no-ref/ + sdmmc2_d47_pins_b: sdmmc2-d47-1 { + pins { + pinmux = , /* SDMMC2_D4 */ + , /* SDMMC2_D5 */ + , /* SDMMC2_D6 */ + ; /* SDMMC2_D7 */ + slew-rate = <2>; + drive-push-pull; + bias-pull-up; + }; + }; + /omit-if-no-ref/ spi1_pins_a: spi1-0 { pins1 { From 281ab10cc888af02c8f27e009b93a882bf58d6c6 Mon Sep 17 00:00:00 2001 From: Mateusz Nowicki Date: Wed, 23 Sep 2026 11:47:40 +0000 Subject: [PATCH 0162/1012] arm64: dts: st: enable eMMC on stm32mp257f-dk Enable the 8-Gbyte eMMC v5.1 connected to SDMMC2 with an 8-bit bus on the STM32MP257F-DK. VDDIO2 is tied to the 1.8 V rail on this board, so the DDR52 and HS200 modes are used. Use the _b pin groups, the _a groups do not give reliable reads in HS200 mode on this board. Tested in HS200 mode at 100 MHz. Signed-off-by: Mateusz Nowicki Link: https://lore.kernel.org/r/20260923114714.140842-3-mateusz.nowicki@posteo.net --- arch/arm64/boot/dts/st/stm32mp257f-dk.dts | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/arch/arm64/boot/dts/st/stm32mp257f-dk.dts b/arch/arm64/boot/dts/st/stm32mp257f-dk.dts index 39ad23d95069c9..a8f18508db3e45 100644 --- a/arch/arm64/boot/dts/st/stm32mp257f-dk.dts +++ b/arch/arm64/boot/dts/st/stm32mp257f-dk.dts @@ -22,6 +22,8 @@ aliases { ethernet0 = ðernet1; serial0 = &usart2; + mmc0 = &sdmmc1; + mmc1 = &sdmmc2; }; chosen { @@ -353,6 +355,23 @@ status = "okay"; }; +&sdmmc2 { + pinctrl-names = "default", "opendrain", "sleep"; + pinctrl-0 = <&sdmmc2_b4_pins_b &sdmmc2_d47_pins_b>; + pinctrl-1 = <&sdmmc2_b4_od_pins_b &sdmmc2_d47_pins_b>; + pinctrl-2 = <&sdmmc2_b4_sleep_pins_a &sdmmc2_d47_sleep_pins_a>; + non-removable; + no-sd; + no-sdio; + st,neg-edge; + bus-width = <8>; + mmc-ddr-1_8v; + mmc-hs200-1_8v; + vmmc-supply = <&scmi_vdd_emmc>; + vqmmc-supply = <&scmi_vddio2>; + status = "okay"; +}; + &usart2 { pinctrl-names = "default", "idle", "sleep"; pinctrl-0 = <&usart2_pins_a>; From 9e693c68f093138002f34e15c11fe7616c7c6c18 Mon Sep 17 00:00:00 2001 From: Dario Binacchi Date: Wed, 23 Sep 2026 17:53:09 +0200 Subject: [PATCH 0163/1012] arm64: dts: st: move Engicam MicroGEA-STM32MP257D-RMM display to an overlay The board is fitted with different displays depending on the product variant. Keep in the board dts only what is common to all variants and move the RK050HR345-CT106A RGB panel and its touchscreen to an overlay. Signed-off-by: Dario Binacchi Link: https://lore.kernel.org/r/20260923155333.610134-2-dario.binacchi@amarulasolutions.com Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/Makefile | 6 ++ ...icrogea-rmm-overlay-rk050hr345-ct106a.dtso | 68 +++++++++++++++++++ .../st/stm32mp257d-engicam-microgea-rmm.dts | 46 +------------ 3 files changed, 75 insertions(+), 45 deletions(-) create mode 100644 arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a.dtso diff --git a/arch/arm64/boot/dts/st/Makefile b/arch/arm64/boot/dts/st/Makefile index 931a64f87d8882..62105a9f87ad29 100644 --- a/arch/arm64/boot/dts/st/Makefile +++ b/arch/arm64/boot/dts/st/Makefile @@ -4,6 +4,10 @@ stm32mp255c-dhcos-dhsbc-overlay-imx219-x10-dtbs := \ stm32mp255c-dhcos-dhsbc.dtb \ stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtbo +stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a-dtbs := \ + stm32mp257d-engicam-microgea-rmm.dtb \ + stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a.dtbo + dtb-$(CONFIG_ARCH_STM32) += \ stm32mp215f-dk.dtb \ stm32mp235f-dk.dtb \ @@ -13,5 +17,7 @@ dtb-$(CONFIG_ARCH_STM32) += \ stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtb \ stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtbo \ stm32mp257d-engicam-microgea-rmm.dtb \ + stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a.dtb \ + stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a.dtbo \ stm32mp257f-dk.dtb \ stm32mp257f-ev1.dtb diff --git a/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a.dtso b/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a.dtso new file mode 100644 index 00000000000000..f89641bdb23b6f --- /dev/null +++ b/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a.dtso @@ -0,0 +1,68 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Copyright (C) 2026 Amarula Solutions, Dario Binacchi + * Copyright (C) 2026 Engicam srl + * + * Rocktech RK050HR345-CT106A 5" 480x854 RGB panel with EDT FT5306 + * I2C touchscreen controller. + */ + +/dts-v1/; +/plugin/; + +#include +#include + +&i2c1 { + pinctrl-names = "default", "sleep"; + pinctrl-0 = <&i2c1_pins_a>; + pinctrl-1 = <&i2c1_sleep_pins_a>; + i2c-scl-rising-time-ns = <185>; + i2c-scl-falling-time-ns = <20>; + #address-cells = <1>; + #size-cells = <0>; + status = "okay"; + + touchscreen@38 { + compatible = "edt,edt-ft5306"; + reg = <0x38>; + interrupt-parent = <&gpiob>; + interrupts = <0 IRQ_TYPE_EDGE_FALLING>; + reset-gpios = <&gpiod 1 GPIO_ACTIVE_LOW>; + touchscreen-size-x = <480>; + touchscreen-size-y = <854>; + }; +}; + +<dc { + pinctrl-names = "default", "sleep"; + pinctrl-0 = <<dc_pins_a>; + pinctrl-1 = <<dc_sleep_pins_a>; + status = "okay"; + + port { + ltdc_out: endpoint { + remote-endpoint = <&panel_in>; + }; + }; +}; + +&spi1 { + #address-cells = <1>; + #size-cells = <0>; + + display@0 { + compatible = "rocktech,rk050hr345-ct106a", "ilitek,ili9806e"; + reg = <0>; + vdd-supply = <®_3v3>; + spi-max-frequency = <10000000>; + reset-gpios = <&gpiob 6 GPIO_ACTIVE_LOW>; + backlight = <&backlight>; + + port { + panel_in: endpoint { + remote-endpoint = <<dc_out>; + }; + }; + }; +}; diff --git a/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm.dts b/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm.dts index f5af0913bf1b8c..faedddc2c914f1 100644 --- a/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm.dts +++ b/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm.dts @@ -116,25 +116,9 @@ }; &i2c1 { - pinctrl-names = "default", "sleep"; - pinctrl-0 = <&i2c1_pins_a>; - pinctrl-1 = <&i2c1_sleep_pins_a>; - i2c-scl-rising-time-ns = <185>; - i2c-scl-falling-time-ns = <20>; - status = "okay"; - /* spare dmas for other usage */ + /* dt overlays don't support /delete-property/ */ /delete-property/dmas; /delete-property/dma-names; - - touchscreen@38 { - compatible = "edt,edt-ft5306"; - reg = <0x38>; - interrupt-parent = <&gpiob>; - interrupts = <0 IRQ_TYPE_EDGE_FALLING>; - reset-gpios = <&gpiod 1 GPIO_ACTIVE_LOW>; - touchscreen-size-x = <480>; - touchscreen-size-y = <854>; - }; }; &i2c2 { @@ -179,19 +163,6 @@ }; }; -<dc { - pinctrl-names = "default", "sleep"; - pinctrl-0 = <<dc_pins_a>; - pinctrl-1 = <<dc_sleep_pins_a>; - status = "okay"; - - port { - ltdc_out: endpoint { - remote-endpoint = <&panel_in>; - }; - }; -}; - &m_can1 { pinctrl-names = "default", "sleep"; pinctrl-0 = <&m_can1_pins_a>; @@ -259,21 +230,6 @@ #size-cells = <0>; cs-gpios = <&gpioh 8 GPIO_ACTIVE_LOW>, <&gpioh 3 GPIO_ACTIVE_HIGH>; status = "okay"; - - display: display@0 { - compatible = "rocktech,rk050hr345-ct106a", "ilitek,ili9806e"; - reg = <0>; - vdd-supply = <®_3v3>; - spi-max-frequency = <10000000>; - reset-gpios = <&gpiob 6 GPIO_ACTIVE_LOW>; - backlight = <&backlight>; - - port { - panel_in: endpoint { - remote-endpoint = <<dc_out>; - }; - }; - }; }; &timers2 { From c38a11600dfde7b2d8bcdcc30bac6c329d24796b Mon Sep 17 00:00:00 2001 From: Dario Binacchi Date: Wed, 23 Sep 2026 17:53:10 +0200 Subject: [PATCH 0164/1012] arm64: dts: st: add Engicam MicroGEA-STM32MP257D-RMM LVDS display overlay Add an overlay for the variant fitted with the AM-1280800W8TZQW-T00H 10.1" 1280x800 LVDS panel. Its touchscreen is connected over USB, so there is no touchscreen node. Signed-off-by: Dario Binacchi Link: https://lore.kernel.org/r/20260923155333.610134-3-dario.binacchi@amarulasolutions.com Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/Makefile | 6 ++ ...gea-rmm-overlay-am-1280800w8tzqw-t00h.dtso | 67 +++++++++++++++++++ .../st/stm32mp257d-engicam-microgea-rmm.dts | 2 +- 3 files changed, 74 insertions(+), 1 deletion(-) create mode 100644 arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm-overlay-am-1280800w8tzqw-t00h.dtso diff --git a/arch/arm64/boot/dts/st/Makefile b/arch/arm64/boot/dts/st/Makefile index 62105a9f87ad29..9829a5b104fef1 100644 --- a/arch/arm64/boot/dts/st/Makefile +++ b/arch/arm64/boot/dts/st/Makefile @@ -4,6 +4,10 @@ stm32mp255c-dhcos-dhsbc-overlay-imx219-x10-dtbs := \ stm32mp255c-dhcos-dhsbc.dtb \ stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtbo +stm32mp257d-engicam-microgea-rmm-overlay-am-1280800w8tzqw-t00h-dtbs := \ + stm32mp257d-engicam-microgea-rmm.dtb \ + stm32mp257d-engicam-microgea-rmm-overlay-am-1280800w8tzqw-t00h.dtbo + stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a-dtbs := \ stm32mp257d-engicam-microgea-rmm.dtb \ stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a.dtbo @@ -17,6 +21,8 @@ dtb-$(CONFIG_ARCH_STM32) += \ stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtb \ stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtbo \ stm32mp257d-engicam-microgea-rmm.dtb \ + stm32mp257d-engicam-microgea-rmm-overlay-am-1280800w8tzqw-t00h.dtb \ + stm32mp257d-engicam-microgea-rmm-overlay-am-1280800w8tzqw-t00h.dtbo \ stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a.dtb \ stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a.dtbo \ stm32mp257f-dk.dtb \ diff --git a/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm-overlay-am-1280800w8tzqw-t00h.dtso b/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm-overlay-am-1280800w8tzqw-t00h.dtso new file mode 100644 index 00000000000000..7655c1bf56b1ef --- /dev/null +++ b/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm-overlay-am-1280800w8tzqw-t00h.dtso @@ -0,0 +1,67 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Copyright (C) 2026 Amarula Solutions, Dario Binacchi + * Copyright (C) 2026 Engicam srl + * + * Ampire AM-1280800W8TZQW-T00H 10.1" 1280x800 LVDS panel. Its touchscreen + * controller is connected over USB. + */ + +/dts-v1/; +/plugin/; + +#include + +&{/} { + panel-lvds { + compatible = "ampire,am-1280800w8tzqw-t00h"; + backlight = <&backlight>; + power-supply = <®_3v3>; + + port { + panel_in: endpoint { + remote-endpoint = <&lvds_out>; + }; + }; + }; +}; + +&framebuffer { + clocks = <&rcc CK_BUS_LTDC>, <&rcc CK_KER_LTDC>, + <&rcc CK_BUS_LVDS>, <&rcc CK_KER_LVDSPHY>; +}; + +<dc { + status = "okay"; + + port { + ltdc_out: endpoint { + remote-endpoint = <&lvds_in>; + }; + }; +}; + +&lvds { + status = "okay"; + + ports { + #address-cells = <1>; + #size-cells = <0>; + + port@0 { + reg = <0>; + + lvds_in: endpoint { + remote-endpoint = <<dc_out>; + }; + }; + + port@1 { + reg = <1>; + + lvds_out: endpoint { + remote-endpoint = <&panel_in>; + }; + }; + }; +}; diff --git a/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm.dts b/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm.dts index faedddc2c914f1..96b7cb6b5bc104 100644 --- a/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm.dts +++ b/arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm.dts @@ -43,7 +43,7 @@ #size-cells = <2>; ranges; - framebuffer { + framebuffer: framebuffer { compatible = "simple-framebuffer"; clocks = <&rcc CK_BUS_LTDC>, <&rcc CK_KER_LTDC>; lcd-supply = <®_3v3>; From bf3e03e9bae427addb3e57c976745af49242fd0c Mon Sep 17 00:00:00 2001 From: Mateusz Nowicki Date: Thu, 24 Sep 2026 19:23:10 +0000 Subject: [PATCH 0165/1012] arm64: dts: st: add digital temperature sensor on stm32mp251 The digital temperature sensor (DTS) of the STM32MP25 is a Moortec MR75203 PVT controller with two temperature sensors, handled by the mr75203 hwmon driver. The reference manual does not document the process detector and voltage monitor blocks that the controller reports and the driver maps. They were found at offsets 0x180 and 0x400 on the STM32MP257F-DK. Use the temperature coefficients from the STM32MP25 datasheet, as the driver defaults read about 1 degree Celsius too high. Signed-off-by: Mateusz Nowicki Link: https://lore.kernel.org/r/20260924192259.20517-2-mateusz.nowicki@posteo.net Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/stm32mp251.dtsi | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/arch/arm64/boot/dts/st/stm32mp251.dtsi b/arch/arm64/boot/dts/st/stm32mp251.dtsi index 992d77e14e74d1..60811a9e914154 100644 --- a/arch/arm64/boot/dts/st/stm32mp251.dtsi +++ b/arch/arm64/boot/dts/st/stm32mp251.dtsi @@ -1829,6 +1829,21 @@ }; }; + dts: thermal-sensor@44070000 { + compatible = "moortec,mr75203"; + reg = <0x44070000 0x80>, + <0x44070080 0xc0>, + <0x44070180 0x80>, + <0x44070400 0x400>; + reg-names = "common", "ts", "pd", "vm"; + clocks = <&rcc CK_KER_DTS>; + resets = <&rcc DTS_R>; + #thermal-sensor-cells = <1>; + moortec,ts-coeff-g = <58500>; + moortec,ts-coeff-h = <201200>; + moortec,ts-coeff-j = <0>; + }; + hdp: pinctrl@44090000 { compatible = "st,stm32mp251-hdp"; reg = <0x44090000 0x400>; From 2e5db74f0bf0adb30162a0c13ba971086aea32fa Mon Sep 17 00:00:00 2001 From: Mateusz Nowicki Date: Thu, 24 Sep 2026 19:23:11 +0000 Subject: [PATCH 0166/1012] arm64: defconfig: enable Moortec MR75203 PVT controller Enable the Moortec MR75203 PVT controller driver as a module. It is used by the digital temperature sensor of the STM32MP25 SoCs. Signed-off-by: Mateusz Nowicki Link: https://lore.kernel.org/r/20260924192259.20517-3-mateusz.nowicki@posteo.net Signed-off-by: Alexandre Torgue --- arch/arm64/configs/defconfig | 1 + 1 file changed, 1 insertion(+) diff --git a/arch/arm64/configs/defconfig b/arch/arm64/configs/defconfig index f6d3d94ab7db14..fd4da6a61ce6b1 100644 --- a/arch/arm64/configs/defconfig +++ b/arch/arm64/configs/defconfig @@ -765,6 +765,7 @@ CONFIG_SENSORS_ARM_SCPI=y CONFIG_SENSORS_GPIO_FAN=m CONFIG_SENSORS_JC42=m CONFIG_SENSORS_MACSMC_HWMON=m +CONFIG_SENSORS_MR75203=m CONFIG_SENSORS_LM75=m CONFIG_SENSORS_LM90=m CONFIG_SENSORS_PWM_FAN=m From a565b7669b35df8aca237d27a61b53a7ee63c270 Mon Sep 17 00:00:00 2001 From: Mateusz Nowicki Date: Thu, 24 Sep 2026 19:38:09 +0000 Subject: [PATCH 0167/1012] arm64: dts: st: fix sai3b unit address on stm32mp251 The unit address of the sai3b node, 502b0024, is the address of the SAI_BCR1 register in the secure alias of SAI3. The reg property of the node translates to the non-secure address 0x402b0024, which all other SAI nodes use. Fix the unit address to match. Fixes: bf26d75a95f1 ("arm64: dts: st: add sai support on stm32mp251") Signed-off-by: Mateusz Nowicki Link: https://lore.kernel.org/r/20260924193759.27881-2-mateusz.nowicki@posteo.net Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/stm32mp251.dtsi | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm64/boot/dts/st/stm32mp251.dtsi b/arch/arm64/boot/dts/st/stm32mp251.dtsi index 60811a9e914154..b624f998b87c1f 100644 --- a/arch/arm64/boot/dts/st/stm32mp251.dtsi +++ b/arch/arm64/boot/dts/st/stm32mp251.dtsi @@ -1291,7 +1291,7 @@ status = "disabled"; }; - sai3b: audio-controller@502b0024 { + sai3b: audio-controller@402b0024 { compatible = "st,stm32-sai-sub-b"; reg = <0x24 0x20>; #sound-dai-cells = <0>; From e6cdc9f49eebe486c256e134be84175865a357e9 Mon Sep 17 00:00:00 2001 From: Mateusz Nowicki Date: Thu, 24 Sep 2026 19:38:10 +0000 Subject: [PATCH 0168/1012] arm64: dts: st: fix sai3b unit address on stm32mp231 The unit address of the sai3b node, 502b0024, is the address of the SAI_BCR1 register in the secure alias of SAI3. The reg property of the node translates to the non-secure address 0x402b0024, which all other SAI nodes use. Fix the unit address to match. Fixes: e9b03ef21386 ("arm64: dts: st: introduce stm32mp23 SoCs family") Signed-off-by: Mateusz Nowicki Link: https://lore.kernel.org/r/20260924193759.27881-3-mateusz.nowicki@posteo.net Signed-off-by: Alexandre Torgue --- arch/arm64/boot/dts/st/stm32mp231.dtsi | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm64/boot/dts/st/stm32mp231.dtsi b/arch/arm64/boot/dts/st/stm32mp231.dtsi index f72ee55ea2c3ca..a60ae2765d4d5d 100644 --- a/arch/arm64/boot/dts/st/stm32mp231.dtsi +++ b/arch/arm64/boot/dts/st/stm32mp231.dtsi @@ -631,7 +631,7 @@ status = "disabled"; }; - sai3b: audio-controller@502b0024 { + sai3b: audio-controller@402b0024 { compatible = "st,stm32-sai-sub-b"; reg = <0x24 0x20>; #sound-dai-cells = <0>; From d1b210c045f706927eaf6f1bda41a07d37672e9b Mon Sep 17 00:00:00 2001 From: Tangudu Tilak Tirumalesh Date: Fri, 25 Sep 2026 22:57:22 +0530 Subject: [PATCH 0169/1012] drm/xe/xe3p_lpg: Add support for Wa_16030224309 Opt in to the GuC workaround, to ignore context completion status with preemption or preemption-to-idle in the CSB when the multi-queue feature is enabled, which is supported on Graphics Version 35.10. This WA is supported starting from GuC 70.73. Apply Wa_16030224309 to Graphics Version 35.10. v2: Place this code near to KLV WAs. -Lin Shuicheng Signed-off-by: Tangudu Tilak Tirumalesh Reviewed-by: Gustavo Sousa Reviewed-by: Shuicheng Lin Link: https://patch.msgid.link/20260925172722.238924-1-tilak.tirumalesh.tangudu@intel.com Signed-off-by: Matt Roper --- drivers/gpu/drm/xe/abi/guc_klvs_abi.h | 1 + drivers/gpu/drm/xe/xe_guc_ads.c | 6 ++++++ drivers/gpu/drm/xe/xe_wa_oob.rules | 1 + 3 files changed, 8 insertions(+) diff --git a/drivers/gpu/drm/xe/abi/guc_klvs_abi.h b/drivers/gpu/drm/xe/abi/guc_klvs_abi.h index 685c4ef17b7329..4bcdc1a96785eb 100644 --- a/drivers/gpu/drm/xe/abi/guc_klvs_abi.h +++ b/drivers/gpu/drm/xe/abi/guc_klvs_abi.h @@ -525,6 +525,7 @@ enum xe_guc_klv_ids { GUC_WA_KLV_CLR_CS_INDIRECT_RING_STATE_IF_IDLE_AT_CTX_REG = 0x900e, GUC_WA_KLV_REMAP_RANGED_TLB_INV = 0x900f, GUC_WA_KLV_IGNORE_MMIO_READ_SEM_TOKEN_64 = 0x9010, + GUC_WA_KLV_IGNORE_MULTIQ_CTX_COMPLETE_WITH_PREEMPT_IN_CSB = 0x9011, }; /** diff --git a/drivers/gpu/drm/xe/xe_guc_ads.c b/drivers/gpu/drm/xe/xe_guc_ads.c index 9ceb662ca40e4c..43a2951a66d648 100644 --- a/drivers/gpu/drm/xe/xe_guc_ads.c +++ b/drivers/gpu/drm/xe/xe_guc_ads.c @@ -383,6 +383,12 @@ static void guc_waklv_init(struct xe_guc_ads *ads) guc_waklv_enable(ads, NULL, 0, &offset, &remain, GUC_WA_KLV_IGNORE_MMIO_READ_SEM_TOKEN_64); + /* GuC only applies this WA for scheduled MultiQ contexts; no KMD-side gating needed. */ + if (XE_GT_WA(gt, 16030224309) && + GUC_FIRMWARE_VER_AT_LEAST(>->uc.guc, 70, 73)) + guc_waklv_enable(ads, NULL, 0, &offset, &remain, + GUC_WA_KLV_IGNORE_MULTIQ_CTX_COMPLETE_WITH_PREEMPT_IN_CSB); + /* * On GuC firmware 70.66 and above, use the Feature KLV (shared with the * WA KLV buffer); older firmware uses GUC_CTL_DISABLE_MULTI_QUEUE in diff --git a/drivers/gpu/drm/xe/xe_wa_oob.rules b/drivers/gpu/drm/xe/xe_wa_oob.rules index 6662987effd739..f35181fc4e49b6 100644 --- a/drivers/gpu/drm/xe/xe_wa_oob.rules +++ b/drivers/gpu/drm/xe/xe_wa_oob.rules @@ -75,3 +75,4 @@ 14027054324 GRAPHICS_VERSION(3511) 14025941587 GRAPHICS_VERSION_RANGE(2001, 3511), FUNC(xe_rtp_match_not_sriov_vf) MEDIA_VERSION_RANGE(1301, 3503), FUNC(xe_rtp_match_not_sriov_vf) +16030224309 GRAPHICS_VERSION(3510) From 28f76170fd90051dd65a8330004fff3f386a56da Mon Sep 17 00:00:00 2001 From: Yichong Chen Date: Fri, 7 Aug 2026 18:00:51 +0800 Subject: [PATCH 0170/1012] interconnect: debugfs: replace writable string helper debugfs_create_str() is being made read-only because its generic write path is hard to make safe without adding more locking to the helper. Convert the interconnect debugfs client src_node and dst_node entries to local file operations before removing writable string support from debugfs_create_str(). Protect the string replacement and path lookup with the existing debugfs_lock. The old code duplicated the strings under rcu_read_lock(), so it had to use GFP_ATOMIC. The local file operations protect src_node and dst_node with debugfs_lock instead, so the allocation can use GFP_KERNEL. Signed-off-by: Yichong Chen Link: https://patch.msgid.link/20260807100053.1089834-2-chenyichong@uniontech.com Signed-off-by: Georgi Djakov --- drivers/interconnect/debugfs-client.c | 81 +++++++++++++++++++++------ 1 file changed, 64 insertions(+), 17 deletions(-) diff --git a/drivers/interconnect/debugfs-client.c b/drivers/interconnect/debugfs-client.c index 91f86d9237a688..238d282303ffad 100644 --- a/drivers/interconnect/debugfs-client.c +++ b/drivers/interconnect/debugfs-client.c @@ -5,6 +5,7 @@ #include #include #include +#include #include "internal.h" @@ -36,6 +37,59 @@ struct debugfs_path { struct list_head list; }; +static ssize_t icc_node_read(struct file *file, char __user *user_buf, + size_t count, loff_t *ppos) +{ + char **node = file->private_data; + char *copy; + size_t len; + ssize_t ret; + + scoped_guard(mutex, &debugfs_lock) { + copy = kstrdup(*node ?: "", GFP_KERNEL); + } + if (!copy) + return -ENOMEM; + + len = strlen(copy); + copy[len++] = '\n'; + ret = simple_read_from_buffer(user_buf, count, ppos, copy, len); + kfree(copy); + return ret; +} + +static ssize_t icc_node_write(struct file *file, const char __user *user_buf, + size_t count, loff_t *ppos) +{ + char **node = file->private_data; + char *old, *new; + + if (*ppos) + return -EINVAL; + if (count + 1 > PAGE_SIZE) + return -E2BIG; + + new = memdup_user_nul(user_buf, count); + if (IS_ERR(new)) + return PTR_ERR(new); + strim(new); + + scoped_guard(mutex, &debugfs_lock) { + old = *node; + *node = new; + } + + kfree(old); + return count; +} + +static const struct file_operations icc_node_fops = { + .open = simple_open, + .read = icc_node_read, + .write = icc_node_write, + .llseek = default_llseek, +}; + static struct icc_path *get_path(const char *src, const char *dst) { struct debugfs_path *path; @@ -54,26 +108,19 @@ static int icc_get_set(void *data, u64 val) char *src, *dst; int ret = 0; - mutex_lock(&debugfs_lock); - - rcu_read_lock(); - src = rcu_dereference(src_node); - dst = rcu_dereference(dst_node); + guard(mutex)(&debugfs_lock); /* * If we've already looked up a path, then use the existing one instead * of calling icc_get() again. This allows for updating previous BW * votes when "get" is written to multiple times for multiple paths. */ - cur_path = get_path(src, dst); - if (cur_path) { - rcu_read_unlock(); + cur_path = get_path(src_node, dst_node); + if (cur_path) goto out; - } - src = kstrdup(src, GFP_ATOMIC); - dst = kstrdup(dst, GFP_ATOMIC); - rcu_read_unlock(); + src = kstrdup(src_node, GFP_KERNEL); + dst = kstrdup(dst_node, GFP_KERNEL); if (!src || !dst) { ret = -ENOMEM; @@ -105,7 +152,6 @@ static int icc_get_set(void *data, u64 val) kfree(src); kfree(dst); out: - mutex_unlock(&debugfs_lock); return ret; } @@ -115,7 +161,7 @@ static int icc_commit_set(void *data, u64 val) { int ret; - mutex_lock(&debugfs_lock); + guard(mutex)(&debugfs_lock); if (!cur_path) { ret = -EINVAL; @@ -130,7 +176,6 @@ static int icc_commit_set(void *data, u64 val) icc_set_tag(cur_path, tag); ret = icc_set_bw(cur_path, avg_bw, peak_bw); out: - mutex_unlock(&debugfs_lock); return ret; } @@ -162,8 +207,10 @@ int icc_debugfs_client_init(struct dentry *icc_dir) client_dir = debugfs_create_dir("test_client", icc_dir); - debugfs_create_str("src_node", 0600, client_dir, &src_node); - debugfs_create_str("dst_node", 0600, client_dir, &dst_node); + debugfs_create_file("src_node", 0600, client_dir, &src_node, + &icc_node_fops); + debugfs_create_file("dst_node", 0600, client_dir, &dst_node, + &icc_node_fops); debugfs_create_file("get", 0200, client_dir, NULL, &icc_get_fops); debugfs_create_u32("avg_bw", 0600, client_dir, &avg_bw); debugfs_create_u32("peak_bw", 0600, client_dir, &peak_bw); From 07767d98ce0a40972bc7dadc8c1436f0490eb143 Mon Sep 17 00:00:00 2001 From: Rosen Penev Date: Wed, 23 Sep 2026 11:50:59 -0700 Subject: [PATCH 0171/1012] interconnect: qcom: fix endian annotations of BCM aux data struct qcom_icc_bcm::aux_data keeps a copy of the struct bcm_db read from the command db, whose unit and width fields are annotated as __le32/__le16 to describe the little-endian on-disk format. Using those restricted types directly in bandwidth calculations makes sparse complain about endianness. Keep aux_data typed as struct bcm_db, copied verbatim from the command db buffer, and convert the fields with le32_to_cpu()/le16_to_cpu() at the places where unit and width are used. No functional change. Reported-by: kernel test robot Closes: https://lore.kernel.org/oe-kbuild-all/202609221350.3Y8MOce9-lkp@intel.com/ Assisted-by: LLM Signed-off-by: Rosen Penev Link: https://patch.msgid.link/20260923185059.25196-1-rosenp@gmail.com Signed-off-by: Georgi Djakov --- drivers/interconnect/qcom/bcm-voter.c | 8 ++++---- drivers/interconnect/qcom/icc-rpmh.c | 13 +++++-------- 2 files changed, 9 insertions(+), 12 deletions(-) diff --git a/drivers/interconnect/qcom/bcm-voter.c b/drivers/interconnect/qcom/bcm-voter.c index c15abb57cd24db..8afeb4604d40f8 100644 --- a/drivers/interconnect/qcom/bcm-voter.c +++ b/drivers/interconnect/qcom/bcm-voter.c @@ -99,20 +99,20 @@ static void bcm_aggregate(struct qcom_icc_bcm *bcm) for (bucket = 0; bucket < QCOM_ICC_NUM_BUCKETS; bucket++) { for (i = 0; i < bcm->num_nodes; i++) { node = bcm->nodes[i]; - temp = bcm_div(node->sum_avg[bucket] * bcm->aux_data.width, + temp = bcm_div(node->sum_avg[bucket] * le16_to_cpu(bcm->aux_data.width), node->buswidth * node->channels); agg_avg[bucket] = max(agg_avg[bucket], temp); - temp = bcm_div(node->max_peak[bucket] * bcm->aux_data.width, + temp = bcm_div(node->max_peak[bucket] * le16_to_cpu(bcm->aux_data.width), node->buswidth); agg_peak[bucket] = max(agg_peak[bucket], temp); } temp = agg_avg[bucket] * bcm->vote_scale; - bcm->vote_x[bucket] = bcm_div(temp, bcm->aux_data.unit); + bcm->vote_x[bucket] = bcm_div(temp, le32_to_cpu(bcm->aux_data.unit)); temp = agg_peak[bucket] * bcm->vote_scale; - bcm->vote_y[bucket] = bcm_div(temp, bcm->aux_data.unit); + bcm->vote_y[bucket] = bcm_div(temp, le32_to_cpu(bcm->aux_data.unit)); } if (bcm->keepalive && bcm->vote_x[QCOM_ICC_BUCKET_AMC] == 0 && diff --git a/drivers/interconnect/qcom/icc-rpmh.c b/drivers/interconnect/qcom/icc-rpmh.c index 38c8c3cb9a38f2..4a2a746a4e59db 100644 --- a/drivers/interconnect/qcom/icc-rpmh.c +++ b/drivers/interconnect/qcom/icc-rpmh.c @@ -166,19 +166,19 @@ static int qcom_icc_get_bw(struct icc_node *node, u32 *avg, u32 *peak) peak_max = INT_MAX; } else { if (x) { - x *= bcm->aux_data.unit; + x *= le32_to_cpu(bcm->aux_data.unit); do_div(x, bcm->vote_scale); x *= qn->buswidth * qn->channels; - do_div(x, bcm->aux_data.width); + do_div(x, le16_to_cpu(bcm->aux_data.width)); avg_max = max(avg_max, x); } if (y) { - y *= bcm->aux_data.unit; + y *= le32_to_cpu(bcm->aux_data.unit); do_div(y, bcm->vote_scale); y *= qn->buswidth; - do_div(y, bcm->aux_data.width); + do_div(y, le16_to_cpu(bcm->aux_data.width)); peak_max = max(peak_max, y); } @@ -228,10 +228,7 @@ int qcom_icc_bcm_init(struct qcom_icc_bcm *bcm, struct device *dev) return -EINVAL; } - bcm->aux_data.unit = le32_to_cpu(data->unit); - bcm->aux_data.width = le16_to_cpu(data->width); - bcm->aux_data.vcd = data->vcd; - bcm->aux_data.reserved = data->reserved; + bcm->aux_data = *data; INIT_LIST_HEAD(&bcm->list); INIT_LIST_HEAD(&bcm->ws_list); From 7996f5d105d8b90104d3daffa669aa5a746a4db1 Mon Sep 17 00:00:00 2001 From: Breno Rodrigues Alves Date: Sat, 5 Sep 2026 11:01:11 -0300 Subject: [PATCH 0172/1012] interconnect: mediatek: fix Makefile typo for mt8196 Correct a copy-paste typo in the MediaTek interconnect Makefile that mapped mt8196.o to CONFIG_INTERCONNECT_MTK_MT8195 instead of MT8196. Assisted-by: OpenCode AI Signed-off-by: Breno Rodrigues Alves Link: https://patch.msgid.link/20260905140113.21638-1-breno3011alves@gmail.com Signed-off-by: Georgi Djakov --- drivers/interconnect/mediatek/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/interconnect/mediatek/Makefile b/drivers/interconnect/mediatek/Makefile index 6bd656668f5d8b..64170ab163ff7b 100644 --- a/drivers/interconnect/mediatek/Makefile +++ b/drivers/interconnect/mediatek/Makefile @@ -3,4 +3,4 @@ obj-$(CONFIG_INTERCONNECT_MTK_DVFSRC_EMI) += icc-emi.o obj-$(CONFIG_INTERCONNECT_MTK_MT8183) += mt8183.o obj-$(CONFIG_INTERCONNECT_MTK_MT8195) += mt8195.o -obj-$(CONFIG_INTERCONNECT_MTK_MT8195) += mt8196.o +obj-$(CONFIG_INTERCONNECT_MTK_MT8196) += mt8196.o From 69e41b870d0ad4f9ede112425cf694c8fdbb3c96 Mon Sep 17 00:00:00 2001 From: Sean Christopherson Date: Wed, 23 Sep 2026 08:51:06 -0700 Subject: [PATCH 0173/1012] KVM: SVM: Add paranoid helper for checking if vCPU is AVIC-addressable Add a helper to check if a vCPU is addressable by AVIC hardware, i.e. has an APIC ID that fits in the physical ID table, and use the more paranoid helper when determining if a vCPU is compatible with AVIC when initializing the vCPU. KVM is supposed to reject vCPU creation if the vCPU's ID is greater than or equal to max_vcpu_ids, i.e. simply checking the architectural maximum *should* suffice. But piecing together why this is safe is unnecessarily difficult, and there is no meaningful downside to being extra cautious. Suggested-by: Naveen N Rao (AMD) Reviewed-by: Naveen N Rao (AMD) Link: https://patch.msgid.link/20260923155108.1550622-2-seanjc@google.com Signed-off-by: Sean Christopherson --- arch/x86/kvm/svm/avic.c | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/arch/x86/kvm/svm/avic.c b/arch/x86/kvm/svm/avic.c index 3b037e3855234e..0e4b5eb6ac82ba 100644 --- a/arch/x86/kvm/svm/avic.c +++ b/arch/x86/kvm/svm/avic.c @@ -395,6 +395,11 @@ static phys_addr_t avic_get_backing_page_address(struct vcpu_svm *svm) return __sme_set(__pa(svm->vcpu.arch.apic->regs)); } +static bool avic_is_addressable_vcpu(struct kvm_vcpu *vcpu) +{ + return vcpu->vcpu_id <= __avic_get_max_physical_id(vcpu->kvm, NULL); +} + void avic_init_vmcb(struct vcpu_svm *svm, struct vmcb *vmcb) { struct kvm_svm *kvm_svm = to_kvm_svm(svm->vcpu.kvm); @@ -412,7 +417,6 @@ void avic_init_vmcb(struct vcpu_svm *svm, struct vmcb *vmcb) static int avic_init_backing_page(struct kvm_vcpu *vcpu) { - u32 max_id = x2avic_enabled ? x2avic_max_physical_id : AVIC_MAX_PHYSICAL_ID; struct kvm_svm *kvm_svm = to_kvm_svm(vcpu->kvm); struct vcpu_svm *svm = to_svm(vcpu); u32 id = vcpu->vcpu_id; @@ -425,7 +429,7 @@ static int avic_init_backing_page(struct kvm_vcpu *vcpu) * avic_vcpu_load() expects to be called if and only if the vCPU has * fully initialized AVIC. */ - if (id > max_id) { + if (!avic_is_addressable_vcpu(vcpu)) { kvm_set_apicv_inhibit(vcpu->kvm, APICV_INHIBIT_REASON_PHYSICAL_ID_TOO_BIG); vcpu->arch.apic->apicv_active = false; return 0; From ce464958a432f675b896915f74a5366d02e08693 Mon Sep 17 00:00:00 2001 From: Sean Christopherson Date: Wed, 23 Sep 2026 08:51:07 -0700 Subject: [PATCH 0174/1012] KVM: SVM: Use "is AVIC-addressable" helper to sanity check load()/put() Use avic_is_addressable_vcpu() instead of open coding a check on the bounds of the allocated table for the sanity checks when loading/putting AVIC state for a vCPU. If KVM botches the allocation, then KVM will already have performed an OOB write in avic_init_backing_page(), i.e. being super paranoid in load()/put() doesn't provide meaningful protection in practice. Cc: Naveen N Rao (AMD) Reviewed-by: Naveen N Rao (AMD) Link: https://patch.msgid.link/20260923155108.1550622-3-seanjc@google.com Signed-off-by: Sean Christopherson --- arch/x86/kvm/svm/avic.c | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/arch/x86/kvm/svm/avic.c b/arch/x86/kvm/svm/avic.c index 0e4b5eb6ac82ba..4173a30dfe608f 100644 --- a/arch/x86/kvm/svm/avic.c +++ b/arch/x86/kvm/svm/avic.c @@ -1049,8 +1049,7 @@ static void __avic_vcpu_load(struct kvm_vcpu *vcpu, int cpu, if (WARN_ON(h_physical_id & ~AVIC_PHYSICAL_ID_ENTRY_HOST_PHYSICAL_ID_MASK)) return; - if (WARN_ON_ONCE(vcpu->vcpu_id * sizeof(entry) >= - PAGE_SIZE << avic_get_physical_id_table_order(vcpu->kvm))) + if (WARN_ON_ONCE(!avic_is_addressable_vcpu(vcpu))) return; /* @@ -1112,8 +1111,7 @@ static void __avic_vcpu_put(struct kvm_vcpu *vcpu, enum avic_vcpu_action action) lockdep_assert_preemption_disabled(); - if (WARN_ON_ONCE(vcpu->vcpu_id * sizeof(entry) >= - PAGE_SIZE << avic_get_physical_id_table_order(vcpu->kvm))) + if (WARN_ON_ONCE(!avic_is_addressable_vcpu(vcpu))) return; /* From 1b84a8226fb2d91862f9292ab4a7d12b48889675 Mon Sep 17 00:00:00 2001 From: "Naveen N Rao (AMD)" Date: Wed, 23 Sep 2026 08:51:08 -0700 Subject: [PATCH 0175/1012] KVM: SVM: Clear AVIC Physical ID table entry if vCPU creation fails If vCPU creation fails after kvm_arch_vcpu_create(), the AVIC Physical ID table entry corresponding to that vCPU continues to point to the freed APIC backing page which can result in UAF. Address this by clearing out the corresponding AVIC Physical ID table entry in the vcpu_free() callback, similar to the VMX commit b41f2ca6c060 ("KVM: VMX: Fix stale PID-pointer table entry left after vCPU free"). Though unlikely, it is also possible that svm_vcpu_create() itself fails after the AVIC Physical ID table entry has been setup if memory allocation fails in svm_vcpu_alloc_msrpm(). Clear the entry in this path as well. Note: this change depends on commit 97d65b544f48 ("KVM: Check for duplicate vcpu_id as early as possible"), which ensures that a vCPU with a duplicate ID is never created. Otherwise, a valid AVIC Physical ID table entry for an existing vCPU will be cleared. Fixes: 44a95dae1d22 ("KVM: x86: Detect and Initialize AVIC support") Signed-off-by: Naveen N Rao (AMD) Tested-by: Atish Patra [sean: use avic_is_addressable_vcpu()] Link: https://patch.msgid.link/20260923155108.1550622-4-seanjc@google.com Signed-off-by: Sean Christopherson --- arch/x86/kvm/svm/avic.c | 8 ++++++++ arch/x86/kvm/svm/svm.c | 6 +++++- arch/x86/kvm/svm/svm.h | 1 + 3 files changed, 14 insertions(+), 1 deletion(-) diff --git a/arch/x86/kvm/svm/avic.c b/arch/x86/kvm/svm/avic.c index 4173a30dfe608f..96ca39c0045f10 100644 --- a/arch/x86/kvm/svm/avic.c +++ b/arch/x86/kvm/svm/avic.c @@ -889,6 +889,14 @@ int avic_init_vcpu(struct vcpu_svm *svm) return ret; } +void avic_vcpu_free(struct kvm_vcpu *vcpu) +{ + struct kvm_svm *kvm_svm = to_kvm_svm(vcpu->kvm); + + if (kvm_svm->avic_physical_id_table && avic_is_addressable_vcpu(vcpu)) + WRITE_ONCE(kvm_svm->avic_physical_id_table[vcpu->vcpu_id], 0); +} + void avic_apicv_post_state_restore(struct kvm_vcpu *vcpu) { avic_handle_dfr_update(vcpu); diff --git a/arch/x86/kvm/svm/svm.c b/arch/x86/kvm/svm/svm.c index fe0cba4731b2ef..c054803d4673df 100644 --- a/arch/x86/kvm/svm/svm.c +++ b/arch/x86/kvm/svm/svm.c @@ -1328,7 +1328,7 @@ static int svm_vcpu_create(struct kvm_vcpu *vcpu) svm->msrpm = svm_vcpu_alloc_msrpm(); if (!svm->msrpm) { err = -ENOMEM; - goto error_free_sev; + goto error_free_avic; } svm->x2avic_msrs_intercepted = true; @@ -1342,6 +1342,8 @@ static int svm_vcpu_create(struct kvm_vcpu *vcpu) return 0; +error_free_avic: + avic_vcpu_free(vcpu); error_free_sev: sev_free_vcpu(vcpu); error_free_vmcb_page: @@ -1356,6 +1358,8 @@ static void svm_vcpu_free(struct kvm_vcpu *vcpu) WARN_ON_ONCE(!list_empty(&svm->ir_list)); + avic_vcpu_free(vcpu); + svm_leave_nested(vcpu); svm_free_nested(svm); diff --git a/arch/x86/kvm/svm/svm.h b/arch/x86/kvm/svm/svm.h index 84f19026d3e8bb..29bde741874ad1 100644 --- a/arch/x86/kvm/svm/svm.h +++ b/arch/x86/kvm/svm/svm.h @@ -955,6 +955,7 @@ void avic_init_vmcb(struct vcpu_svm *svm, struct vmcb *vmcb); int avic_incomplete_ipi_interception(struct kvm_vcpu *vcpu); int avic_unaccelerated_access_interception(struct kvm_vcpu *vcpu); int avic_init_vcpu(struct vcpu_svm *svm); +void avic_vcpu_free(struct kvm_vcpu *vcpu); void avic_vcpu_load(struct kvm_vcpu *vcpu, int cpu); void avic_vcpu_put(struct kvm_vcpu *vcpu); void avic_apicv_post_state_restore(struct kvm_vcpu *vcpu); From 9e389110fd668958735141af24d04ed040776235 Mon Sep 17 00:00:00 2001 From: Jann Horn Date: Mon, 10 Aug 2026 17:33:59 +0200 Subject: [PATCH 0176/1012] KVM: SEV: Fix page dirtying in sev_gmem_post_populate() set_page_dirty() requires that the caller holds some kind of lock to ensure that the page's mapping does not concurrently go away. That is not the case for a random page we got from get_user_pages_fast(), so use set_page_dirty_lock(). Fixes: 97cd21d57e9b ("KVM: SEV: Mark source page dirty when writing back CPUID data on failure") Cc: stable@vger.kernel.org Signed-off-by: Jann Horn Link: https://patch.msgid.link/20260810-x86-kvm-setpagedirty-v1-1-85f180892d4f@google.com [sean: tag for stable] Signed-off-by: Sean Christopherson --- arch/x86/kvm/svm/sev.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/x86/kvm/svm/sev.c b/arch/x86/kvm/svm/sev.c index 3448d56520c69a..40a1339a501a34 100644 --- a/arch/x86/kvm/svm/sev.c +++ b/arch/x86/kvm/svm/sev.c @@ -2408,7 +2408,7 @@ static int sev_gmem_post_populate(struct kvm *kvm, gfn_t gfn, kvm_pfn_t pfn, void *dst_vaddr = kmap_local_pfn(pfn); memcpy(src_vaddr, dst_vaddr, PAGE_SIZE); - set_page_dirty(src_page); + set_page_dirty_lock(src_page); kunmap_local(dst_vaddr); kunmap_local(src_vaddr); From 56329611a670eedc78e1870ae96c2b7ccc580709 Mon Sep 17 00:00:00 2001 From: Jinu Kim Date: Fri, 7 Aug 2026 20:28:18 +0900 Subject: [PATCH 0177/1012] KVM: x86: Use active memslots for the per-vCPU MMIO cache KVM tags the per-vCPU MMIO cache with a memslot generation so that a memslot update invalidates cached MMIO information. Both the fill and validation paths use kvm_memslots(), which unconditionally selects address space 0, even when the vCPU is running in SMM and using address space 1. Commit 56f17dd3fbc4 ("kvm: x86: fix stale mmio cache bug") added the memslot generation to the cache key so that a memslot update could not leave a stale entry valid. When commit 699023e23965 ("KVM: x86: add SMM to the MMU role, support SMRAM address space") added the SMM address space, these helpers were not converted to use the active memslots. Consequently, an update to the SMM memslots can leave an entry from the old SMM address space apparently valid after the vCPU returns to the normal address space. A guest can then cause an access to valid RAM at the same GFN to be returned to userspace as KVM_EXIT_MMIO. Completing that exit through the VMM's RAM address space writes the backing page without going through KVM's write-tracking path. If the page backs a nested EPT table, KVM can continue using shadow translations derived from the old contents. The WARN_ON_ONCE() in kvm_mmu_write_protect_fault() catches this invalid cache and memslot combination. Use the vCPU's active memslots when caching and validating the entry. Memslot generations are unique across address spaces, so the generation also distinguishes the normal and SMM views. This matches the MMIO SPTE cache, which already uses kvm_vcpu_memslots(). On an unpatched current-mainline kernel, a regression test that fills the cache in SMM and updates only the SMM memslots fails the intended RAM write and triggers the kvm_mmu_write_protect_fault() warning. With this change, the same test completes the RAM write without a warning. The x86/smm_test, set_memory_region_test, and memslot_modification_stress_test selftests also pass. Fixes: 699023e23965 ("KVM: x86: add SMM to the MMU role, support SMRAM address space") Cc: stable@vger.kernel.org Signed-off-by: Jinu Kim Link: https://patch.msgid.link/20260807112818.958190-1-kimjw04271234@gmail.com Signed-off-by: Sean Christopherson --- arch/x86/kvm/x86.h | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/arch/x86/kvm/x86.h b/arch/x86/kvm/x86.h index 0f5919b092e475..f72897ae56bc6d 100644 --- a/arch/x86/kvm/x86.h +++ b/arch/x86/kvm/x86.h @@ -250,7 +250,7 @@ static inline bool is_noncanonical_invlpg_address(u64 la, struct kvm_vcpu *vcpu) static inline void vcpu_cache_mmio_info(struct kvm_vcpu *vcpu, gva_t gva, gfn_t gfn, unsigned access) { - u64 gen = kvm_memslots(vcpu->kvm)->generation; + u64 gen = kvm_vcpu_memslots(vcpu)->generation; if (unlikely(gen & KVM_MEMSLOT_GEN_UPDATE_IN_PROGRESS)) return; @@ -267,7 +267,7 @@ static inline void vcpu_cache_mmio_info(struct kvm_vcpu *vcpu, static inline bool vcpu_match_mmio_gen(struct kvm_vcpu *vcpu) { - return vcpu->arch.mmio_gen == kvm_memslots(vcpu->kvm)->generation; + return vcpu->arch.mmio_gen == kvm_vcpu_memslots(vcpu)->generation; } /* From 0ad5dc33c83c6bfdf7190b470fe8b2609052c9e4 Mon Sep 17 00:00:00 2001 From: Josua Mayer Date: Mon, 28 Sep 2026 19:33:55 +0200 Subject: [PATCH 0178/1012] arm64: dts: renesas: Add support for solidrun rzg2l som and hb-iiot evb Add support for the SolidRun RZ/G2L SoM [1] on HummingBoard IIoT [2]. The SoM features: - 2x 1Gbps Ethernet with PHY - eMMC - 1/2GB DDR - WiFi + Bluetooth - SDHI Mux switching between eMMC and Carrier Board The HummingBoard IIoT features: - 3x USB-2.0 Type A connector - 2x 1Gbps RJ45 Ethernet - USB Type-C Console Port - microSD connector - RTC with backup battery - RGB Status LED - 1x M.2 B-Key connector with USB-2.0 + SIM card holder - 1x DSI Display Connector - GPIO header - 2x RS232/RS485 ports (configurable) - 2x CAN Descriptions for eMMC, microSD and RS485 are provided as overlays due to their dependency on configurable mux states. [1] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/rz-g2l-som/ [2] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/hummingboard-rz-series-sbcs/hummingboard-rz-g2l-iot-sbc/ Reviewed-by: Geert Uytterhoeven Signed-off-by: Josua Mayer Link: https://patch.msgid.link/20260928-rzg2-sr-boards-v10-4-91752ea39cbc@solid-run.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/Makefile | 13 + .../renesas/r9a07g044l2-hummingboard-iiot.dts | 16 + .../rzg2l-hummingboard-iiot-common.dtsi | 560 ++++++++++++++++++ .../rzg2l-hummingboard-iiot-microsd.dtso | 26 + .../rzg2l-hummingboard-iiot-rs485-a.dtso | 17 + .../rzg2l-hummingboard-iiot-rs485-b.dtso | 17 + .../dts/renesas/rzg2l-hummingboard-iiot.dtsi | 49 ++ .../boot/dts/renesas/rzg2l-sr-som-emmc.dtso | 44 ++ arch/arm64/boot/dts/renesas/rzg2l-sr-som.dtsi | 473 +++++++++++++++ 9 files changed, 1215 insertions(+) create mode 100644 arch/arm64/boot/dts/renesas/r9a07g044l2-hummingboard-iiot.dts create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-common.dtsi create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-microsd.dtso create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-rs485-a.dtso create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-rs485-b.dtso create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot.dtsi create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-sr-som-emmc.dtso create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-sr-som.dtsi diff --git a/arch/arm64/boot/dts/renesas/Makefile b/arch/arm64/boot/dts/renesas/Makefile index 10a13821622188..c5d34a0ef43605 100644 --- a/arch/arm64/boot/dts/renesas/Makefile +++ b/arch/arm64/boot/dts/renesas/Makefile @@ -176,6 +176,19 @@ dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-smarc-cru-csi-ov5645.dtbo r9a07g044c2-smarc-cru-csi-ov5645-dtbs := r9a07g044c2-smarc.dtb r9a07g044c2-smarc-cru-csi-ov5645.dtbo dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-smarc-cru-csi-ov5645.dtb +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-iiot.dtb +dtb-$(CONFIG_ARCH_R9A07G044) += rzg2l-sr-som-emmc.dtbo +r9a07g044l2-hummingboard-iiot-emmc-dtbs := r9a07g044l2-hummingboard-iiot.dtb rzg2l-sr-som-emmc.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-iiot-emmc.dtb +dtb-$(CONFIG_ARCH_R9A07G044) += rzg2l-hummingboard-iiot-microsd.dtbo +r9a07g044l2-hummingboard-iiot-microsd-dtbs := r9a07g044l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-microsd.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-iiot-microsd.dtb +dtb-$(CONFIG_ARCH_R9A07G044) += rzg2l-hummingboard-iiot-rs485-a.dtbo +r9a07g044l2-hummingboard-iiot-rs485-a-dtbs := r9a07g044l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-a.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-iiot-rs485-a.dtb +dtb-$(CONFIG_ARCH_R9A07G044) += rzg2l-hummingboard-iiot-rs485-b.dtbo +r9a07g044l2-hummingboard-iiot-rs485-b-dtbs := r9a07g044l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-b.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-iiot-rs485-b.dtb dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-remi-pi.dtb dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-smarc.dtb dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-smarc-cru-csi-ov5645.dtbo diff --git a/arch/arm64/boot/dts/renesas/r9a07g044l2-hummingboard-iiot.dts b/arch/arm64/boot/dts/renesas/r9a07g044l2-hummingboard-iiot.dts new file mode 100644 index 00000000000000..eba4f423c8f050 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/r9a07g044l2-hummingboard-iiot.dts @@ -0,0 +1,16 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + */ + +/dts-v1/; + +#include "r9a07g044l2.dtsi" +#include "rzg2l-sr-som.dtsi" +#include "rzg2l-hummingboard-iiot.dtsi" + +/ { + compatible = "solidrun,rzg2l-hummingboard-iiot", "solidrun,rzg2l-sr-som", + "renesas,r9a07g044l2", "renesas,r9a07g044"; + model = "SolidRun RZ/G2L HummingBoard IIoT"; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-common.dtsi b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-common.dtsi new file mode 100644 index 00000000000000..9f46c73ba3b5ac --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-common.dtsi @@ -0,0 +1,560 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + */ + +#include +#include +/ { + /* power for M.2 B-Key connector (J6) */ + regulator-m2-b { + compatible = "regulator-fixed"; + regulator-name = "m2-b"; + gpios = <&tca6416_u20 5 GPIO_ACTIVE_HIGH>; + regulator-always-on; + regulator-max-microvolt = <3300000>; + regulator-min-microvolt = <3300000>; + enable-active-high; + }; + + /* power for M.2 M-Key connector (J4) */ + regulator-m2-m { + compatible = "regulator-fixed"; + regulator-name = "m2-m"; + gpios = <&tca6416_u20 6 GPIO_ACTIVE_HIGH>; + regulator-always-on; + regulator-max-microvolt = <3300000>; + regulator-min-microvolt = <3300000>; + enable-active-high; + }; + + /* power for USB-A J27 behind USB Hub Port 3 */ + regulator-vbus-2 { + compatible = "regulator-fixed"; + regulator-name = "vbus2"; + regulator-always-on; + regulator-max-microvolt = <5000000>; + regulator-min-microvolt = <5000000>; + gpios = <&tca6416_u20 12 GPIO_ACTIVE_HIGH>; + enable-active-high; + }; + + /* power for USB-A J27 behind USB Hub Port 4 */ + regulator-vbus-3 { + compatible = "regulator-fixed"; + regulator-name = "vbus3"; + regulator-always-on; + regulator-max-microvolt = <5000000>; + regulator-min-microvolt = <5000000>; + gpios = <&tca6416_u20 13 GPIO_ACTIVE_HIGH>; + enable-active-high; + }; + + aliases { + i2c3 = &i2c_exp; + i2c4 = &i2c_csi; + i2c5 = &i2c_dsi; + i2c6 = &i2c_lvds; + rtc0 = &carrier_rtc; + rtc1 = &pmic; + serial3 = &scif3; + }; + + gpio-keys { + compatible = "gpio-keys"; + + wakeup-event { + interrupts-extended = <&tca6416_u21 11 IRQ_TYPE_EDGE_FALLING>; + label = "m2-m-wakeup"; + wakeup-source; + linux,code = ; + }; + }; + + can_mux: mux-controller-1 { + compatible = "gpio-mux"; + /* default J9-55/57/59/61 to on-board transceivers */ + idle-state = <0>; + #mux-control-cells = <0>; + /* + * Mux routes CAN bus signals between SoM connector pins, + * expansion connector (J22) and on-board transceivers using + * two GPIO: + * - IO3: 0 = on-board transceivers, 1 = expansion connector + * - IO4: 0 = J9-55/57/59/61, 1 = J7-12/16 & J9-54/56 + */ + mux-gpios = <&tca6416_u20 3 GPIO_ACTIVE_HIGH>, + <&tca6416_u20 4 GPIO_ACTIVE_HIGH>; + }; + + spi_mux: mux-controller-2 { + compatible = "gpio-mux"; + /* default on-board */ + idle-state = <0>; + /* + * Mux switches spi bus between on-board tpm + * and expansion connector (J22). + */ + mux-gpios = <&tca6416_u21 0 GPIO_ACTIVE_HIGH>; + #mux-control-cells = <0>; + }; + + scif1_scif3_b2b_mux: mux-controller-3 { + compatible = "gpio-mux"; + /* default on-board */ + idle-state = <0>; + #mux-control-cells = <0>; + /* + * Mux switches both scif1 and scif3 tx/rx between expansion + * connector (J22) and on-board rs232/rs485 transceivers + * using one GPIO: 0 = on-board, 1 = connector. + */ + mux-gpios = <&tca6416_u20 0 GPIO_ACTIVE_HIGH>; + }; + + scif1_rs_232_485_mux: mux-controller-4 { + compatible = "gpio-mux"; + /* default rs232 */ + idle-state = <0>; + #mux-control-cells = <0>; + /* + * Mux switches scif1 tx/rx between rs232 and rs485 + * transceivers. using one GPIO: 0 = rs232, 1 = rs485. + */ + mux-gpios = <&tca6416_u20 1 GPIO_ACTIVE_HIGH>; + }; + + scif3_rs_232_485_mux: mux-controller-5 { + compatible = "gpio-mux"; + /* default rs232 */ + idle-state = <0>; + #mux-control-cells = <0>; + /* + * Mux switches scif3 tx/rx between rs232 and rs485 + * transceivers. using one GPIO: 0 = rs232, 1 = rs485. + */ + mux-gpios = <&tca6416_u20 2 GPIO_ACTIVE_HIGH>; + }; + + v_1_2: regulator-1v2 { + compatible = "regulator-fixed"; + regulator-name = "1v2"; + regulator-max-microvolt = <1200000>; + regulator-min-microvolt = <1200000>; + }; + + v_3_3: regulator-3v3 { + compatible = "regulator-fixed"; + regulator-name = "3v3"; + regulator-max-microvolt = <3300000>; + regulator-min-microvolt = <3300000>; + }; + + reg_dsi_panel: regulator-dsi-panel { + compatible = "regulator-fixed"; + regulator-name = "dsi-panel"; + gpios = <&tca6416_u20 15 GPIO_ACTIVE_HIGH>; + regulator-max-microvolt = <11200000>; + regulator-min-microvolt = <11200000>; + enable-active-high; + }; + + vmmc: regulator-mmc { + compatible = "regulator-fixed"; + regulator-name = "vmmc"; + regulator-max-microvolt = <3300000>; + regulator-min-microvolt = <3300000>; + startup-delay-us = <250>; + vin-supply = <&v_3_3>; + gpios = <&pinctrl RZG2L_GPIO(4, 1) GPIO_ACTIVE_HIGH>; + enable-active-high; + }; + + /* power for USB-A J5003 */ + vbus1: regulator-vbus-1 { + compatible = "regulator-fixed"; + regulator-name = "vbus1"; + regulator-max-microvolt = <5000000>; + regulator-min-microvolt = <5000000>; + gpios = <&tca6416_u20 14 GPIO_ACTIVE_HIGH>; + enable-active-high; + }; + + rfkill-m2-b-gnss { + compatible = "rfkill-gpio"; + /* rfkill-gpio inverts internally */ + shutdown-gpios = <&tca6416_u20 10 GPIO_ACTIVE_LOW>; + label = "m2-b gnss"; + radio-type = "gps"; + }; + + rfkill-m2-b-wwan { + compatible = "rfkill-gpio"; + /* rfkill-gpio inverts internally */ + shutdown-gpios = <&tca6416_u20 9 GPIO_ACTIVE_HIGH>; + label = "m2-b radio"; + radio-type = "wwan"; + }; +}; + +&ehci1 { + #address-cells = <1>; + #size-cells = <0>; + + hub_2_0: hub@1 { + compatible = "usb4b4,6502", "usb4b4,6506"; + reg = <1>; + reset-gpios = <&tca6416_u20 11 GPIO_ACTIVE_LOW>; + vdd2-supply = <&v_3_3>; + vdd-supply = <&v_1_2>; + }; +}; + +&i2c0 { + /* highest i2c clock supported by all peripherals is 400kHz */ + + tca6416_u20: gpio@20 { + compatible = "ti,tcal6416"; + reg = <0x20>; + #gpio-cells = <2>; + gpio-controller; + gpio-line-names = "TCA_INT/EXT_UART", "TCA_UARTA_232/485", + "TCA_UARTB_232/485", "TCA_INT/EXT_CAN", + "TCA_NXP/REN", "TCA_M.2B_3V3_EN", + "TCA_M.2M_3V3_EN", "TCA_M.2M_RESET#", + "TCA_M.2B_RESET#", "TCA_M.2B_W_DIS#", + "TCA_M.2B_GPS_EN#", "TCA_USB-HUB_RST#", + "TCA_USB_HUB3_PWR_EN", "TCA_USB_HUB4_PWR_EN", + "TCA_USB1_PWR_EN", "TCA_VIDEO_PWR_EN"; + + m2-b-reset-hog { + gpios = <8 GPIO_ACTIVE_LOW>; + gpio-hog; + line-name = "m2-b-reset"; + output-low; + }; + + m2-m-reset-hog { + gpios = <7 GPIO_ACTIVE_LOW>; + gpio-hog; + line-name = "m2-m-reset"; + /* + * M.2 Key-M connector only supports PCI, + * but RZ/G2L(C) has no pci controller. + * Keep any card in reset. + */ + output-high; + }; + }; + + tca6416_u21: gpio@21 { + compatible = "ti,tcal6416"; + reg = <0x21>; + #interrupt-cells = <2>; + interrupt-controller; + #gpio-cells = <2>; + gpio-controller; + gpio-line-names = "TCA_SPI_TPM/EXT", "TCA_TPM_RST#", + "TCA_I2C_RST", "TCA_RS232_SHTD#", + "TCA_LCD_I2C_RST", "TCA_DIG_OUT1", + "TCA_bDIG_IN1", "TCA_SENS_INT", + "TCA_ALERT#", "TCA_TPM_PIRQ#", + "TCA_RTC_INT", "TCA_M.2M_WAKW_ON_LAN", + "TCA_M.2M_CLKREQ#", "TCA_LVDS_INT#", + "", "TCA_POE_AT"; + /* + * Level interrupts are typical for gpio expander but not supported on rz/g2l + * gpios. However since TCAL6416 latches interrupts and clears only on reading + * port register, falling edge trigger doesn't lose events. + */ + interrupts-extended = <&pinctrl RZG2L_GPIO(4, 0) IRQ_TYPE_EDGE_FALLING>; + + lcd-i2c-reset-hog { + gpios = <4 (GPIO_ACTIVE_LOW|GPIO_PULL_UP|GPIO_OPEN_DRAIN)>; + line-name = "lcd-i2c-reset"; + output-low; + /* + * reset shared between U37 and U48, to be + * supported once gpio-pca953x switches to + * reset framework. + */ + gpio-hog; + }; + + lvds-irq-hog { + gpios = <13 (GPIO_ACTIVE_LOW | GPIO_PULL_UP)>; + gpio-hog; + input; + line-name = "lvds-irq"; + }; + + m2-m-clkreq-hog { + gpios = <12 GPIO_ACTIVE_LOW>; + gpio-hog; + input; + line-name = "m2-m-clkreq"; + }; + + rs232_shutdown: rs232-shutdown-hog { + gpios = <3 GPIO_ACTIVE_LOW>; + gpio-hog; + line-name = "rs232-shutdown"; + output-low; + }; + + sensor-irq-hog { + gpios = <7 (GPIO_ACTIVE_LOW | GPIO_PULL_UP)>; + gpio-hog; + input; + line-name = "sensor-irq"; + }; + + tpm-irq-hog { + gpios = <9 (GPIO_ACTIVE_LOW | GPIO_PULL_UP)>; + gpio-hog; + input; + line-name = "tpm-irq"; + }; + }; + + led-controller@30 { + compatible = "ti,lp5562"; + reg = <0x30>; + #address-cells = <1>; + #size-cells = <0>; + /* use internal clock, could use external generated by rtc */ + clock-mode = /bits/ 8 <1>; + + multi-led@0 { + reg = <0x0>; + #address-cells = <1>; + #size-cells = <0>; + color = ; + label = "D7"; + + led@0 { + reg = <0x0>; + color = ; + led-cur = /bits/ 8 <0x32>; + max-cur = /bits/ 8 <0x64>; + }; + + led@1 { + reg = <0x1>; + color = ; + led-cur = /bits/ 8 <0x19>; + max-cur = /bits/ 8 <0x32>; + }; + + led@2 { + reg = <0x2>; + color = ; + led-cur = /bits/ 8 <0x19>; + max-cur = /bits/ 8 <0x32>; + }; + }; + + led@3 { + reg = <0x3>; + chan-name = "D8"; + color = ; + label = "D8"; + led-cur = /bits/ 8 <0x19>; + max-cur = /bits/ 8 <0x64>; + }; + }; + + light-sensor@44 { + compatible = "isil,isl29023"; + reg = <0x44>; + /* IRQ shared between accelerometer, light-sensor and Tamper input (J5007) */ + interrupts-extended = <&tca6416_u21 7 IRQ_TYPE_LEVEL_LOW>; + }; + + accelerometer@53 { + compatible = "adi,adxl345"; + reg = <0x53>; + interrupts-extended = <&tca6416_u21 7 IRQ_TYPE_LEVEL_LOW>; + /* IRQ shared between accelerometer, light-sensor and Tamper input (J5007) */ + interrupt-names = "INT1"; + }; + + carrier_eeprom: eeprom@57 { + compatible = "atmel,24c02"; + reg = <0x57>; + pagesize = <8>; + }; + + carrier_rtc: rtc@69 { + compatible = "abracon,ab1805"; + reg = <0x69>; + /* + * AM1805 RTC used on this board has only nTIRQ pins wired, + * which is for countdown timer irqs only. + * Driver does not support this, disable for now. + * + * interrupts-extended = <&tca6416_u21 10 IRQ_TYPE_LEVEL_LOW>; + */ + abracon,tc-diode = "schottky"; + abracon,tc-resistor = <3>; + }; +}; + +&i2c1 { + /* highest i2c clock supported by all peripherals is 400kHz */ + + i2c-mux@70 { + compatible = "nxp,pca9546"; + reg = <0x70>; + #address-cells = <1>; + #size-cells = <0>; + /* + * This reset is open drain with HW 10k pull-up resistor, + * but reset core does not support GPIO_OPEN_DRAIN flag. + * The pull-up and GPIO voltages match, push-pull is safe. + */ + reset-gpios = <&tca6416_u21 2 GPIO_ACTIVE_LOW>; + + /* channel 0 routed to expansion connector (J22) */ + i2c_exp: i2c@0 { + reg = <0>; + #address-cells = <1>; + #size-cells = <0>; + }; + + /* channel 1 routed to mipi-csi connector (J23) */ + i2c_csi: i2c@1 { + reg = <1>; + #address-cells = <1>; + #size-cells = <0>; + }; + + /* channel 2 routed to mipi-dsi connector (J25) */ + i2c_dsi: i2c@2 { + reg = <2>; + #address-cells = <1>; + #size-cells = <0>; + + tca6408_u48: gpio@21 { + compatible = "ti,tca6408"; + reg = <0x21>; + #gpio-cells = <2>; + gpio-line-names = "CAM_RST#", "DSI_RESET", + "DSI_STBYB", "DSI_PWM_BL", + "DSI_L/R", "DSI_U/D", + "DSI_CTP_/RST", "CAM_TRIG"; + /* + * reset shared between U37 and U48, to be + * supported once gpio-pca953x switches to + * reset framework. + * + * reset-gpios = <&tca6416_u21 4 + * (GPIO_ACTIVE_LOW|GPIO_PULL_UP|GPIO_OPEN_DRAIN)>; + */ + gpio-controller; + }; + }; + + /* channel 3 routed to lvds connector (J24) */ + i2c_lvds: i2c@3 { + reg = <3>; + #address-cells = <1>; + #size-cells = <0>; + + tca6408_u37: gpio@20 { + compatible = "ti,tca6408"; + reg = <0x20>; + #gpio-cells = <2>; + gpio-line-names = "SELB", "LVDS_RESET", + "LVDS_STBYB", "LVDS_PWM_BL", + "LVDS_L/R", "LVDS_U/D", + "LVDS_CTP_/RST", ""; + /* + * reset shared between U37 and U48, to be + * supported once gpio-pca953x switches to + * reset framework. + * + * reset-gpios = <&tca6416_u21 4 + * (GPIO_ACTIVE_LOW|GPIO_PULL_UP|GPIO_OPEN_DRAIN)>; + */ + gpio-controller; + }; + }; + }; +}; + +&ohci0 { + dr_mode = "host"; +}; + +&phy0 { + leds { + #address-cells = <1>; + #size-cells = <0>; + + /* LED_0 pin */ + led@0 { + reg = <0>; + color = ; + default-state = "keep"; + function = LED_FUNCTION_LAN; + }; + }; +}; + +&pinctrl { + /* UARTA */ + scif1_pins: scif1 { + pinmux = , /* SCIF1_RXD */ + ; /* SCIF1_TXD */ + }; + + /* UARTB */ + scif3_pins: scif3 { + pinmux = , /* SCIF3_RXD */ + ; /* SCIF3_TXD */ + }; +}; + +&scif1 { + pinctrl-0 = <&scif1_pins>; + pinctrl-names = "default"; + status = "okay"; +}; + +&scif3 { + pinctrl-0 = <&scif3_pins>; + pinctrl-names = "default"; + status = "okay"; +}; + +&spi1 { + /* native cs does not support cs persistence required for tpm */ + cs-gpios = <&pinctrl RZG2L_GPIO(44, 3) GPIO_ACTIVE_LOW>; + num-cs = <1>; + pinctrl-0 = <&spi1_pins>; + pinctrl-names = "default"; + status = "okay"; + + spi1_muxed: spi@0 { + compatible = "spi-mux"; + reg = <0>; + #address-cells = <1>; + #size-cells = <0>; + mux-controls = <&spi_mux>; + /* mux bandwidth is 2GHz, soc max. spi clock is P0/2 = 50MHz */ + spi-max-frequency = <50000000>; + + tpm@0 { + compatible = "infineon,slb9670", "tcg,tpm_tis-spi"; + reg = <0>; + interrupts-extended = <&tca6416_u21 9 IRQ_TYPE_LEVEL_LOW>; + reset-gpios = <&tca6416_u21 1 (GPIO_ACTIVE_LOW | GPIO_OPEN_DRAIN)>; + spi-max-frequency = <43000000>; + }; + }; +}; + +&usb2_phy0 { + vbus-supply = <&vbus1>; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-microsd.dtso b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-microsd.dtso new file mode 100644 index 00000000000000..0f92a9d2ccd98f --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-microsd.dtso @@ -0,0 +1,26 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Device Tree Overlay for the RZ/G2L(C) Solidrun SOM SD + * + * Copyright (C) 2024 SolidRun Ltd. + */ + +/dts-v1/; +/plugin/; + +#include +#include + +&sdhi0 { + bus-width = <4>; + full-pwr-cycle; + mux-states = <&sdhi0_mux 1>; + pinctrl-0 = <&sdhi0_4bit_pins>, <&sdhi0_cd_pins>; + pinctrl-1 = <&sdhi0_4bit_uhs_pins>, <&sdhi0_cd_pins>; + pinctrl-names = "default", "state_uhs"; + sd-uhs-sdr104; + sd-uhs-sdr50; + vmmc-supply = <&vmmc>; + vqmmc-supply = <®_pmic_ldo1>; + status = "okay"; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-rs485-a.dtso b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-rs485-a.dtso new file mode 100644 index 00000000000000..4bcb22d518f053 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-rs485-a.dtso @@ -0,0 +1,17 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + * + * Overlay for enabling HummingBoard IIoT on-board RS485 Port A on connector J5004. + * + * Because Renesas uart driver does not support rs485, + * users must manually toggle P41_1 between RX & TX. + */ + +/dts-v1/; +/plugin/; + +&scif1_rs_232_485_mux { + /* select rs485 */ + idle-state = <1>; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-rs485-b.dtso b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-rs485-b.dtso new file mode 100644 index 00000000000000..6f460b3e0b256f --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-rs485-b.dtso @@ -0,0 +1,17 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + * + * Overlay for enabling HummingBoard IIoT on-board RS485 Port B on connector J5004. + * + * Because Renesas uart driver does not support rs485, + * users must manually toggle P41_0 between RX & TX. + */ + +/dts-v1/; +/plugin/; + +&scif3_rs_232_485_mux { + /* select rs485 */ + idle-state = <1>; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot.dtsi b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot.dtsi new file mode 100644 index 00000000000000..22f066079e69ab --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot.dtsi @@ -0,0 +1,49 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + */ + +#include "rzg2l-hummingboard-iiot-common.dtsi" + +&canfd { + pinctrl-names = "default"; + pinctrl-0 = <&can0_pins>, <&can1_pins>; + status = "okay"; + + channel0 { + status = "okay"; + }; + + channel1 { + status = "okay"; + }; +}; + +&phy1 { + leds { + #address-cells = <1>; + #size-cells = <0>; + + /* LED_0 pin */ + led@0 { + reg = <0>; + color = ; + function = LED_FUNCTION_LAN; + default-state = "keep"; + }; + }; +}; + +&pinctrl { + /* CANA */ + can0_pins: can0 { + pinmux = , /* CAN0_TX */ + ; /* CAN0_RX */ + }; + + /* CANB */ + can1_pins: can1 { + pinmux = , /* CAN1_TX */ + ; /* CAN1_RX */ + }; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-sr-som-emmc.dtso b/arch/arm64/boot/dts/renesas/rzg2l-sr-som-emmc.dtso new file mode 100644 index 00000000000000..92e3406be7dfc4 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2l-sr-som-emmc.dtso @@ -0,0 +1,44 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Device Tree Overlay for the RZ/G2L(C) Solidrun SOM eMMC + * + * Copyright (C) 2024 SolidRun Ltd. + */ + +/dts-v1/; +/plugin/; + +#include +#include + +®_pmic_ldo1 { + /* + * This ldo can switch mmc host controller io voltage between + * 1.8V and 3.3V. The eMMC IO voltage however is supplied from + * reg_pmic_buck3 which is fixed at 1.8V. + * Lower this ldo maximum voltage to 1.8V to prevent setting 3.3V. + */ + regulator-max-microvolt = <1800000>; +}; + +&sdhi0 { + /* + * Host controller and eMMC have separate io voltage regulators: + * reg_pmic_ldo1 (1.8V/3.3V); reg_pmic_buck3 (1.8V only). + * Link to the switchable regulator ensuring that it gets configured. + */ + vqmmc-supply = <®_pmic_ldo1>; + bus-width = <8>; + cap-mmc-hw-reset; + mmc-hs200-1_8v; + mux-states = <&sdhi0_mux 0>; + non-removable; + no-sdio; + pinctrl-0 = <&sdhi0_8bit_uhs_pins>, <&sdhi0_rst_pins>; + pinctrl-1 = <&sdhi0_8bit_uhs_pins>, <&sdhi0_rst_pins>; + pinctrl-names = "default", "state_uhs"; + vmmc-supply = <®_pmic_buck4>; + /* emmc io voltage is hard-wired for 1.8V, disable sd modes */ + no-sd; + status = "okay"; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-sr-som.dtsi b/arch/arm64/boot/dts/renesas/rzg2l-sr-som.dtsi new file mode 100644 index 00000000000000..c1dcc5ea430aa0 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2l-sr-som.dtsi @@ -0,0 +1,473 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Device Tree Source for the RZ/G2L Solidrun SoM + * + * Copyright 2023 SolidRun Ltd. + * Copyright 2025 Josua Mayer + */ + +#include +#include + +/ { + aliases { + ethernet0 = ð0; + ethernet1 = ð1; + i2c0 = &i2c0; + i2c1 = &i2c1; + i2c2 = &i2c3; + mmc0 = &sdhi0; + mmc1 = &sdhi1; + rtc0 = &pmic; + serial0 = &scif0; + serial1 = &scif1; + serial2 = &scif2; + }; + + chosen { + stdout-path = "serial0:115200n8"; + }; + + memory@48000000 { + /* + * Physical RAM starts at 0x40000000, but Renesas RZ/G2 BSPs + * reserve and hide the first 128MB for TrustZone by default. + * + * Minimum size for this SoM is 1GB, and the bootloader patches + * the dtb according to the actual RAM size and reservations. + */ + reg = <0x0 0x48000000 0x0 0x38000000>; + device_type = "memory"; + }; + + sdhi0_mux: mux-controller-0 { + compatible = "gpio-mux"; + #mux-control-cells = <0>; + #mux-state-cells = <1>; + /* + * Mux switches SD0_DATA[0-3], SD0_CMD & SD0_CLK between + * on-SoM eMMC and board-to-board connector using one gpio: + * Low = connector, High = eMMC. + * Define active-low for logical mux state 0 = eMMC, 1 = connector. + */ + mux-gpios = <&pinctrl RZG2L_GPIO(22, 1) GPIO_ACTIVE_LOW>; + }; + + reg_pmic_buck1: regulator-pmic-buck1 { + compatible = "regulator-fixed"; + regulator-name = "pmic-buck1"; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <1100000>; + regulator-min-microvolt = <1100000>; + }; + + reg_pmic_buck3: regulator-pmic-buck3 { + compatible = "regulator-fixed"; + regulator-name = "pmic-buck3"; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <1800000>; + regulator-min-microvolt = <1800000>; + }; + + reg_pmic_buck4: regulator-pmic-buck4 { + compatible = "regulator-fixed"; + regulator-name = "pmic-buck4"; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <3300000>; + regulator-min-microvolt = <3300000>; + }; + + reg_pmic_ldo1: regulator-pmic-ldo1 { + compatible = "regulator-gpio"; + regulator-name = "pmic-ldo1"; + gpios = <&pinctrl RZG2L_GPIO(39, 0) GPIO_ACTIVE_HIGH>; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <3300000>; + regulator-min-microvolt = <1800000>; + states = <3300000 1>, <1800000 0>; + }; + + reg_pmic_ldo2: regulator-pmic-ldo2 { + compatible = "regulator-fixed"; + regulator-name = "pmic-ldo2"; + /* + * This ldo can switch mmc host controller io voltage between + * 1.8V and 3.3V by assembly option of pull-up / pull-dow. + * Default assembly is 3.3V. + */ + regulator-min-microvolt = <3300000>; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <3300000>; + }; + + reserved-memory { + ranges; + #address-cells = <2>; + #size-cells = <2>; + + global_cma: linux,cma@58000000 { + compatible = "shared-dma-pool"; + reg = <0x0 0x58000000 0x0 0x10000000>; + reusable; + linux,cma-default; + }; + }; + + sdhi1_pwrseq: sdhi1-pwrseq { + compatible = "mmc-pwrseq-simple"; + reset-gpios = <&pinctrl RZG2L_GPIO(23, 1) GPIO_ACTIVE_LOW>; + }; + + /* 32.768kHz crystal */ + x2: x2-clock { + compatible = "fixed-clock"; + #clock-cells = <0>; + clock-frequency = <32768>; + }; +}; + +&ehci0 { + status = "okay"; +}; + +&ehci1 { + status = "okay"; +}; + +ð0 { + phy-handle = <&phy0>; + pinctrl-0 = <ð0_pins>; + pinctrl-names = "default"; + /* + * ravb driver does not configure mac internal delays for RZ/G2L(C), + * instead delays are added by the MxL86110 phy driver. + */ + phy-mode = "rgmii-id"; + status = "okay"; + + phy0: ethernet-phy@0 { + reg = <0>; + /* Level interrupts are typical for phy but not supported on RZ/G2L gpios. */ + interrupts-extended = <&pinctrl RZG2L_GPIO(27, 0) IRQ_TYPE_EDGE_FALLING>; + }; +}; + +ð1 { + phy-handle = <&phy1>; + pinctrl-0 = <ð1_pins>; + pinctrl-names = "default"; + /* + * ravb driver does not configure mac internal delays for RZ/G2L(C), + * instead delays are added by the MxL86110 phy driver. + */ + phy-mode = "rgmii-id"; + status = "okay"; + + phy1: ethernet-phy@4 { + reg = <4>; + /* Level interrupts are typical for phy but not supported on RZ/G2L gpios. */ + interrupts-extended = <&pinctrl RZG2L_GPIO(42, 4) IRQ_TYPE_EDGE_FALLING>; + }; +}; + +&extal_clk { + clock-frequency = <24000000>; +}; + +&gpu { + mali-supply = <®_pmic_buck1>; +}; + +&i2c0 { + clock-frequency = <400000>; + pinctrl-0 = <&i2c0_pins>; + pinctrl-names = "default"; + status = "okay"; +}; + +&i2c1 { + clock-frequency = <400000>; + pinctrl-0 = <&i2c1_pins>; + pinctrl-names = "default"; + status = "okay"; + + eeprom: eeprom@50 { + compatible = "atmel,24c01"; + reg = <0x50>; + pagesize = <16>; + }; +}; + +&i2c3 { + clock-frequency = <400000>; + pinctrl-0 = <&i2c3_pins>; + pinctrl-names = "default"; + status = "okay"; + + pmic: pmic@12 { + compatible = "renesas,raa215300"; + reg = <0x12>, <0x6f>; + reg-names = "main", "rtc"; + clocks = <&x2>; + clock-names = "xin"; + }; +}; + +&ohci0 { + status = "okay"; +}; + +&ohci1 { + status = "okay"; +}; + +&ostm1 { + status = "okay"; +}; + +&ostm2 { + status = "okay"; +}; + +&phyrst { + status = "okay"; +}; + +&pinctrl { + eth0_pins: eth0 { + pinmux = , /* ET0_LINKSTA */ + , /* ET0_MDC */ + , /* ET0_MDIO */ + , /* ET0_TXC */ + , /* ET0_TX_CTL */ + , /* ET0_TXD0 */ + , /* ET0_TXD1 */ + , /* ET0_TXD2 */ + , /* ET0_TXD3 */ + , /* ET0_RXC */ + , /* ET0_RX_CTL */ + , /* ET0_RXD0 */ + , /* ET0_RXD1 */ + , /* ET0_RXD2 */ + ; /* ET0_RXD3 */ + }; + + eth1_pins: eth1 { + pinmux = , /* ET1_LINKSTA */ + , /* ET1_MDC */ + , /* ET1_MDIO */ + , /* ET1_TXC */ + , /* ET1_TX_CTL */ + , /* ET1_TXD0 */ + , /* ET1_TXD1 */ + , /* ET1_TXD2 */ + , /* ET1_TXD3 */ + , /* ET1_RXC */ + , /* ET1_RX_CTL */ + , /* ET1_RXD0 */ + , /* ET1_RXD1 */ + , /* ET1_RXD2 */ + ; /* ET1_RXD3 */ + }; + + i2c0_pins: i2c0 { + input-enable; + pins = "RIIC0_SDA", "RIIC0_SCL"; + }; + + i2c1_pins: i2c1 { + input-enable; + pins = "RIIC1_SDA", "RIIC1_SCL"; + }; + + i2c3_pins: i2c3 { + pinmux = , /* RIIC3_SDA */ + ; /* RIIC3_SCL */ + }; + + qspi0_pins: qspi0 { + pins = "QSPI0_IO0", "QSPI0_IO1", "QSPI0_IO2", "QSPI0_IO3", + "QSPI0_SPCLK", "QSPI0_SSL"; + power-source = <1800>; + }; + + scif0_pins: scif0 { + pinmux = , /* SCIF0_TXD */ + ; /* SCIF0_RXD */ + }; + + scif2_pins: scif2 { + pinmux = , /* SCIF2_TXD */ + , /* SCIF2_RXD */ + , /* SCIF2_CTS# */ + ; /* SCIF2_RTS# */ + }; + + sdhi0_4bit_pins: sdhi0-4bit { + sd0_ctrl_dat03 { + pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3", + "SD0_CLK", "SD0_CMD"; + power-source = <3300>; + }; + + /* + * Pins 4-7 are hard-wired to eMMC with 1.8V IO voltage, + * with all pins sharing a single voltage domain. + * + * These pins are fixed-function lacking independently configurable + * pull-up, pull-down or output-disable. + * + * While unused, disable the input buffers to avoid sampling the still + * physically connected but unused eMMC lines. + * For output direction rely on correct controller bus-width configuration + * to avoid over-voltage to eMMC e.g. when operating at 3.3V for a 4-bit microSD. + */ + sd0_dat47 { + input-disable; + pins = "SD0_DATA4", "SD0_DATA5", "SD0_DATA6", "SD0_DATA7"; + }; + }; + + sdhi0_4bit_uhs_pins: sdhi0-4bit-uhs { + sd0_ctrl_dat03 { + pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3", + "SD0_CLK", "SD0_CMD"; + power-source = <1800>; + }; + + /* + * Pins 4-7 are hard-wired to eMMC with 1.8V IO voltage, + * with all pins sharing a single voltage domain. + * + * These pins are fixed-function, lacking configurable pull-up, pull-down or + * output-disable, while sharing the voltage domain with all SD0_* pins. + * + * While unused, disable the input buffers to avoid sampling the still + * physically connected but unused eMMC lines. + * For output direction rely on correct controller bus-width configuration + * to avoid over-voltage to eMMC e.g. when operating at 3.3V for a 4-bit microSD. + */ + sd0_dat47 { + input-disable; + pins = "SD0_DATA4", "SD0_DATA5", "SD0_DATA6", "SD0_DATA7"; + }; + }; + + sdhi0_8bit_uhs_pins: sdhi0-8bit-uhs { + sd0_ctrl { + pins = "SD0_CLK", "SD0_CMD"; + power-source = <1800>; + }; + + sd0_dat { + input-enable; + pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3", + "SD0_DATA4", "SD0_DATA5", "SD0_DATA6", "SD0_DATA7"; + }; + }; + + sdhi0_cd_pins: sdhi0-cd { + pinmux = ; /* SD0_CD */ + }; + + /* SD0_RST is only routed to eMMC which uses fixed 1.8V IO voltage */ + sdhi0_rst_pins: sdhi0-rst { + pins = "SD0_RST#"; + power-source = <1800>; + }; + + sdhi1_pins: sdhi1 { + pins = "SD1_DATA0", "SD1_DATA1", "SD1_DATA2", "SD1_DATA3", + "SD1_CLK", "SD1_CMD"; + power-source = <3300>; + }; + + spi1_pins: spi1 { + pinmux = , /* RSPI1_MISO */ + , /* RSPI1_MOSI */ + ; /* RSPI1_CK */ + }; + + spi1_cs_pins: spi1-cs { + pinmux = ; /* RSPI1_SSL */ + }; +}; + +&sbc { + pinctrl-0 = <&qspi0_pins>; + pinctrl-names = "default"; + status = "okay"; + + flash@0 { + compatible = "winbond,w25q80bl", "jedec,spi-nor"; + reg = <0>; + spi-max-frequency = <50000000>; + m25p,fast-read; + }; +}; + +&scif0 { + pinctrl-0 = <&scif0_pins>; + pinctrl-names = "default"; + status = "okay"; +}; + +&scif2 { + pinctrl-0 = <&scif2_pins>; + pinctrl-names = "default"; + uart-has-rtscts; + status = "okay"; + + bluetooth { + /* + * The BT is based on BCM4345C0. Murata 1MW variant seems to + * require driving RTS on open. + * Use compatible bcm43438 to accomplish this. + */ + compatible = "brcm,bcm43438-bt"; + max-speed = <115200>; + shutdown-gpios = <&pinctrl RZG2L_GPIO(23, 0) GPIO_ACTIVE_HIGH>; + vbat-supply = <®_pmic_buck4>; + vddio-supply = <®_pmic_buck4>; + }; +}; + +/* WiFi */ +&sdhi1 { + /* Murata 1MW max rate is 50MHz */ + max-frequency = <50000000>; + bus-width = <4>; + cap-sdio-irq; + mmc-pwrseq = <&sdhi1_pwrseq>; + non-removable; + no-1-8-v; + no-sd; + pinctrl-0 = <&sdhi1_pins>; + pinctrl-names = "default"; + vmmc-supply = <®_pmic_buck4>; + /* + * Host controller IO voltage is provided from reg_pmic_ldo2, + * WiFi module IO voltage from reg_pmic_buck4. + * Neither is configurable at run-time so either can be set here. + */ + vqmmc-supply = <®_pmic_ldo2>; + status = "okay"; +}; + +&usb2_phy0 { + vbus-supply = <&usb0_vbus_otg>; + status = "okay"; +}; + +&usb2_phy1 { + status = "okay"; +}; + +&wdt0 { + status = "okay"; +}; From 823703cb5e3a6825daf8ce5dbfc03aeafbc5b688 Mon Sep 17 00:00:00 2001 From: Josua Mayer Date: Mon, 28 Sep 2026 19:33:56 +0200 Subject: [PATCH 0179/1012] arm64: dts: renesas: Add support for solidrun rzv2l som and hb-iiot evb Add support for the SolidRun RZ/V2L [1] SoM on HummingBoard IIoT [2]. The SoM features: - 2x 1Gbps Ethernet with PHY - eMMC - 1/2GB DDR - WiFi + Bluetooth - SDHI Mux switching between eMMC and Carrier Board The HummingBoard IIoT features: - 3x USB-2.0 Type A connector - 2x 1Gbps RJ45 Ethernet - USB Type-C Console Port - microSD connector - RTC with backup battery - RGB Status LED - 1x M.2 B-Key connector with USB-2.0 + SIM card holder - 1x DSI Display Connector - GPIO header - 2x RS232/RS485 ports (configurable) - 2x CAN The RZ/V2L SoM shares PCB with RZ/G2L, differing only in the SoC itself. RZ/V2L is adding a powerful DRP-AI NPU which G2L lacks. Due to the similarities most code is shared, including DT overlays for eMMC, microSD, and RS485. [1] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/rz-v2l-som/ [2] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/hummingboard-rz-series-sbcs/hummingboard-rz-g2l-iot-sbc/ Reviewed-by: Geert Uytterhoeven Signed-off-by: Josua Mayer Link: https://patch.msgid.link/20260928-rzg2-sr-boards-v10-5-91752ea39cbc@solid-run.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/Makefile | 13 +++++++++++++ .../renesas/r9a07g054l2-hummingboard-iiot.dts | 16 ++++++++++++++++ 2 files changed, 29 insertions(+) create mode 100644 arch/arm64/boot/dts/renesas/r9a07g054l2-hummingboard-iiot.dts diff --git a/arch/arm64/boot/dts/renesas/Makefile b/arch/arm64/boot/dts/renesas/Makefile index c5d34a0ef43605..31be50bbb205bb 100644 --- a/arch/arm64/boot/dts/renesas/Makefile +++ b/arch/arm64/boot/dts/renesas/Makefile @@ -195,6 +195,19 @@ dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-smarc-cru-csi-ov5645.dtbo r9a07g044l2-smarc-cru-csi-ov5645-dtbs := r9a07g044l2-smarc.dtb r9a07g044l2-smarc-cru-csi-ov5645.dtbo dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-smarc-cru-csi-ov5645.dtb +dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-iiot.dtb +dtb-$(CONFIG_ARCH_R9A07G054) += rzg2l-sr-som-emmc.dtbo +r9a07g054l2-hummingboard-iiot-emmc-dtbs := r9a07g054l2-hummingboard-iiot.dtb rzg2l-sr-som-emmc.dtbo +dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-iiot-emmc.dtb +dtb-$(CONFIG_ARCH_R9A07G054) += rzg2l-hummingboard-iiot-microsd.dtbo +r9a07g054l2-hummingboard-iiot-microsd-dtbs := r9a07g054l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-microsd.dtbo +dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-iiot-microsd.dtb +dtb-$(CONFIG_ARCH_R9A07G054) += rzg2l-hummingboard-iiot-rs485-a.dtbo +r9a07g054l2-hummingboard-iiot-rs485-a-dtbs := r9a07g054l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-a.dtbo +dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-iiot-rs485-a.dtb +dtb-$(CONFIG_ARCH_R9A07G054) += rzg2l-hummingboard-iiot-rs485-b.dtbo +r9a07g054l2-hummingboard-iiot-rs485-b-dtbs := r9a07g054l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-b.dtbo +dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-iiot-rs485-b.dtb dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-smarc.dtb dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-smarc-cru-csi-ov5645.dtbo r9a07g054l2-smarc-cru-csi-ov5645-dtbs := r9a07g054l2-smarc.dtb r9a07g054l2-smarc-cru-csi-ov5645.dtbo diff --git a/arch/arm64/boot/dts/renesas/r9a07g054l2-hummingboard-iiot.dts b/arch/arm64/boot/dts/renesas/r9a07g054l2-hummingboard-iiot.dts new file mode 100644 index 00000000000000..d77a6ff163bea1 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/r9a07g054l2-hummingboard-iiot.dts @@ -0,0 +1,16 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + */ + +/dts-v1/; + +#include "r9a07g054l2.dtsi" +#include "rzg2l-sr-som.dtsi" +#include "rzg2l-hummingboard-iiot.dtsi" + +/ { + compatible = "solidrun,rzv2l-hummingboard-iiot", "solidrun,rzv2l-sr-som", + "renesas,r9a07g054l2", "renesas,r9a07g054"; + model = "SolidRun RZ/V2L HummingBoard IIoT"; +}; From 17ed07aa3b1222da7c33635cf47c4832d6c8f910 Mon Sep 17 00:00:00 2001 From: Josua Mayer Date: Mon, 28 Sep 2026 19:33:57 +0200 Subject: [PATCH 0180/1012] arm64: dts: renesas: Add support for solidrun rzg2lc som and hb-iiot evb Add support for the SolidRun RZ/G2LC SoM [1] on HummingBoard IIoT [2]. The SoM features: - 100Mbps Ethernet with PHY - eMMC - 1/2GB DDR - WiFi + Bluetooth - SDHI Mux switching between eMMC and Carrier Board The HummingBoard IIoT features: - 3x USB-2.0 Type A connector - 1x 100Mbps RJ45 Ethernet - USB Type-C Console Port - microSD connector - RTC with backup battery - RGB Status LED - 1x M.2 B-Key connector with USB-2.0 + SIM card holder - 1x DSI Display Connector - GPIO header - 2x RS232/RS485 ports (configurable) The RZ/G2LC SoM was designed to be pin compatible to G2L SoM, with slightly reduced feature set. Descriptions for eMMC, microSD, and RS485 are shared with G2L. [1] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/rz-g2lc-som/ [2] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/hummingboard-rz-series-sbcs/hummingboard-rz-g2l-iot-sbc/ Reviewed-by: Geert Uytterhoeven Signed-off-by: Josua Mayer Link: https://patch.msgid.link/20260928-rzg2-sr-boards-v10-6-91752ea39cbc@solid-run.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/Makefile | 9 + .../renesas/r9a07g044c2-hummingboard-iiot.dts | 20 + .../arm64/boot/dts/renesas/rzg2lc-sr-som.dtsi | 436 ++++++++++++++++++ 3 files changed, 465 insertions(+) create mode 100644 arch/arm64/boot/dts/renesas/r9a07g044c2-hummingboard-iiot.dts create mode 100644 arch/arm64/boot/dts/renesas/rzg2lc-sr-som.dtsi diff --git a/arch/arm64/boot/dts/renesas/Makefile b/arch/arm64/boot/dts/renesas/Makefile index 31be50bbb205bb..4296f893eea46d 100644 --- a/arch/arm64/boot/dts/renesas/Makefile +++ b/arch/arm64/boot/dts/renesas/Makefile @@ -171,6 +171,15 @@ dtb-$(CONFIG_ARCH_R9A07G043) += r9a07g043u11-smarc-du-adv7513.dtb r9a07g043u11-smarc-pmod-dtbs := r9a07g043u11-smarc.dtb r9a07g043-smarc-pmod.dtbo dtb-$(CONFIG_ARCH_R9A07G043) += r9a07g043u11-smarc-pmod.dtb +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-iiot.dtb +r9a07g044c2-hummingboard-iiot-emmc-dtbs := r9a07g044c2-hummingboard-iiot.dtb rzg2l-sr-som-emmc.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-iiot-emmc.dtb +r9a07g044c2-hummingboard-iiot-microsd-dtbs := r9a07g044c2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-microsd.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-iiot-microsd.dtb +r9a07g044c2-hummingboard-iiot-rs485-a-dtbs := r9a07g044c2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-a.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-iiot-rs485-a.dtb +r9a07g044c2-hummingboard-iiot-rs485-b-dtbs := r9a07g044c2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-b.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-iiot-rs485-b.dtb dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-smarc.dtb dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-smarc-cru-csi-ov5645.dtbo r9a07g044c2-smarc-cru-csi-ov5645-dtbs := r9a07g044c2-smarc.dtb r9a07g044c2-smarc-cru-csi-ov5645.dtbo diff --git a/arch/arm64/boot/dts/renesas/r9a07g044c2-hummingboard-iiot.dts b/arch/arm64/boot/dts/renesas/r9a07g044c2-hummingboard-iiot.dts new file mode 100644 index 00000000000000..9cc21ae32ed4db --- /dev/null +++ b/arch/arm64/boot/dts/renesas/r9a07g044c2-hummingboard-iiot.dts @@ -0,0 +1,20 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + */ + +/dts-v1/; + +#include "r9a07g044c2.dtsi" +#include "rzg2lc-sr-som.dtsi" +#include "rzg2l-hummingboard-iiot-common.dtsi" + +/ { + compatible = "solidrun,rzg2lc-hummingboard-iiot", "solidrun,rzg2lc-sr-som", + "renesas,r9a07g044c2", "renesas,r9a07g044"; + model = "SolidRun RZ/G2LC HummingBoard IIoT"; +}; + +&vmmc { + gpios = <&pinctrl RZG2L_GPIO(18, 1) GPIO_ACTIVE_HIGH>; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2lc-sr-som.dtsi b/arch/arm64/boot/dts/renesas/rzg2lc-sr-som.dtsi new file mode 100644 index 00000000000000..bf3d9e8ed109cd --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2lc-sr-som.dtsi @@ -0,0 +1,436 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Device Tree Source for the RZ/G2LC Solidrun SOM + * + * Copyright 2023 SolidRun Ltd. + * Copyright 2025 Josua Mayer + */ + +#include +#include + +/ { + aliases { + ethernet0 = ð0; + i2c0 = &i2c0; + i2c1 = &i2c1; + i2c2 = &i2c2; + mmc0 = &sdhi0; + mmc1 = &sdhi1; + rtc0 = &pmic; + serial0 = &scif0; + serial1 = &scif1; + serial2 = &scif2; + }; + + chosen { + stdout-path = "serial0:115200n8"; + }; + + memory@48000000 { + /* + * Physical RAM starts at 0x40000000, but Renesas RZ/G2 BSPs + * reserve and hide the first 128MB for TrustZone by default. + * + * Minimum size for this SoM is 1GB, and the bootloader patches + * the dtb according to the actual RAM size and reservations. + */ + reg = <0x0 0x48000000 0x0 0x38000000>; + device_type = "memory"; + }; + + sdhi0_mux: mux-controller-0 { + compatible = "gpio-mux"; + #mux-control-cells = <0>; + #mux-state-cells = <1>; + /* + * Mux switches SD0_DATA[0-3], SD0_CMD & SD0_CLK between + * on-SoM eMMC and board-to-board connector using one gpio: + * High = connector, Low = eMMC. + * Define active-high for logical mux state 0 = eMMC, 1 = connector. + */ + mux-gpios = <&pinctrl RZG2L_GPIO(22, 1) GPIO_ACTIVE_HIGH>; + }; + + reg_pmic_buck1: regulator-pmic-buck1 { + compatible = "regulator-fixed"; + regulator-name = "pmic-buck1"; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <1100000>; + regulator-min-microvolt = <1100000>; + }; + + reg_pmic_buck3: regulator-pmic-buck3 { + compatible = "regulator-fixed"; + regulator-name = "pmic-buck3"; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <1800000>; + regulator-min-microvolt = <1800000>; + }; + + reg_pmic_buck4: regulator-pmic-buck4 { + compatible = "regulator-fixed"; + regulator-name = "pmic-buck4"; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <3300000>; + regulator-min-microvolt = <3300000>; + }; + + reg_pmic_ldo1: regulator-pmic-ldo1 { + compatible = "regulator-gpio"; + regulator-name = "pmic-ldo1"; + gpios = <&pinctrl RZG2L_GPIO(39, 0) GPIO_ACTIVE_HIGH>; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <3300000>; + regulator-min-microvolt = <1800000>; + states = <3300000 1>, <1800000 0>; + }; + + reg_pmic_ldo2: regulator-pmic-ldo2 { + compatible = "regulator-fixed"; + regulator-name = "pmic-ldo2"; + /* + * This ldo can switch mmc host controller io voltage between + * 1.8V and 3.3V by assembly option of pull-up / pull-dow. + * Default assembly is 3.3V. + */ + regulator-min-microvolt = <3300000>; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <3300000>; + }; + + reserved-memory { + ranges; + #address-cells = <2>; + #size-cells = <2>; + + global_cma: linux,cma@58000000 { + compatible = "shared-dma-pool"; + reg = <0x0 0x58000000 0x0 0x10000000>; + reusable; + linux,cma-default; + }; + }; + + sdhi1_pwrseq: sdhi1-pwrseq { + compatible = "mmc-pwrseq-simple"; + reset-gpios = <&pinctrl RZG2L_GPIO(23, 0) GPIO_ACTIVE_LOW>; + }; + + /* 32.768kHz crystal */ + x2: x2-clock { + compatible = "fixed-clock"; + #clock-cells = <0>; + clock-frequency = <32768>; + }; +}; + +&ehci0 { + status = "okay"; +}; + +&ehci1 { + status = "okay"; +}; + +ð0 { + phy-handle = <&phy0>; + pinctrl-0 = <ð0_pins>; + pinctrl-names = "default"; + /* + * ravb driver does not configure mac internal delays for RZ/G2L(C), + * instead delays are added by the ADIN1200 phy driver. + */ + phy-mode = "rgmii-id"; + status = "okay"; + + phy0: ethernet-phy@0 { + reg = <0>; + /* Level interrupts are typical for phy but not supported on RZ/G2L gpios. */ + interrupts-extended = <&pinctrl RZG2L_GPIO(27, 0) IRQ_TYPE_EDGE_FALLING>; + }; +}; + +&extal_clk { + clock-frequency = <24000000>; +}; + +&gpu { + mali-supply = <®_pmic_buck1>; +}; + +&i2c0 { + clock-frequency = <400000>; + pinctrl-0 = <&i2c0_pins>; + pinctrl-names = "default"; + status = "okay"; + + eeprom: eeprom@50 { + compatible = "atmel,24c01"; + reg = <0x50>; + pagesize = <16>; + }; +}; + +&i2c1 { + clock-frequency = <400000>; + pinctrl-0 = <&i2c1_pins>; + pinctrl-names = "default"; + status = "okay"; +}; + +&i2c2 { + clock-frequency = <400000>; + pinctrl-0 = <&i2c2_pins>; + pinctrl-names = "default"; + status = "okay"; + + pmic: pmic@12 { + compatible = "renesas,raa215300"; + reg = <0x12>, <0x6f>; + reg-names = "main", "rtc"; + clocks = <&x2>; + clock-names = "xin"; + }; +}; + +&ohci0 { + status = "okay"; +}; + +&ohci1 { + status = "okay"; +}; + +&ostm1 { + status = "okay"; +}; + +&ostm2 { + status = "okay"; +}; + +&phyrst { + status = "okay"; +}; + +&pinctrl { + eth0_pins: eth0 { + pinmux = , /* ET0_LINKSTA */ + , /* ET0_MDC */ + , /* ET0_MDIO */ + , /* ET0_TXC */ + , /* ET0_TX_CTL */ + , /* ET0_TXD0 */ + , /* ET0_TXD1 */ + , /* ET0_TXD2 */ + , /* ET0_TXD3 */ + , /* ET0_RXC */ + , /* ET0_RX_CTL */ + , /* ET0_RXD0 */ + , /* ET0_RXD1 */ + , /* ET0_RXD2 */ + ; /* ET0_RXD3 */ + }; + + i2c0_pins: i2c0 { + input-enable; + pins = "RIIC0_SDA", "RIIC0_SCL"; + }; + + i2c1_pins: i2c1 { + input-enable; + pins = "RIIC1_SDA", "RIIC1_SCL"; + }; + + i2c2_pins: i2c2 { + pinmux = , /* RIIC2_SDA */ + ; /* RIIC2_SCL */ + }; + + qspi0_pins: qspi0 { + pins = "QSPI0_IO0", "QSPI0_IO1", "QSPI0_IO2", "QSPI0_IO3", + "QSPI0_SPCLK", "QSPI0_SSL"; + power-source = <1800>; + }; + + scif0_pins: scif0 { + pinmux = , /* SCIF0_TXD */ + ; /* SCIF0_RXD */ + }; + + scif2_pins: scif2 { + pinmux = , /* SCIF2_TXD */ + , /* SCIF2_RXD */ + , /* SCIF2_CTS# */ + ; /* SCIF2_RTS# */ + }; + + sdhi0_4bit_pins: sdhi0-4bit { + sd0_ctrl_dat03 { + pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3", + "SD0_CLK", "SD0_CMD"; + power-source = <3300>; + }; + + /* + * Pins 4-7 are hard-wired to eMMC with 1.8V IO voltage, + * with all pins sharing a single voltage domain. + * + * These pins are fixed-function lacking independently configurable + * pull-up, pull-down or output-disable. + * + * While unused, disable the input buffers to avoid sampling the still + * physically connected but unused eMMC lines. + * For output direction rely on correct controller bus-width configuration + * to avoid over-voltage to eMMC e.g. when operating at 3.3V for a 4-bit microSD. + */ + sd0_dat47 { + input-disable; + pins = "SD0_DATA4", "SD0_DATA5", "SD0_DATA6", "SD0_DATA7"; + }; + }; + + sdhi0_4bit_uhs_pins: sdhi0-4bit-uhs { + sd0_ctrl_dat03 { + pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3", + "SD0_CLK", "SD0_CMD"; + power-source = <1800>; + }; + + /* + * Pins 4-7 are hard-wired to eMMC with 1.8V IO voltage, + * with all pins sharing a single voltage domain. + * + * These pins are fixed-function, lacking configurable pull-up, pull-down or + * output-disable, while sharing the voltage domain with all SD0_* pins. + * + * While unused, disable the input buffers to avoid sampling the still + * physically connected but unused eMMC lines. + * For output direction rely on correct controller bus-width configuration + * to avoid over-voltage to eMMC e.g. when operating at 3.3V for a 4-bit microSD. + */ + sd0_dat47 { + input-disable; + pins = "SD0_DATA4", "SD0_DATA5", "SD0_DATA6", "SD0_DATA7"; + }; + }; + + sdhi0_8bit_uhs_pins: sdhi0-8bit-uhs { + sd0_ctrl { + pins = "SD0_CLK", "SD0_CMD"; + power-source = <1800>; + }; + + sd0_dat { + input-enable; + pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3", + "SD0_DATA4", "SD0_DATA5", "SD0_DATA6", "SD0_DATA7"; + }; + }; + + sdhi0_cd_pins: sdhi0-cd { + pinmux = ; /* SD0_CD */ + }; + + /* SD0_RST is only routed to eMMC which uses fixed 1.8V IO voltage */ + sdhi0_rst_pins: sdhi0-rst { + pins = "SD0_RST#"; + power-source = <1800>; + }; + + sdhi1_pins: sdhi1 { + pins = "SD1_DATA0", "SD1_DATA1", "SD1_DATA2", "SD1_DATA3", + "SD1_CLK", "SD1_CMD"; + power-source = <3300>; + }; + + spi1_pins: spi1 { + pinmux = , /* RSPI1_MISO */ + , /* RSPI1_MOSI */ + ; /* RSPI1_CK */ + }; + + spi1_cs_pins: spi1-cs { + pinmux = ; /* RSPI1_SSL */ + }; +}; + +&sbc { + pinctrl-0 = <&qspi0_pins>; + pinctrl-names = "default"; + status = "okay"; + + flash@0 { + compatible = "winbond,w25q80bl", "jedec,spi-nor"; + reg = <0>; + spi-max-frequency = <50000000>; + m25p,fast-read; + }; +}; + +&scif0 { + pinctrl-0 = <&scif0_pins>; + pinctrl-names = "default"; + status = "okay"; +}; + +&scif2 { + pinctrl-0 = <&scif2_pins>; + pinctrl-names = "default"; + uart-has-rtscts; + status = "okay"; + + bluetooth { + /* + * The BT is based on BCM4343A2. Murata 1YN variant seems to + * require driving RTS on open. + * Use compatible bcm43438 to accomplish this. + */ + compatible = "brcm,bcm43438-bt"; + max-speed = <115200>; + shutdown-gpios = <&pinctrl RZG2L_GPIO(23, 1) GPIO_ACTIVE_HIGH>; + vbat-supply = <®_pmic_buck4>; + vddio-supply = <®_pmic_buck4>; + }; +}; + +/* WiFi */ +&sdhi1 { + /* Murata 1YN max rate is 50MHz */ + max-frequency = <50000000>; + bus-width = <4>; + cap-sdio-irq; + mmc-pwrseq = <&sdhi1_pwrseq>; + non-removable; + no-1-8-v; + no-sd; + pinctrl-0 = <&sdhi1_pins>; + pinctrl-names = "default"; + vmmc-supply = <®_pmic_buck4>; + /* + * Host controller IO voltage is provided from reg_pmic_ldo2, + * WiFi module IO voltage from reg_pmic_buck4. + * Neither is configurable at run-time so either can be set here. + */ + vqmmc-supply = <®_pmic_ldo2>; + status = "okay"; +}; + +&usb2_phy0 { + vbus-supply = <&usb0_vbus_otg>; + status = "okay"; +}; + +&usb2_phy1 { + status = "okay"; +}; + +&wdt0 { + status = "okay"; +}; From 1b16a69bb19bdfafc41fc4bf5ef3059c9511fa13 Mon Sep 17 00:00:00 2001 From: Josua Mayer Date: Mon, 28 Sep 2026 19:33:58 +0200 Subject: [PATCH 0181/1012] arm64: dts: renesas: rzg2l(c)/rzv2l hb-iiot: Add dsi panel dt overlay The SolidRun HummingBoard IIoT features a DSI connector supporting touch-screen displays and is usable with RZ/G2L, RZ/G2LC and RZ/V2L SoMs. Provide addon for the Winstar WJ70N3TYJHMNG0 1024x600 dsi panel. Add the addon in Makefile for all supported SoMs SoC targets. Reviewed-by: Geert Uytterhoeven Signed-off-by: Josua Mayer Link: https://patch.msgid.link/20260928-rzg2-sr-boards-v10-7-91752ea39cbc@solid-run.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/Makefile | 8 ++ ...ngboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtso | 79 +++++++++++++++++++ 2 files changed, 87 insertions(+) create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtso diff --git a/arch/arm64/boot/dts/renesas/Makefile b/arch/arm64/boot/dts/renesas/Makefile index 4296f893eea46d..d538ef85017a92 100644 --- a/arch/arm64/boot/dts/renesas/Makefile +++ b/arch/arm64/boot/dts/renesas/Makefile @@ -176,6 +176,8 @@ r9a07g044c2-hummingboard-iiot-emmc-dtbs := r9a07g044c2-hummingboard-iiot.dtb rzg dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-iiot-emmc.dtb r9a07g044c2-hummingboard-iiot-microsd-dtbs := r9a07g044c2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-microsd.dtbo dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-iiot-microsd.dtb +r9a07g044c2-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0-dtbs := r9a07g044c2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtb r9a07g044c2-hummingboard-iiot-rs485-a-dtbs := r9a07g044c2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-a.dtbo dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-iiot-rs485-a.dtb r9a07g044c2-hummingboard-iiot-rs485-b-dtbs := r9a07g044c2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-b.dtbo @@ -192,6 +194,9 @@ dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-iiot-emmc.dtb dtb-$(CONFIG_ARCH_R9A07G044) += rzg2l-hummingboard-iiot-microsd.dtbo r9a07g044l2-hummingboard-iiot-microsd-dtbs := r9a07g044l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-microsd.dtbo dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-iiot-microsd.dtb +dtb-$(CONFIG_ARCH_R9A07G044) += rzg2l-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtbo +r9a07g044l2-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0-dtbs := r9a07g044l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtb dtb-$(CONFIG_ARCH_R9A07G044) += rzg2l-hummingboard-iiot-rs485-a.dtbo r9a07g044l2-hummingboard-iiot-rs485-a-dtbs := r9a07g044l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-a.dtbo dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-iiot-rs485-a.dtb @@ -211,6 +216,9 @@ dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-iiot-emmc.dtb dtb-$(CONFIG_ARCH_R9A07G054) += rzg2l-hummingboard-iiot-microsd.dtbo r9a07g054l2-hummingboard-iiot-microsd-dtbs := r9a07g054l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-microsd.dtbo dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-iiot-microsd.dtb +dtb-$(CONFIG_ARCH_R9A07G054) += rzg2l-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtbo +r9a07g054l2-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0-dtbs := r9a07g054l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtbo +dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtb dtb-$(CONFIG_ARCH_R9A07G054) += rzg2l-hummingboard-iiot-rs485-a.dtbo r9a07g054l2-hummingboard-iiot-rs485-a-dtbs := r9a07g054l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-a.dtbo dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-iiot-rs485-a.dtb diff --git a/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtso b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtso new file mode 100644 index 00000000000000..f8584c473fafed --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtso @@ -0,0 +1,79 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + * + * Overlay for enabling HummingBoard IIoT MIPI-DSI connector + * with Winstar WJ70N3TYJHMNG0 panel. + */ + +/dts-v1/; +/plugin/; + +#include +#include + +&{/} { + dsi_backlight: dsi-backlight { + compatible = "gpio-backlight"; + gpios = <&tca6408_u48 3 GPIO_ACTIVE_LOW>; + }; +}; + +&i2c_dsi { + #address-cells = <1>; + #size-cells = <0>; + + touchscreen@41 { + compatible = "ilitek,ili2130"; + reg = <0x41>; + /* + * Touchscreen fires short active-low pulses which level trigger + * on i2c gpio expander can miss. Trigger on falling edge instead. + */ + interrupts-extended = <&tca6416_u21 13 IRQ_TYPE_EDGE_FALLING>; + reset-gpios = <&tca6408_u48 6 GPIO_ACTIVE_LOW>; + }; +}; + +&dsi { + #address-cells = <1>; + #size-cells = <0>; + status = "okay"; + + panel@0 { + /* This is a Winstar panel, but the ronbo panel uses same controls. */ + compatible = "ronbo,rb070d30"; + reg = <0>; + /* reset is active-low but driver inverts it internally */ + reset-gpios = <&tca6408_u48 1 GPIO_ACTIVE_HIGH>; + backlight = <&dsi_backlight>; + power-gpios = <&tca6408_u48 2 GPIO_ACTIVE_HIGH>; + shlr-gpios = <&tca6408_u48 4 GPIO_ACTIVE_LOW>; + updn-gpios = <&tca6408_u48 5 GPIO_ACTIVE_HIGH>; + vcc-lcd-supply = <®_dsi_panel>; + + port { + dsi_in_panel: endpoint { + remote-endpoint = <&mipi_dsi_out>; + }; + }; + }; + + ports { + #address-cells = <1>; + #size-cells = <0>; + + port@1 { + reg = <1>; + + mipi_dsi_out: endpoint { + data-lanes = <1 2 3 4>; + remote-endpoint = <&dsi_in_panel>; + }; + }; + }; +}; + +&du { + status = "okay"; +}; From dadb5237822b112cf51e2b27aa0613068294b289 Mon Sep 17 00:00:00 2001 From: Josua Mayer Date: Mon, 28 Sep 2026 19:33:59 +0200 Subject: [PATCH 0182/1012] arm64: dts: renesas: Add support for solidrun hb-ripple with rzg2l som Add support for the SolidRun HummingBoard Ripple [2] with RZ/G2L SoM [1]. The HummingBoard Ripple is a reduced version of HummingBoard Pulse, featuring: - 2x USB-2.0 Type-A connector - 1x 1Gbps RJ45 Ethernet - micro-HDMI connector - microSD connector - mini-PCI-E connector with SIM slot supporting USB-2.0 interface - MIPI-CSI Camera Connector (not described without specific camera) - RTC with backup battery The carrier board is identical between RZ/G2L, RZ/G2LC, RZ/G2UL and RZ/V2L SoMs, yet only the RZ/G2LC combination has a product page [2]. While the variant being supported here is named "Ripple", shared include files are still named according to the full board for consistency with schematics, silk screen labels and other SoMs on same board. Description for microSD is shared with HummingBoard IIoT. [1] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/rz-g2l-som/ [2] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/hummingboard-rz-series-sbcs/hummingboard-rz-g2lc-base/ Reviewed-by: Geert Uytterhoeven Signed-off-by: Josua Mayer Link: https://patch.msgid.link/20260928-rzg2-sr-boards-v10-8-91752ea39cbc@solid-run.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/Makefile | 5 + .../r9a07g044l2-hummingboard-ripple.dts | 18 +++ .../rzg2l-hummingboard-pulse-common.dtsi | 116 +++++++++++++++++ .../rzg2l-hummingboard-pulse-micro-hdmi.dtsi | 79 ++++++++++++ .../renesas/rzg2l-hummingboard-ripple.dtsi | 122 ++++++++++++++++++ 5 files changed, 340 insertions(+) create mode 100644 arch/arm64/boot/dts/renesas/r9a07g044l2-hummingboard-ripple.dts create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-pulse-common.dtsi create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-pulse-micro-hdmi.dtsi create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-ripple.dtsi diff --git a/arch/arm64/boot/dts/renesas/Makefile b/arch/arm64/boot/dts/renesas/Makefile index d538ef85017a92..8a524ffa8f4144 100644 --- a/arch/arm64/boot/dts/renesas/Makefile +++ b/arch/arm64/boot/dts/renesas/Makefile @@ -203,6 +203,11 @@ dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-iiot-rs485-a.dtb dtb-$(CONFIG_ARCH_R9A07G044) += rzg2l-hummingboard-iiot-rs485-b.dtbo r9a07g044l2-hummingboard-iiot-rs485-b-dtbs := r9a07g044l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-b.dtbo dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-iiot-rs485-b.dtb +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-ripple.dtb +r9a07g044l2-hummingboard-ripple-emmc-dtbs := r9a07g044l2-hummingboard-ripple.dtb rzg2l-sr-som-emmc.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-ripple-emmc.dtb +r9a07g044l2-hummingboard-ripple-microsd-dtbs := r9a07g044l2-hummingboard-ripple.dtb rzg2l-hummingboard-iiot-microsd.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-hummingboard-ripple-microsd.dtb dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-remi-pi.dtb dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-smarc.dtb dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044l2-smarc-cru-csi-ov5645.dtbo diff --git a/arch/arm64/boot/dts/renesas/r9a07g044l2-hummingboard-ripple.dts b/arch/arm64/boot/dts/renesas/r9a07g044l2-hummingboard-ripple.dts new file mode 100644 index 00000000000000..95a36f8804b558 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/r9a07g044l2-hummingboard-ripple.dts @@ -0,0 +1,18 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + */ + +/dts-v1/; + +#include "r9a07g044l2.dtsi" +#include "rzg2l-sr-som.dtsi" +#include "rzg2l-hummingboard-pulse-common.dtsi" +#include "rzg2l-hummingboard-pulse-micro-hdmi.dtsi" +#include "rzg2l-hummingboard-ripple.dtsi" + +/ { + compatible = "solidrun,rzg2l-hummingboard-ripple", "solidrun,rzg2l-sr-som", + "renesas,r9a07g044l2", "renesas,r9a07g044"; + model = "SolidRun RZ/G2L HummingBoard Ripple"; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-pulse-common.dtsi b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-pulse-common.dtsi new file mode 100644 index 00000000000000..3fd49333ff1953 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-pulse-common.dtsi @@ -0,0 +1,116 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + */ + +#include + +/ { + aliases { + rtc0 = &carrier_rtc; + rtc1 = &pmic; + }; + + v_1_2: regulator-1v2 { + compatible = "regulator-fixed"; + regulator-name = "1v2"; + regulator-max-microvolt = <1200000>; + regulator-min-microvolt = <1200000>; + vin-supply = <®_pmic_buck4>; + }; + + vmmc: regulator-mmc { + compatible = "regulator-fixed"; + regulator-name = "vmmc"; + regulator-max-microvolt = <3300000>; + regulator-min-microvolt = <3300000>; + startup-delay-us = <250>; + vin-supply = <®_pmic_buck4>; + }; + + vmpcie: regulator-mpcie { + /* supplies mpcie and m2 connectors */ + regulator-name = "vmpcie"; + compatible = "regulator-fixed"; + regulator-always-on; + regulator-max-microvolt = <3300000>; + regulator-min-microvolt = <3300000>; + enable-active-high; + }; +}; + +&ehci1 { + #address-cells = <1>; + #size-cells = <0>; + + hub_2_0: hub@1 { + compatible = "usb4b4,6502", "usb4b4,6506"; + reg = <1>; + vdd2-supply = <®_pmic_buck4>; + vdd-supply = <&v_1_2>; + }; +}; + +&i2c0 { + carrier_eeprom: eeprom@57 { + compatible = "st,24c02", "atmel,24c02"; + reg = <0x57>; + pagesize = <16>; + }; + + carrier_rtc: rtc@69 { + compatible = "abracon,ab1805"; + reg = <0x69>; + abracon,tc-diode = "schottky"; + abracon,tc-resistor = <3>; + }; +}; + +&ohci0 { + dr_mode = "host"; +}; + +&ohci1 { + dr_mode = "host"; +}; + +&phy0 { + leds { + #address-cells = <1>; + #size-cells = <0>; + + led@0 { + reg = <0>; + color = ; + default-state = "keep"; + function = LED_FUNCTION_LAN; + }; + + led@1 { + reg = <1>; + color = ; + default-state = "keep"; + function = LED_FUNCTION_LAN; + }; + }; +}; + +/* mikrobus uart */ +&scif1 { + pinctrl-0 = <&mikro_uart_pins>; + pinctrl-names = "default"; + status = "okay"; +}; + +/* mikrobus spi */ +&spi1 { + num-cs = <1>; + pinctrl-0 = <&spi1_pins>, <&spi1_cs_pins>; + pinctrl-names = "default"; + status = "okay"; +}; + +&usb2_phy0 { + pinctrl-0 = <&usb0_vbus_pins>; + pinctrl-names = "default"; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-pulse-micro-hdmi.dtsi b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-pulse-micro-hdmi.dtsi new file mode 100644 index 00000000000000..80e43510142ab5 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-pulse-micro-hdmi.dtsi @@ -0,0 +1,79 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + */ + +/ { + hdmi-connector { + compatible = "hdmi-connector"; + label = "hdmi"; + type = "d"; + + port { + hdmi_connector_in: endpoint { + remote-endpoint = <&adv7535_out>; + }; + }; + }; + + vmipi: regulator-mipi-1-8 { + compatible = "regulator-fixed"; + regulator-name = "vmipi"; + regulator-always-on; + regulator-max-microvolt = <1800000>; + regulator-min-microvolt = <1800000>; + vin-supply = <®_pmic_buck4>; + }; +}; + +&dsi { + status = "okay"; + + ports { + port@1 { + mipi_dsi_out: endpoint { + data-lanes = <1 2 3 4>; + remote-endpoint = <&adv7535_from_dsim>; + }; + }; + }; +}; + +&du { + status = "okay"; +}; + +&i2c0 { + hdmi: hdmi@3d { + compatible = "adi,adv7535"; + reg = <0x3d>, <0x3f>, <0x3c>, <0x38>; + reg-names = "main", "edid", "cec", "packet"; + a2vdd-supply = <&vmipi>; + avdd-supply = <&vmipi>; + dvdd-supply = <&vmipi>; + pvdd-supply = <&vmipi>; + v3p3-supply = <®_pmic_buck4>; + adi,dsi-lanes = <4>; + + ports { + #address-cells = <1>; + #size-cells = <0>; + + port@0 { + reg = <0>; + + adv7535_from_dsim: endpoint { + remote-endpoint = <&mipi_dsi_out>; + }; + }; + + port@1 { + reg = <1>; + + adv7535_out: endpoint { + remote-endpoint = <&hdmi_connector_in>; + }; + }; + }; + }; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-ripple.dtsi b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-ripple.dtsi new file mode 100644 index 00000000000000..aeb92630f4c6b4 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2l-hummingboard-ripple.dtsi @@ -0,0 +1,122 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + */ + +/ { + aliases { + /* this board does not use second phy / ethernet on SoM */ + /delete-property/ ethernet1; + }; + + leds { + compatible = "gpio-leds"; + + led-0 { + color = ; + default-state = "on"; + gpios = <&pinctrl RZG2L_GPIO(0, 1) GPIO_ACTIVE_LOW>; + label = "D30"; + }; + + led-1 { + color = ; + default-state = "on"; + gpios = <&pinctrl RZG2L_GPIO(47, 2) GPIO_ACTIVE_LOW>; + label = "D31"; + }; + + led-2 { + color = ; + default-state = "on"; + gpios = <&pinctrl RZG2L_GPIO(47, 1) GPIO_ACTIVE_LOW>; + label = "D32"; + }; + + led-3 { + color = ; + default-state = "on"; + gpios = <&pinctrl RZG2L_GPIO(48, 2) GPIO_ACTIVE_LOW>; + label = "D33"; + }; + + led-4 { + color = ; + default-state = "on"; + gpios = <&pinctrl RZG2L_GPIO(7, 2) GPIO_ACTIVE_LOW>; + label = "D34"; + }; + }; + + rfkill-mpcie { + /* rfkill-gpio inverts internally */ + shutdown-gpios = <&pinctrl RZG2L_GPIO(14, 1) GPIO_ACTIVE_HIGH>; + label = "mpcie radio"; + /* + * The mpcie connector only has USB, + * therefore this rfkill is for cellular radios only. + */ + radio-type = "wwan"; + compatible = "rfkill-gpio"; + }; +}; + +ð1 { + /* this board does not use second phy / ethernet on SoM */ + status = "disabled"; +}; + +&gpt { + /* mikrobus header pwm pin is pwm channel 6 */ + pinctrl-0 = <&mikro_pwm_pins>; + pinctrl-names = "default"; + status = "okay"; +}; + +&hdmi { + interrupts-extended = <&pinctrl RZG2L_GPIO(9, 1) IRQ_TYPE_EDGE_FALLING>; + pd-gpios = <&pinctrl RZG2L_GPIO(3, 0) GPIO_ACTIVE_LOW>; +}; + +&hub_2_0 { + reset-gpios = <&pinctrl RZG2L_GPIO(39, 1) GPIO_ACTIVE_LOW>; +}; + +&pinctrl { + mpcie_rst_hog: mpcie-reset-hog { + gpios = ; + gpio-hog; + line-name = "mpcie-reset"; + output-low; + }; + + mikro_pwm_pins: pinctrl-mikro-pwm-grp { + pinmux = ; /* GTIOC3A */ + }; + + mikro_uart_pins: pinctrl-mikro-uart-grp { + pinmux = , /* SCIF1_RXD */ + ; /* SCIF1_TXD */ + }; + + usb0_vbus_pins: usb0-vbus { + pinmux = ; /* USB0_VBUSEN */ + }; + + usb1_vbus_pins: usb1-vbus { + pinmux = ; /* USB1_VBUSEN */ + }; +}; + +&usb2_phy1 { + pinctrl-0 = <&usb1_vbus_pins>; + pinctrl-names = "default"; +}; + +&vmmc { + gpios = <&pinctrl RZG2L_GPIO(4, 1) GPIO_ACTIVE_LOW>; +}; + +&vmpcie { + gpios = <&pinctrl RZG2L_GPIO(17, 0) GPIO_ACTIVE_HIGH>; +}; From 1fbab7d1380655531e08f5a3c81c5889f5ca6368 Mon Sep 17 00:00:00 2001 From: Josua Mayer Date: Mon, 28 Sep 2026 19:34:00 +0200 Subject: [PATCH 0183/1012] arm64: dts: renesas: Add support for solidrun hb-ripple with rzv2l som Add support for the SolidRun HummingBoard Ripple [2] with RZ/V2L SoM [1]. The HummingBoard Ripple is a reduced version of HummingBoard Pulse, featuring: - 2x USB-2.0 Type-A connector - 1x 1Gbps RJ45 Ethernet - micro-HDMI connector - microSD connector - mini-PCI-E connector with SIM slot supporting USB-2.0 interface - MIPI-CSI Camera Connector (not described without specific camera) - RTC with backup battery The carrier board is identical between RZ/G2L, RZ/G2LC, RZ/G2UL and RZ/V2L SoMs, yet only the RZ/G2LC combination has a product page [2]. The RZ/V2L SoM shares PCB with RZ/G2L, differing only in the SoC itself. RZ/V2L is adding a powerful DRP-AI NPU which G2L lacks. Due to the similarities, all descriptions except SoC are shared with RZ/G2L. [1] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/rz-v2l-som/ [2] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/hummingboard-rz-series-sbcs/hummingboard-rz-g2lc-base/ Reviewed-by: Geert Uytterhoeven Signed-off-by: Josua Mayer Link: https://patch.msgid.link/20260928-rzg2-sr-boards-v10-9-91752ea39cbc@solid-run.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/Makefile | 5 +++++ .../r9a07g054l2-hummingboard-ripple.dts | 18 ++++++++++++++++++ 2 files changed, 23 insertions(+) create mode 100644 arch/arm64/boot/dts/renesas/r9a07g054l2-hummingboard-ripple.dts diff --git a/arch/arm64/boot/dts/renesas/Makefile b/arch/arm64/boot/dts/renesas/Makefile index 8a524ffa8f4144..dd9cf518bfed54 100644 --- a/arch/arm64/boot/dts/renesas/Makefile +++ b/arch/arm64/boot/dts/renesas/Makefile @@ -230,6 +230,11 @@ dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-iiot-rs485-a.dtb dtb-$(CONFIG_ARCH_R9A07G054) += rzg2l-hummingboard-iiot-rs485-b.dtbo r9a07g054l2-hummingboard-iiot-rs485-b-dtbs := r9a07g054l2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-b.dtbo dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-iiot-rs485-b.dtb +dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-ripple.dtb +r9a07g054l2-hummingboard-ripple-emmc-dtbs := r9a07g054l2-hummingboard-ripple.dtb rzg2l-sr-som-emmc.dtbo +dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-ripple-emmc.dtb +r9a07g054l2-hummingboard-ripple-microsd-dtbs := r9a07g054l2-hummingboard-ripple.dtb rzg2l-hummingboard-iiot-microsd.dtbo +dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-hummingboard-ripple-microsd.dtb dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-smarc.dtb dtb-$(CONFIG_ARCH_R9A07G054) += r9a07g054l2-smarc-cru-csi-ov5645.dtbo r9a07g054l2-smarc-cru-csi-ov5645-dtbs := r9a07g054l2-smarc.dtb r9a07g054l2-smarc-cru-csi-ov5645.dtbo diff --git a/arch/arm64/boot/dts/renesas/r9a07g054l2-hummingboard-ripple.dts b/arch/arm64/boot/dts/renesas/r9a07g054l2-hummingboard-ripple.dts new file mode 100644 index 00000000000000..0c738715245414 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/r9a07g054l2-hummingboard-ripple.dts @@ -0,0 +1,18 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + */ + +/dts-v1/; + +#include "r9a07g054l2.dtsi" +#include "rzg2l-sr-som.dtsi" +#include "rzg2l-hummingboard-pulse-common.dtsi" +#include "rzg2l-hummingboard-pulse-micro-hdmi.dtsi" +#include "rzg2l-hummingboard-ripple.dtsi" + +/ { + compatible = "solidrun,rzv2l-hummingboard-ripple", "solidrun,rzv2l-sr-som", + "renesas,r9a07g054l2", "renesas,r9a07g054"; + model = "SolidRun RZ/V2L HummingBoard Ripple"; +}; From f6b483d441e3dee1a5ece1266ae731c953ed9bb6 Mon Sep 17 00:00:00 2001 From: Josua Mayer Date: Mon, 28 Sep 2026 19:34:01 +0200 Subject: [PATCH 0184/1012] arm64: dts: renesas: Add support for solidrun hb-ripple with rzg2lc som Add support for the SolidRun HummingBoard Ripple [2] with RZ/G2LC SoM [1]. The HummingBoard Ripple is a reduced version of HummingBoard Pulse, featuring: - 2x USB-2.0 Type-A connector - 1x 100Mbps RJ45 Ethernet - micro-HDMI connector - microSD connector - mini-PCI-E connector with SIM slot supporting USB-2.0 interface - MIPI-CSI Camera Connector (not described without specific camera) - RTC with backup battery The carrier board is identical between RZ/G2L, RZ/G2LC, RZ/G2UL and RZ/V2L SoMs, yet only the RZ/G2LC combination has a product page [2]. RZ/G2LC SoM shares SCIF2 pins between on-SoM Bluetooth and carrier board. Provide dt overlay changing pin function from uart to GPIO LED control. While the variant being supported here is named "Ripple", the overlay still carries the "pulse" suffix as it would fit all its variants. Descriptions for eMMC and microSD are shared with G2L SoM and HummingBoard IIoT. [1] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/rz-g2lc-som/ [2] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/hummingboard-rz-series-sbcs/hummingboard-rz-g2lc-base/ Reviewed-by: Geert Uytterhoeven Signed-off-by: Josua Mayer Link: https://patch.msgid.link/20260928-rzg2-sr-boards-v10-10-91752ea39cbc@solid-run.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/Makefile | 8 ++ .../r9a07g044c2-hummingboard-ripple.dts | 93 +++++++++++++++++++ .../rzg2lc-hummingboard-pulse-leds.dtso | 48 ++++++++++ 3 files changed, 149 insertions(+) create mode 100644 arch/arm64/boot/dts/renesas/r9a07g044c2-hummingboard-ripple.dts create mode 100644 arch/arm64/boot/dts/renesas/rzg2lc-hummingboard-pulse-leds.dtso diff --git a/arch/arm64/boot/dts/renesas/Makefile b/arch/arm64/boot/dts/renesas/Makefile index dd9cf518bfed54..50a54abc761889 100644 --- a/arch/arm64/boot/dts/renesas/Makefile +++ b/arch/arm64/boot/dts/renesas/Makefile @@ -182,6 +182,14 @@ r9a07g044c2-hummingboard-iiot-rs485-a-dtbs := r9a07g044c2-hummingboard-iiot.dtb dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-iiot-rs485-a.dtb r9a07g044c2-hummingboard-iiot-rs485-b-dtbs := r9a07g044c2-hummingboard-iiot.dtb rzg2l-hummingboard-iiot-rs485-b.dtbo dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-iiot-rs485-b.dtb +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-ripple.dtb +r9a07g044c2-hummingboard-ripple-emmc-dtbs := r9a07g044c2-hummingboard-ripple.dtb rzg2l-sr-som-emmc.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-ripple-emmc.dtb +dtb-$(CONFIG_ARCH_R9A07G044) += rzg2lc-hummingboard-pulse-leds.dtbo +r9a07g044c2-hummingboard-ripple-leds-dtbs := r9a07g044c2-hummingboard-ripple.dtb rzg2lc-hummingboard-pulse-leds.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-ripple-leds.dtb +r9a07g044c2-hummingboard-ripple-microsd-dtbs := r9a07g044c2-hummingboard-ripple.dtb rzg2l-hummingboard-iiot-microsd.dtbo +dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-ripple-microsd.dtb dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-smarc.dtb dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-smarc-cru-csi-ov5645.dtbo r9a07g044c2-smarc-cru-csi-ov5645-dtbs := r9a07g044c2-smarc.dtb r9a07g044c2-smarc-cru-csi-ov5645.dtbo diff --git a/arch/arm64/boot/dts/renesas/r9a07g044c2-hummingboard-ripple.dts b/arch/arm64/boot/dts/renesas/r9a07g044c2-hummingboard-ripple.dts new file mode 100644 index 00000000000000..ef8e2f62b6827f --- /dev/null +++ b/arch/arm64/boot/dts/renesas/r9a07g044c2-hummingboard-ripple.dts @@ -0,0 +1,93 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Copyright 2025 Josua Mayer + */ + +/dts-v1/; + +#include "r9a07g044c2.dtsi" +#include "rzg2lc-sr-som.dtsi" +#include "rzg2l-hummingboard-pulse-common.dtsi" +#include "rzg2l-hummingboard-pulse-micro-hdmi.dtsi" + +/ { + compatible = "solidrun,rzg2lc-hummingboard-ripple", "solidrun,rzg2lc-sr-som", + "renesas,r9a07g044c2", "renesas,r9a07g044"; + model = "SolidRun RZ/G2LC HummingBoard Ripple"; + + leds: leds { + compatible = "gpio-leds"; + + led-0 { + color = ; + default-state = "on"; + gpios = <&pinctrl RZG2L_GPIO(0, 1) GPIO_ACTIVE_LOW>; + label = "D30"; + }; + }; + + vbus1: regulator-vbus-1 { + compatible = "regulator-fixed"; + regulator-name = "vbus1"; + regulator-max-microvolt = <5000000>; + regulator-min-microvolt = <5000000>; + gpios = <&pinctrl RZG2L_GPIO(5, 0) GPIO_ACTIVE_HIGH>; + enable-active-high; + }; +}; + +&gpt { + /* mikrobus header pwm pin is pwm channel 6 */ + pinctrl-0 = <&mikro_pwm_pins>; + pinctrl-names = "default"; + status = "okay"; +}; + +&hdmi { + interrupts-extended = <&pinctrl RZG2L_GPIO(42, 2) IRQ_TYPE_EDGE_FALLING>; + pd-gpios = <&pinctrl RZG2L_GPIO(40, 2) GPIO_ACTIVE_LOW>; +}; + +&hub_2_0 { + reset-gpios = <&pinctrl RZG2L_GPIO(39, 1) GPIO_ACTIVE_LOW>; +}; + +&pinctrl { + mpcie_rst_hog: mpcie-reset-hog { + gpios = ; + gpio-hog; + line-name = "mpcie-reset"; + output-low; + }; + + mikro_pwm_pins: pinctrl-mikro-pwm-grp { + pinmux = ; /* GTIOC3A */ + }; + + mikro_uart_pins: pinctrl-mikro-uart-grp { + pinmux = , /* SCIF1_RXD */ + ; /* SCIF1_TXD */ + }; + + usb0_vbus_pins: usb0-vbus { + pinmux = ; /* USB0_VBUSEN */ + }; +}; + +&usb2_phy1 { + vbus-supply = <&vbus1>; +}; + +&vmmc { + gpios = <&pinctrl RZG2L_GPIO(18, 1) GPIO_ACTIVE_LOW>; +}; + +&vmpcie { + /* + * G2LC SoM v1.2 moved this IO from J7-28 (CON4-12) + * to J7-18 (this regulator). + * There are no known users of CON4-12 and thus no need + * to differentiate here between SoM v1.1 and v1.2. + */ + gpios = <&pinctrl RZG2L_GPIO(39, 2) GPIO_ACTIVE_HIGH>; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2lc-hummingboard-pulse-leds.dtso b/arch/arm64/boot/dts/renesas/rzg2lc-hummingboard-pulse-leds.dtso new file mode 100644 index 00000000000000..a2572718fa87c5 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2lc-hummingboard-pulse-leds.dtso @@ -0,0 +1,48 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Overlay for SolidRun RZ/G2LC HummingBoard Pulse + * enabling LEDs D31-34 in exchange for on-SoM Bluetooth. + * + * Copyright 2025 Josua Mayer + */ + +/dts-v1/; +/plugin/; + +#include +#include +#include + +&leds { + led-1 { + label = "D31"; + color = ; + gpios = <&pinctrl RZG2L_GPIO(5, 2) GPIO_ACTIVE_LOW>; + default-state = "on"; + }; + + led-2 { + label = "D32"; + color = ; + gpios = <&pinctrl RZG2L_GPIO(5, 1) GPIO_ACTIVE_LOW>; + default-state = "on"; + }; + + led-3 { + label = "D33"; + color = ; + gpios = <&pinctrl RZG2L_GPIO(42, 0) GPIO_ACTIVE_LOW>; + default-state = "on"; + }; + + led-4 { + label = "D34"; + color = ; + gpios = <&pinctrl RZG2L_GPIO(42, 1) GPIO_ACTIVE_LOW>; + default-state = "on"; + }; +}; + +&scif2 { + status = "disabled"; +}; From c9bbb744bec0c9a24c3c5cec4d23b21ef1dfc1ee Mon Sep 17 00:00:00 2001 From: Josua Mayer Date: Mon, 28 Sep 2026 19:34:02 +0200 Subject: [PATCH 0185/1012] arm64: dts: renesas: Add support for solidrun rzg2ul som on hb-ripple Add support for the SolidRun RZ/G2UL SoM [1] on HummingBoard Ripple [2]. The SoM features: - 100Mbps Ethernet with PHY - eMMC - 512MB/1GB DDR - WiFi + Bluetooth - SDHI Mux switching between eMMC and Carrier Board The HummingBoard Ripple is a reduced version of HummingBoard Pulse, featuring: - 2x USB-2.0 Type-A connector - 1x 100Mbps RJ45 Ethernet - micro-HDMI connector (not usable with RZ/G2UL SoM) - microSD connector - mini-PCI-E connector with SIM slot supporting USB-2.0 interface - MIPI-CSI Camera Connector (not described without specific camera) - RTC with backup battery The carrier board is identical between RZ/G2L, RZ/G2LC, RZ/G2UL and RZ/V2L SoMs, yet only the RZ/G2LC combination has a product page [2]. RZ/G2UL SoM shares SCIF2 pins between on-SoM Bluetooth and carrier board. Provide dt overlay changing pin function from uart to GPIO LED control. While the variant being supported here is named "Ripple", the overlay still carries the "pulse" suffix as it would fit all its variants. Descriptions for eMMC and microSD are shared with G2L SoM and HummingBoard IIoT. [1] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/rz-g2ul-som/ [2] https://www.solid-run.com/embedded-industrial-iot/renesas-rz-family/hummingboard-rz-series-sbcs/hummingboard-rz-g2lc-base/ Reviewed-by: Geert Uytterhoeven Signed-off-by: Josua Mayer Link: https://patch.msgid.link/20260928-rzg2-sr-boards-v10-11-91752ea39cbc@solid-run.com Signed-off-by: Geert Uytterhoeven --- arch/arm64/boot/dts/renesas/Makefile | 10 + .../r9a07g043u12-hummingboard-ripple.dts | 89 ++++ .../rzg2ul-hummingboard-pulse-leds.dtso | 48 ++ .../arm64/boot/dts/renesas/rzg2ul-sr-som.dtsi | 418 ++++++++++++++++++ 4 files changed, 565 insertions(+) create mode 100644 arch/arm64/boot/dts/renesas/r9a07g043u12-hummingboard-ripple.dts create mode 100644 arch/arm64/boot/dts/renesas/rzg2ul-hummingboard-pulse-leds.dtso create mode 100644 arch/arm64/boot/dts/renesas/rzg2ul-sr-som.dtsi diff --git a/arch/arm64/boot/dts/renesas/Makefile b/arch/arm64/boot/dts/renesas/Makefile index 50a54abc761889..c6f3718cd19fad 100644 --- a/arch/arm64/boot/dts/renesas/Makefile +++ b/arch/arm64/boot/dts/renesas/Makefile @@ -170,6 +170,16 @@ r9a07g043u11-smarc-du-adv7513-dtbs := r9a07g043u11-smarc.dtb r9a07g043u11-smarc- dtb-$(CONFIG_ARCH_R9A07G043) += r9a07g043u11-smarc-du-adv7513.dtb r9a07g043u11-smarc-pmod-dtbs := r9a07g043u11-smarc.dtb r9a07g043-smarc-pmod.dtbo dtb-$(CONFIG_ARCH_R9A07G043) += r9a07g043u11-smarc-pmod.dtb +dtb-$(CONFIG_ARCH_R9A07G043) += r9a07g043u12-hummingboard-ripple.dtb +dtb-$(CONFIG_ARCH_R9A07G043) += rzg2l-sr-som-emmc.dtbo +r9a07g043u12-hummingboard-ripple-emmc-dtbs := r9a07g043u12-hummingboard-ripple.dtb rzg2l-sr-som-emmc.dtbo +dtb-$(CONFIG_ARCH_R9A07G043) += r9a07g043u12-hummingboard-ripple-emmc.dtb +dtb-$(CONFIG_ARCH_R9A07G043) += rzg2l-hummingboard-iiot-microsd.dtbo +dtb-$(CONFIG_ARCH_R9A07G043) += rzg2ul-hummingboard-pulse-leds.dtbo +r9a07g043u12-hummingboard-ripple-leds-dtbs := r9a07g043u12-hummingboard-ripple.dtb rzg2ul-hummingboard-pulse-leds.dtbo +dtb-$(CONFIG_ARCH_R9A07G043) += r9a07g043u12-hummingboard-ripple-leds.dtb +r9a07g043u12-hummingboard-ripple-microsd-dtbs := r9a07g043u12-hummingboard-ripple.dtb rzg2l-hummingboard-iiot-microsd.dtbo +dtb-$(CONFIG_ARCH_R9A07G043) += r9a07g043u12-hummingboard-ripple-microsd.dtb dtb-$(CONFIG_ARCH_R9A07G044) += r9a07g044c2-hummingboard-iiot.dtb r9a07g044c2-hummingboard-iiot-emmc-dtbs := r9a07g044c2-hummingboard-iiot.dtb rzg2l-sr-som-emmc.dtbo diff --git a/arch/arm64/boot/dts/renesas/r9a07g043u12-hummingboard-ripple.dts b/arch/arm64/boot/dts/renesas/r9a07g043u12-hummingboard-ripple.dts new file mode 100644 index 00000000000000..8ffa51dab56b22 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/r9a07g043u12-hummingboard-ripple.dts @@ -0,0 +1,89 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Description for SolidRun RZ/G2UL Type-2 HummingBoard Ripple. + * + * Copyright 2025 Josua Mayer + */ + +/dts-v1/; + +#include "r9a07g043u.dtsi" +#include "rzg2ul-sr-som.dtsi" +#include "rzg2l-hummingboard-pulse-common.dtsi" + +/ { + compatible = "solidrun,rzg2ul-hummingboard-ripple", "solidrun,rzg2ul-sr-som", + "renesas,r9a07g043u12", "renesas,r9a07g043"; + model = "SolidRun RZ/G2UL HummingBoard Ripple"; + + leds: leds { + compatible = "gpio-leds"; + + led-0 { + color = ; + default-state = "on"; + gpios = <&pinctrl RZG2L_GPIO(12, 1) GPIO_ACTIVE_LOW>; + label = "D30"; + }; + }; + + vbus1: regulator-vbus-1 { + compatible = "regulator-fixed"; + regulator-name = "vbus1"; + regulator-max-microvolt = <5000000>; + regulator-min-microvolt = <5000000>; + gpios = <&pinctrl RZG2L_GPIO(5, 2) GPIO_ACTIVE_HIGH>; + enable-active-high; + }; +}; + +&hub_2_0 { + reset-gpios = <&pinctrl RZG2L_GPIO(13, 3) GPIO_ACTIVE_LOW>; +}; + +&pinctrl { + hdmi_pd_hog: hdmi-pd-hog { + /* + * RZ/G2UL does not have DSI output to feed DSI-to-HDMI + * bridge. Keep its shutdown signal asserted. + */ + gpio-hog; + gpios = ; + line-name = "hdmi-pd"; + output-high; + }; + + mpcie_rst_hog: mpcie-reset-hog { + gpio-hog; + gpios = ; + line-name = "mpcie-reset"; + output-low; + }; + + mikro_uart_pins: pinctrl-mikro-uart-grp { + pinmux = , /* SCIF1_RXD */ + ; /* SCIF1_TXD */ + }; + + usb0_vbus_pins: usb0-vbus { + pinmux = ; /* USB0_VBUSEN */ + }; +}; + +&usb2_phy1 { + vbus-supply = <&vbus1>; +}; + +&vmmc { + gpios = <&pinctrl RZG2L_GPIO(0, 1) GPIO_ACTIVE_LOW>; +}; + +&vmpcie { + /* + * G2UL SoM v1.2 moved this IO from J7-28 (CON4-12) + * to J7-18 (this regulator). + * There are no known users of CON4-12 and thus no need + * to differentiate here between SoM v1.1 and v1.2. + */ + gpios = <&pinctrl RZG2L_GPIO(13, 4) GPIO_ACTIVE_HIGH>; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2ul-hummingboard-pulse-leds.dtso b/arch/arm64/boot/dts/renesas/rzg2ul-hummingboard-pulse-leds.dtso new file mode 100644 index 00000000000000..ab7f826c6e0398 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2ul-hummingboard-pulse-leds.dtso @@ -0,0 +1,48 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Overlay for SolidRun RZ/G2UL HummingBoard Pulse + * enabling LEDs D31-34 in exchange for on-SoM Bluetooth. + * + * Copyright 2025 Josua Mayer + */ + +/dts-v1/; +/plugin/; + +#include +#include +#include + +&leds { + led-1 { + label = "D31"; + color = ; + gpios = <&pinctrl RZG2L_GPIO(5, 4) GPIO_ACTIVE_LOW>; + default-state = "on"; + }; + + led-2 { + label = "D32"; + color = ; + gpios = <&pinctrl RZG2L_GPIO(5, 3) GPIO_ACTIVE_LOW>; + default-state = "on"; + }; + + led-3 { + label = "D33"; + color = ; + gpios = <&pinctrl RZG2L_GPIO(6, 0) GPIO_ACTIVE_LOW>; + default-state = "on"; + }; + + led-4 { + label = "D34"; + color = ; + gpios = <&pinctrl RZG2L_GPIO(6, 1) GPIO_ACTIVE_LOW>; + default-state = "on"; + }; +}; + +&scif2 { + status = "disabled"; +}; diff --git a/arch/arm64/boot/dts/renesas/rzg2ul-sr-som.dtsi b/arch/arm64/boot/dts/renesas/rzg2ul-sr-som.dtsi new file mode 100644 index 00000000000000..5e4d7861066dd5 --- /dev/null +++ b/arch/arm64/boot/dts/renesas/rzg2ul-sr-som.dtsi @@ -0,0 +1,418 @@ +// SPDX-License-Identifier: (GPL-2.0-only OR BSD-2-Clause) +/* + * Device Tree Source for the RZ/G2UL Solidrun SOM + * + * Copyright 2025 Josua Mayer + */ + +#include +#include + +/ { + aliases { + ethernet0 = ð0; + i2c0 = &i2c0; + i2c1 = &i2c1; + i2c2 = &i2c2; + mmc0 = &sdhi0; + mmc1 = &sdhi1; + rtc0 = &pmic; + serial0 = &scif0; + serial1 = &scif1; + serial2 = &scif2; + }; + + chosen { + stdout-path = "serial0:115200n8"; + }; + + memory@48000000 { + /* + * Physical RAM starts at 0x40000000, but Renesas RZ/G2 BSPs + * reserve and hide the first 128MB for TrustZone by default. + * + * Minimum size for this SoM is 512MB, and the bootloader patches + * the dtb according to the actual RAM size and reservations. + */ + reg = <0x0 0x48000000 0x0 0x18000000>; + device_type = "memory"; + }; + + sdhi0_mux: mux-controller-0 { + compatible = "gpio-mux"; + #mux-control-cells = <0>; + #mux-state-cells = <1>; + /* + * Mux switches SD0_DATA[0-3], SD0_CMD & SD0_CLK between + * on-SoM eMMC and board-to-board connector using one gpio: + * High = connector, Low = eMMC. + * Define active-high for logical mux state 0 = eMMC, 1 = connector. + */ + mux-gpios = <&pinctrl RZG2L_GPIO(2, 1) GPIO_ACTIVE_HIGH>; + }; + + reg_pmic_buck1: regulator-pmic-buck1 { + compatible = "regulator-fixed"; + regulator-name = "pmic-buck1"; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <1100000>; + regulator-min-microvolt = <1100000>; + }; + + reg_pmic_buck3: regulator-pmic-buck3 { + compatible = "regulator-fixed"; + regulator-name = "pmic-buck3"; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <1800000>; + regulator-min-microvolt = <1800000>; + }; + + reg_pmic_buck4: regulator-pmic-buck4 { + compatible = "regulator-fixed"; + regulator-name = "pmic-buck4"; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <3300000>; + regulator-min-microvolt = <3300000>; + }; + + reg_pmic_ldo1: regulator-pmic-ldo1 { + compatible = "regulator-gpio"; + regulator-name = "pmic-ldo1"; + gpios = <&pinctrl RZG2L_GPIO(13, 2) GPIO_ACTIVE_HIGH>; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <3300000>; + regulator-min-microvolt = <1800000>; + states = <3300000 1>, <1800000 0>; + }; + + reg_pmic_ldo2: regulator-pmic-ldo2 { + compatible = "regulator-fixed"; + regulator-name = "pmic-ldo2"; + /* + * This ldo can switch mmc host controller io voltage between + * 1.8V and 3.3V by assembly option of pull-up / pull-dow. + * Default assembly is 3.3V. + */ + regulator-min-microvolt = <3300000>; + regulator-always-on; + regulator-boot-on; + regulator-max-microvolt = <3300000>; + }; + + sdhi1_pwrseq: sdhi1-pwrseq { + compatible = "mmc-pwrseq-simple"; + reset-gpios = <&pinctrl RZG2L_GPIO(2, 2) GPIO_ACTIVE_LOW>; + }; + + /* 32.768kHz crystal */ + x2: x2-clock { + compatible = "fixed-clock"; + #clock-cells = <0>; + clock-frequency = <32768>; + }; +}; + +&ehci0 { + status = "okay"; +}; + +&ehci1 { + status = "okay"; +}; + +ð0 { + phy-handle = <&phy0>; + pinctrl-0 = <ð0_pins>; + pinctrl-names = "default"; + /* + * ravb driver does not configure mac internal delays for RZ/G2(UL/L/C), + * instead delays are added by the ADIN1200 phy driver. + */ + phy-mode = "rgmii-id"; + status = "okay"; + + phy0: ethernet-phy@0 { + reg = <0>; + /* Level interrupts are typical for phy but not supported on rz/g2l gpios. */ + interrupts-extended = <&pinctrl RZG2L_GPIO(4, 2) IRQ_TYPE_EDGE_FALLING>; + }; +}; + +&extal_clk { + clock-frequency = <24000000>; +}; + +&i2c0 { + clock-frequency = <400000>; + pinctrl-0 = <&i2c0_pins>; + pinctrl-names = "default"; + status = "okay"; + + eeprom: eeprom@50 { + compatible = "atmel,24c01"; + reg = <0x50>; + pagesize = <16>; + }; +}; + +&i2c1 { + clock-frequency = <400000>; + pinctrl-0 = <&i2c1_pins>; + pinctrl-names = "default"; + status = "okay"; +}; + +&i2c2 { + clock-frequency = <400000>; + pinctrl-0 = <&i2c2_pins>; + pinctrl-names = "default"; + status = "okay"; + + pmic: pmic@12 { + compatible = "renesas,raa215300"; + reg = <0x12>, <0x6f>; + reg-names = "main", "rtc"; + clocks = <&x2>; + clock-names = "xin"; + }; +}; + +&ohci0 { + status = "okay"; +}; + +&ohci1 { + status = "okay"; +}; + +&ostm1 { + status = "okay"; +}; + +&ostm2 { + status = "okay"; +}; + +&phyrst { + status = "okay"; +}; + +&pinctrl { + eth0_pins: eth0 { + pinmux = , /* ET0_LINKSTA */ + , /* ET0_MDC */ + , /* ET0_MDIO */ + , /* ET0_TXC */ + , /* ET0_TX_CTL */ + , /* ET0_TXD0 */ + , /* ET0_TXD1 */ + , /* ET0_TXD2 */ + , /* ET0_TXD3 */ + , /* ET0_RXC */ + , /* ET0_RX_CTL */ + , /* ET0_RXD0 */ + , /* ET0_RXD1 */ + , /* ET0_RXD2 */ + ; /* ET0_RXD3 */ + }; + + i2c0_pins: i2c0 { + input-enable; + pins = "RIIC0_SDA", "RIIC0_SCL"; + }; + + i2c1_pins: i2c1 { + input-enable; + pins = "RIIC1_SDA", "RIIC1_SCL"; + }; + + i2c2_pins: i2c2 { + pinmux = , /* RIIC2_SDA */ + ; /* RIIC2_SCL */ + }; + + qspi0_pins: qspi0 { + pins = "QSPI0_IO0", "QSPI0_IO1", "QSPI0_IO2", "QSPI0_IO3", + "QSPI0_SPCLK", "QSPI0_SSL"; + power-source = <1800>; + }; + + scif0_pins: scif0 { + pinmux = , /* SCIF0_TXD */ + ; /* SCIF0_RXD */ + }; + + scif2_pins: scif2 { + pinmux = , /* SCIF2_TXD */ + , /* SCIF2_RXD */ + , /* SCIF2_CTS# */ + ; /* SCIF2_RTS# */ + }; + + sdhi0_4bit_pins: sdhi0-4bit { + sd0_ctrl_dat03 { + pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3", + "SD0_CLK", "SD0_CMD"; + power-source = <3300>; + }; + + /* + * Pins 4-7 are hard-wired to eMMC with 1.8V IO voltage, + * with all pins sharing a single voltage domain. + * + * These pins are fixed-function lacking independently configurable + * pull-up, pull-down or output-disable. + * + * While unused, disable the input buffers to avoid sampling the still + * physically connected but unused eMMC lines. + * For output direction rely on correct controller bus-width configuration + * to avoid over-voltage to eMMC e.g. when operating at 3.3V for a 4-bit microSD. + */ + sd0_dat47 { + input-disable; + pins = "SD0_DATA4", "SD0_DATA5", "SD0_DATA6", "SD0_DATA7"; + }; + }; + + sdhi0_4bit_uhs_pins: sdhi0-4bit-uhs { + sd0_ctrl_dat03 { + pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3", + "SD0_CLK", "SD0_CMD"; + power-source = <1800>; + }; + + /* + * Pins 4-7 are hard-wired to eMMC with 1.8V IO voltage, + * with all pins sharing a single voltage domain. + * + * These pins are fixed-function, lacking configurable pull-up, pull-down or + * output-disable, while sharing the voltage domain with all SD0_* pins. + * + * While unused, disable the input buffers to avoid sampling the still + * physically connected but unused eMMC lines. + * For output direction rely on correct controller bus-width configuration + * to avoid over-voltage to eMMC e.g. when operating at 3.3V for a 4-bit microSD. + */ + sd0_dat47 { + input-disable; + pins = "SD0_DATA4", "SD0_DATA5", "SD0_DATA6", "SD0_DATA7"; + }; + }; + + sdhi0_8bit_uhs_pins: sdhi0-8bit-uhs { + sd0_ctrl { + pins = "SD0_CLK", "SD0_CMD"; + power-source = <1800>; + }; + + sd0_dat { + input-enable; + pins = "SD0_DATA0", "SD0_DATA1", "SD0_DATA2", "SD0_DATA3", + "SD0_DATA4", "SD0_DATA5", "SD0_DATA6", "SD0_DATA7"; + }; + }; + + sdhi0_cd_pins: sdhi0-cd { + pinmux = ; /* SD0_CD */ + }; + + /* SD0_RST is only routed to eMMC which uses fixed 1.8V IO voltage */ + sdhi0_rst_pins: sdhi0-rst { + pins = "SD0_RST#"; + power-source = <1800>; + }; + + sdhi1_pins: sdhi1 { + pins = "SD1_DATA0", "SD1_DATA1", "SD1_DATA2", "SD1_DATA3", + "SD1_CLK", "SD1_CMD"; + power-source = <3300>; + }; + + spi1_pins: spi1 { + pinmux = , /* RSPI1_MISO */ + , /* RSPI1_MOSI */ + ; /* RSPI1_CK */ + }; + + spi1_cs_pins: spi1-cs { + pinmux = ; /* RSPI1_SSL */ + }; +}; + +&sbc { + pinctrl-0 = <&qspi0_pins>; + pinctrl-names = "default"; + status = "okay"; + + flash@0 { + compatible = "winbond,w25q80bl", "jedec,spi-nor"; + reg = <0>; + spi-max-frequency = <50000000>; + m25p,fast-read; + }; +}; + +&scif0 { + pinctrl-0 = <&scif0_pins>; + pinctrl-names = "default"; + status = "okay"; +}; + +&scif2 { + pinctrl-0 = <&scif2_pins>; + pinctrl-names = "default"; + uart-has-rtscts; + status = "okay"; + + bluetooth { + /* + * The BT is based on BCM4343A2. Murata 1YN variant seems to + * require driving RTS on open. + * Use compatible bcm43438 to accomplish this. + */ + compatible = "brcm,bcm43438-bt"; + max-speed = <115200>; + shutdown-gpios = <&pinctrl RZG2L_GPIO(2, 3) GPIO_ACTIVE_HIGH>; + vbat-supply = <®_pmic_buck4>; + vddio-supply = <®_pmic_buck4>; + }; +}; + +/* WiFi */ +&sdhi1 { + /* Murata 1YN max rate is 50MHz */ + max-frequency = <50000000>; + bus-width = <4>; + cap-sdio-irq; + mmc-pwrseq = <&sdhi1_pwrseq>; + non-removable; + no-1-8-v; + no-sd; + pinctrl-0 = <&sdhi1_pins>; + pinctrl-names = "default"; + vmmc-supply = <®_pmic_buck4>; + /* + * Host controller IO voltage is provided from reg_pmic_ldo2, + * WiFi module IO voltage from reg_pmic_buck4. + * Neither is configurable at run-time so either can be set here. + */ + vqmmc-supply = <®_pmic_ldo2>; + status = "okay"; +}; + +&usb2_phy0 { + vbus-supply = <&usb0_vbus_otg>; + status = "okay"; +}; + +&usb2_phy1 { + status = "okay"; +}; + +&wdt0 { + status = "okay"; +}; From 7849521aaac26b17820642010ab7f7a222ec73d1 Mon Sep 17 00:00:00 2001 From: Junhui Liu Date: Thu, 14 May 2026 17:27:21 +0800 Subject: [PATCH 0186/1012] riscv: dts: anlogic: add clocks and CRU for DR1V90 Add clocks and introduce the CRU (Clock and Reset) unit node for Anlogic DR1V90 SoC, providing both clock and reset support. The DR1V90 SoC uses three external clocks: - A crystal oscillator as the main system clock. - Two optional external clocks (via IO) for the CAN and WDT modules. The main crystal oscillator frequency is board-dependent. For the dr1v90-mlkpai-fs01 board, a 33.33 MHz oscillator is used and defined accordingly. Signed-off-by: Junhui Liu Signed-off-by: Brian Masney --- .../boot/dts/anlogic/dr1v90-mlkpai-fs01.dts | 4 ++ arch/riscv/boot/dts/anlogic/dr1v90.dtsi | 40 ++++++++++++++++++- 2 files changed, 42 insertions(+), 2 deletions(-) diff --git a/arch/riscv/boot/dts/anlogic/dr1v90-mlkpai-fs01.dts b/arch/riscv/boot/dts/anlogic/dr1v90-mlkpai-fs01.dts index 597407655efd2e..af78f1a4eeccb1 100644 --- a/arch/riscv/boot/dts/anlogic/dr1v90-mlkpai-fs01.dts +++ b/arch/riscv/boot/dts/anlogic/dr1v90-mlkpai-fs01.dts @@ -23,6 +23,10 @@ }; }; +&osc { + clock-frequency = <33333333>; +}; + &uart1 { status = "okay"; }; diff --git a/arch/riscv/boot/dts/anlogic/dr1v90.dtsi b/arch/riscv/boot/dts/anlogic/dr1v90.dtsi index 9fe183f5f5c8d3..574c6608aef014 100644 --- a/arch/riscv/boot/dts/anlogic/dr1v90.dtsi +++ b/arch/riscv/boot/dts/anlogic/dr1v90.dtsi @@ -3,6 +3,9 @@ * Copyright (C) 2025 Junhui Liu */ +#include +#include + /dts-v1/; / { #address-cells = <2>; @@ -40,6 +43,26 @@ }; }; + clocks { + can_ext: clock-ext-can { + compatible = "fixed-clock"; + clock-output-names = "can_ext"; + #clock-cells = <0>; + }; + + osc: clock-osc { + compatible = "fixed-clock"; + clock-output-names = "osc"; + #clock-cells = <0>; + }; + + wdt_ext: clock-ext-wdt { + compatible = "fixed-clock"; + clock-output-names = "wdt_ext"; + #clock-cells = <0>; + }; + }; + soc { compatible = "simple-bus"; interrupt-parent = <&plic>; @@ -81,21 +104,34 @@ uart0: serial@f8400000 { compatible = "anlogic,dr1v90-uart", "snps,dw-apb-uart"; reg = <0x0 0xf8400000 0x0 0x1000>; - clock-frequency = <50000000>; + clocks = <&cru CLK_IO_400M_DIV8>, <&cru CLK_CPU_1X>; + clock-names = "baudclk", "apb_pclk"; interrupts = <71>; reg-io-width = <4>; reg-shift = <2>; + resets = <&cru RESET_UART0>; status = "disabled"; }; uart1: serial@f8401000 { compatible = "anlogic,dr1v90-uart", "snps,dw-apb-uart"; reg = <0x0 0xf8401000 0x0 0x1000>; - clock-frequency = <50000000>; + clocks = <&cru CLK_IO_400M_DIV8>, <&cru CLK_CPU_1X>; + clock-names = "baudclk", "apb_pclk"; interrupts = <72>; reg-io-width = <4>; reg-shift = <2>; + resets = <&cru RESET_UART1>; status = "disabled"; }; + + cru: clock-controller@f8801000 { + compatible = "anlogic,dr1v90-cru"; + reg = <0x0 0xf8801000 0 0x400>; + clocks = <&osc>, <&can_ext>, <&wdt_ext>; + clock-names = "osc", "can_ext", "wdt_ext"; + #clock-cells = <1>; + #reset-cells = <1>; + }; }; }; From 5795c6059d10afbc0fbcfe3cd95676be69b30bcb Mon Sep 17 00:00:00 2001 From: Junhui Liu Date: Thu, 14 May 2026 17:27:22 +0800 Subject: [PATCH 0187/1012] MAINTAINERS: Add Anlogic DR1V90 CRU driver entry Add a MAINTAINERS entry for the Anlogic DR1V90 Clock and Reset Unit (CRU) drivers. Signed-off-by: Junhui Liu Signed-off-by: Brian Masney --- MAINTAINERS | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/MAINTAINERS b/MAINTAINERS index 3a729f7326f081..6977bc6245b022 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -2010,6 +2010,15 @@ M: Jiaxun Yang S: Supported F: drivers/rtc/rtc-goldfish.c +ANLOGIC DR1V90 CRU DRIVER +M: Junhui Liu +S: Maintained +F: Documentation/devicetree/bindings/clock/anlogic,dr1v90-cru.yaml +F: drivers/clk/anlogic/cru?dr1* +F: drivers/reset/reset-dr1v90.c +F: include/dt-bindings/clock/anlogic,dr1v90-cru.h +F: include/dt-bindings/reset/anlogic,dr1v90-cru.h + AOA (Apple Onboard Audio) ALSA DRIVER M: Johannes Berg L: linuxppc-dev@lists.ozlabs.org From 1403ae0137015007b6a18fef2ff5effcf2d2c368 Mon Sep 17 00:00:00 2001 From: Badal Nilawar Date: Tue, 29 Sep 2026 22:48:24 +0530 Subject: [PATCH 0188/1012] drm/xe/cri: Expose device UID through sysfs Expose a read-only sysfs attribute, device_uid, that reports the GPU SoC's unique identifier. Bspec: 53048, 53049 Assisted-by: Claude:claude-opus-4.8 Signed-off-by: Badal Nilawar Reviewed-by: Rodrigo Vivi Link: https://patch.msgid.link/20260929171823.3811737-2-badal.nilawar@intel.com Signed-off-by: Rodrigo Vivi --- .../ABI/testing/sysfs-driver-intel-xe-gpu | 10 ++++++ drivers/gpu/drm/xe/regs/xe_regs.h | 2 ++ drivers/gpu/drm/xe/xe_device.c | 9 +++++ drivers/gpu/drm/xe/xe_device_sysfs.c | 36 +++++++++++++++++++ drivers/gpu/drm/xe/xe_device_types.h | 5 +++ drivers/gpu/drm/xe/xe_pci.c | 2 ++ drivers/gpu/drm/xe/xe_pci_types.h | 1 + 7 files changed, 65 insertions(+) create mode 100644 Documentation/ABI/testing/sysfs-driver-intel-xe-gpu diff --git a/Documentation/ABI/testing/sysfs-driver-intel-xe-gpu b/Documentation/ABI/testing/sysfs-driver-intel-xe-gpu new file mode 100644 index 00000000000000..4dc4afc3961d9d --- /dev/null +++ b/Documentation/ABI/testing/sysfs-driver-intel-xe-gpu @@ -0,0 +1,10 @@ +What: /sys/bus/pci/drivers/xe/.../device_uid +Date: October 2026 +KernelVersion: 7.4 +Contact: intel-xe@lists.freedesktop.org +Description: + RO. Unique 64-bit identifier of the GPU SoC device, exposed as + hexadecimal value. + + This sysfs file is present only on supported Intel Xe platforms. + Accessible only to users with administrative privileges. diff --git a/drivers/gpu/drm/xe/regs/xe_regs.h b/drivers/gpu/drm/xe/regs/xe_regs.h index ef4746b7b5d342..437485b5a0af9b 100644 --- a/drivers/gpu/drm/xe/regs/xe_regs.h +++ b/drivers/gpu/drm/xe/regs/xe_regs.h @@ -30,6 +30,8 @@ #define XEHP_MTCFG_ADDR XE_REG(0x101800) #define TILE_COUNT REG_GENMASK(15, 8) +#define CRI_DEVICE_UID XE_REG(0x102008) + #define GGC XE_REG(0x108040) #define GMS_MASK REG_GENMASK(15, 8) #define GGMS_MASK REG_GENMASK(7, 6) diff --git a/drivers/gpu/drm/xe/xe_device.c b/drivers/gpu/drm/xe/xe_device.c index 44975e7823be2b..e1979bbcf7bc60 100644 --- a/drivers/gpu/drm/xe/xe_device.c +++ b/drivers/gpu/drm/xe/xe_device.c @@ -672,6 +672,7 @@ static void vf_update_device_info(struct xe_device *xe) xe->info.skip_guc_pc = 1; xe->info.skip_pcode = 1; xe->info.has_drm_ras = false; + xe->info.has_device_uid = false; } static int xe_device_vram_alloc(struct xe_device *xe) @@ -872,6 +873,12 @@ static int xe_debug_page_size_alloc_ctrl_init(struct xe_device *xe) } #endif +static void xe_uid_probe(struct xe_device *xe) +{ + if (xe->info.has_device_uid) + xe->device_uid = xe_mmio_read64_2x32(xe_root_tile_mmio(xe), CRI_DEVICE_UID); +} + int xe_device_probe(struct xe_device *xe) { struct xe_tile *tile; @@ -879,6 +886,8 @@ int xe_device_probe(struct xe_device *xe) int err; u8 id; + xe_uid_probe(xe); + xe_pat_init_early(xe); err = xe_sriov_init(xe); diff --git a/drivers/gpu/drm/xe/xe_device_sysfs.c b/drivers/gpu/drm/xe/xe_device_sysfs.c index a73e0e957cb0bb..18783a286a830e 100644 --- a/drivers/gpu/drm/xe/xe_device_sysfs.c +++ b/drivers/gpu/drm/xe/xe_device_sysfs.c @@ -8,6 +8,7 @@ #include #include +#include "regs/xe_regs.h" #include "xe_device.h" #include "xe_device_sysfs.h" #include "xe_mmio.h" @@ -264,6 +265,35 @@ static const struct attribute_group auto_link_downgrade_attr_group = { .attrs = auto_link_downgrade_attrs, }; +/** + * DOC: Device Unique ID + * + * On supported platforms, Xe driver exposes a unique 64-bit GPU SOC + * device identifier through the 'device_uid' sysfs entry. + * + * See Documentation/ABI/testing/sysfs-driver-intel-xe-gpu for the ABI + * specification. + */ + +static ssize_t +device_uid_show(struct device *dev, struct device_attribute *attr, char *buf) +{ + struct pci_dev *pdev = to_pci_dev(dev); + struct xe_device *xe = pdev_to_xe_device(pdev); + + return sysfs_emit(buf, "0x%016llx\n", xe->device_uid); +} +static DEVICE_ATTR_ADMIN_RO(device_uid); + +static struct attribute *device_uid_attrs[] = { + &dev_attr_device_uid.attr, + NULL +}; + +static const struct attribute_group device_uid_attr_group = { + .attrs = device_uid_attrs, +}; + int xe_device_sysfs_init(struct xe_device *xe) { struct device *dev = xe->drm.dev; @@ -285,5 +315,11 @@ int xe_device_sysfs_init(struct xe_device *xe) return ret; } + if (xe->info.has_device_uid) { + ret = devm_device_add_group(dev, &device_uid_attr_group); + if (ret) + return ret; + } + return 0; } diff --git a/drivers/gpu/drm/xe/xe_device_types.h b/drivers/gpu/drm/xe/xe_device_types.h index 448d1e1cd333c8..4e0e1bc040c19e 100644 --- a/drivers/gpu/drm/xe/xe_device_types.h +++ b/drivers/gpu/drm/xe/xe_device_types.h @@ -179,6 +179,8 @@ struct xe_device { u8 has_cached_pt:1; /** @info.has_device_atomics_on_smem: Supports device atomics on SMEM */ u8 has_device_atomics_on_smem:1; + /** @info.has_device_uid: Device supports unique 64-bit GPU SOC ID */ + u8 has_device_uid:1; /** @info.has_drm_ras: Device supports drm_ras (Reliability, Availability, Serviceability) */ u8 has_drm_ras:1; /** @info.has_fan_control: Device supports fan control */ @@ -265,6 +267,9 @@ struct xe_device { bool oob_initialized; } wa_active; + /** @device_uid: unique 64-bit GPU SOC identifier */ + u64 device_uid; + /** @survivability: survivability information for device */ struct xe_survivability survivability; diff --git a/drivers/gpu/drm/xe/xe_pci.c b/drivers/gpu/drm/xe/xe_pci.c index 620f8a75086f17..3719427dd17b89 100644 --- a/drivers/gpu/drm/xe/xe_pci.c +++ b/drivers/gpu/drm/xe/xe_pci.c @@ -472,6 +472,7 @@ static const struct xe_device_desc cri_desc = { PLATFORM(CRESCENTISLAND), .dma_mask_size = 52, .has_display = false, + .has_device_uid = true, .has_drm_ras = true, .has_flat_ccs = false, .has_gsc_nvm = 1, @@ -792,6 +793,7 @@ static int xe_info_init_early(struct xe_device *xe, xe->info.is_dgfx = desc->is_dgfx; xe->info.has_cached_pt = desc->has_cached_pt; + xe->info.has_device_uid = desc->has_device_uid; xe->info.has_drm_ras = desc->has_drm_ras; xe->info.has_fan_control = desc->has_fan_control; /* runtime fusing may force flat_ccs to disabled later */ diff --git a/drivers/gpu/drm/xe/xe_pci_types.h b/drivers/gpu/drm/xe/xe_pci_types.h index 71068cdb35589b..448bbe5ba1d77b 100644 --- a/drivers/gpu/drm/xe/xe_pci_types.h +++ b/drivers/gpu/drm/xe/xe_pci_types.h @@ -40,6 +40,7 @@ struct xe_device_desc { u8 has_cached_pt:1; u8 has_display:1; + u8 has_device_uid:1; u8 has_drm_ras:1; u8 has_fan_control:1; u8 has_flat_ccs:1; From e7ac8fd885298225076abd92b82c1f31a6bc8320 Mon Sep 17 00:00:00 2001 From: Daehyeon Ko <4ncienth@gmail.com> Date: Fri, 25 Sep 2026 17:05:48 +0900 Subject: [PATCH 0189/1012] assoc_array: discard shortcut when collapsing a leaf-only node assoc_array_delete() can collapse a subtree into a node that contains only leaves while retaining the shortcut that led to it. If that node later fills, all_leaves_cluster_together replaces it with another shortcut. The first shortcut then points directly to the second one. assoc_array_apply_edit() publishes this topology and propagates branch counts from the new child node. It skips the inner shortcut, encounters the outer shortcut where it requires a node and triggers the BUG_ON(). Linux v7.2 and v6.12.105 are affected. The same root remains at the base-commit below and in every supported stable branch checked down to 5.10. It requires CONFIG_KEYS, but no capability, user namespace or race. A UID/GID 1000 process produced: CONTROL_BEGIN mode=exact uid=1000 gid=1000 CONTROL_CapEff: 0000000000000000 kernel BUG at lib/assoc_array.c:1388! Oops: invalid opcode: 0000 [#1] SMP KASAN NOPTI CPU: 0 UID: 1000 PID: 154 Comm: exploit RIP: assoc_array_apply_edit+0x4aa/0x690 Call Trace: __key_link __key_instantiate_and_link __key_create_or_update __do_sys_add_key Kernel panic - not syncing: Fatal exception When deletion produces a leaf-only node, bypass its preceding shortcut as garbage collection already does. Retire the shortcut and old node together after an RCU grace period; reused leaves keep their references and the deleted leaf is still freed separately. The exact trigger reached the BUG in 3/3 unmodified v7.2 KASAN boots and completed cleanly in 3/3 fixed boots. Fixed v6.12.105 also passed 3/3. A source reproducer is available privately on request. Fixes: 3cb989501c26 ("Add a generic associative array implementation.") Cc: stable@vger.kernel.org # v5.10+ Assisted-by: LLM Signed-off-by: Daehyeon Ko <4ncienth@gmail.com> Link: https://lore.kernel.org/r/20260925080548.2505640-1-4ncienth@gmail.com Reviewed-by: Jarkko Sakkinen Signed-off-by: Jarkko Sakkinen --- lib/assoc_array.c | 36 ++++++++++++++++++++++-------------- 1 file changed, 22 insertions(+), 14 deletions(-) diff --git a/lib/assoc_array.c b/lib/assoc_array.c index b6c9723e12ced5..841dfe07dc962a 100644 --- a/lib/assoc_array.c +++ b/lib/assoc_array.c @@ -1210,8 +1210,22 @@ struct assoc_array_edit *assoc_array_delete(struct assoc_array *array, goto enomem; edit->new_meta[0] = assoc_array_node_to_ptr(new_n0); - new_n0->back_pointer = node->back_pointer; - new_n0->parent_slot = node->parent_slot; + /* A shortcut above a leaf-only node is redundant. Drop it as + * GC does so that a later split can't create two shortcuts in a row. + */ + ptr = node->back_pointer; + if (assoc_array_ptr_is_shortcut(ptr)) { + struct assoc_array_shortcut *s = + assoc_array_ptr_to_shortcut(ptr); + + new_n0->back_pointer = s->back_pointer; + new_n0->parent_slot = s->parent_slot; + edit->excised_subtree = ptr; + } else { + new_n0->back_pointer = ptr; + new_n0->parent_slot = node->parent_slot; + edit->excised_subtree = assoc_array_node_to_ptr(node); + } new_n0->nr_leaves_on_branch = node->nr_leaves_on_branch; edit->adjust_count_on = new_n0; @@ -1225,21 +1239,15 @@ struct assoc_array_edit *assoc_array_delete(struct assoc_array *array, pr_devel("collapsed %d,%lu\n", collapse.slot, new_n0->nr_leaves_on_branch); BUG_ON(collapse.slot != new_n0->nr_leaves_on_branch - 1); - if (!node->back_pointer) { + if (!new_n0->back_pointer) { edit->set[1].ptr = &array->root; - } else if (assoc_array_ptr_is_leaf(node->back_pointer)) { - BUG(); - } else if (assoc_array_ptr_is_node(node->back_pointer)) { - struct assoc_array_node *p = - assoc_array_ptr_to_node(node->back_pointer); - edit->set[1].ptr = &p->slots[node->parent_slot]; - } else if (assoc_array_ptr_is_shortcut(node->back_pointer)) { - struct assoc_array_shortcut *s = - assoc_array_ptr_to_shortcut(node->back_pointer); - edit->set[1].ptr = &s->next_node; + } else { + struct assoc_array_node *p; + + p = assoc_array_ptr_to_node(new_n0->back_pointer); + edit->set[1].ptr = &p->slots[new_n0->parent_slot]; } edit->set[1].to = assoc_array_node_to_ptr(new_n0); - edit->excised_subtree = assoc_array_node_to_ptr(node); } } From 37db375ad90467d7e3ce4e450b3d1c4f297d4ed3 Mon Sep 17 00:00:00 2001 From: Zhan Xusheng Date: Wed, 16 Sep 2026 10:32:49 +0800 Subject: [PATCH 0190/1012] vdso/math64: Use OPTIMIZER_HIDE_VAR() in __iter_div_u64_rem() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit __iter_div_u64_rem() divides by repeated subtraction, because its callers only ever produce a small quotient. A barrier inside the subtraction loop keeps the compiler from replacing the loop with a division: asm("" : "+rm"(dividend)); The "rm" constraint permits a memory operand. clang picks it and spills the dividend inside the loop, at 32-bit and 64-bit alike. The helper sits on the clock_gettime() fast path through vdso_set_timespec(), so the spill is not free: with clang 18 the x86 vDSO is 64 bytes larger in vdso64 text and 112 bytes larger in vdso32 text than with a register-only barrier. Use OPTIMIZER_HIDE_VAR(), which is the register-only form of the same barrier and the form the rest of the kernel uses. gcc 13 generates identical code either way, and neither compiler turns the loop into a division at either width. Put the function signature on one line while touching it, and reformat the comment, which described the asm() that is now gone. Signed-off-by: Zhan Xusheng Signed-off-by: Thomas Gleixner Reviewed-by: Thomas Weißschuh Link: https://patch.msgid.link/20260916023252.418473-2-zhanxusheng@xiaomi.com --- include/vdso/math64.h | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/include/vdso/math64.h b/include/vdso/math64.h index 22ae212f8b28fe..c628d6cf447c8a 100644 --- a/include/vdso/math64.h +++ b/include/vdso/math64.h @@ -2,15 +2,16 @@ #ifndef __VDSO_MATH64_H #define __VDSO_MATH64_H -static __always_inline u32 -__iter_div_u64_rem(u64 dividend, u32 divisor, u64 *remainder) +static __always_inline u32 __iter_div_u64_rem(u64 dividend, u32 divisor, u64 *remainder) { u32 ret = 0; while (dividend >= divisor) { - /* The following asm() prevents the compiler from - optimising this loop into a modulo operation. */ - asm("" : "+rm"(dividend)); + /* + * Prevent the compiler from optimising this loop into a + * modulo operation. + */ + OPTIMIZER_HIDE_VAR(dividend); dividend -= divisor; ret++; From 5c82c995b63748cfa5026f4ca813eee754263f62 Mon Sep 17 00:00:00 2001 From: Zhan Xusheng Date: Wed, 16 Sep 2026 10:32:50 +0800 Subject: [PATCH 0191/1012] vdso/math64: Add and use __iter_div64_u64_rem() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The vDSO basetimes for CLOCK_MONOTONIC and CLOCK_BOOTTIME are stored in the scaled nanoseconds of tkr_mono, so normalising them requires a division by NSEC_PER_SEC << shift. That divisor does not fit the u32 parameter of __iter_div_u64_rem(), so update_vdso_time_data() open-codes the same iterative division twice. Repeated subtraction is the appropriate form at these two sites because the quotient never exceeds one. accumulate_nsecs_to_secs() keeps xtime_nsec below one scaled second, and the offset added to it is a normalised timespec64 fraction, so the dividend stays below twice the divisor. That bound holds on every architecture, which matters more here than the cost of a division on any particular one. Add __iter_div64_u64_rem(), the u64-divisor counterpart of __iter_div_u64_rem(), and use it at both sites. Return the quotient as a u32 for consistency with the u32-divisor variant; the bound above leaves no use for a wider type. Store the remainder directly into the basetime, as the coarse clocks already do. The CLOCK_BOOTTIME copy of the CLOCK_MONOTONIC values then takes both fields from the same place. Folding the two open-coded loops into one inlined helper removes 16 bytes of vsyscall.o text on x86-64 with gcc 13. No functional change. Suggested-by: David Laight Signed-off-by: Zhan Xusheng Signed-off-by: Thomas Gleixner Reviewed-by: Thomas Weißschuh Link: https://patch.msgid.link/20260916023252.418473-3-zhanxusheng@xiaomi.com --- include/vdso/math64.h | 20 ++++++++++++++++++++ kernel/time/vsyscall.c | 16 +++++----------- 2 files changed, 25 insertions(+), 11 deletions(-) diff --git a/include/vdso/math64.h b/include/vdso/math64.h index c628d6cf447c8a..55b45f5cf615ac 100644 --- a/include/vdso/math64.h +++ b/include/vdso/math64.h @@ -22,6 +22,26 @@ static __always_inline u32 __iter_div_u64_rem(u64 dividend, u32 divisor, u64 *re return ret; } +static __always_inline u32 __iter_div64_u64_rem(u64 dividend, u64 divisor, u64 *remainder) +{ + u32 ret = 0; + + while (dividend >= divisor) { + /* + * Prevent the compiler from optimising this loop into a + * modulo operation. + */ + OPTIMIZER_HIDE_VAR(dividend); + + dividend -= divisor; + ret++; + } + + *remainder = dividend; + + return ret; +} + #if defined(CONFIG_ARCH_SUPPORTS_INT128) && defined(__SIZEOF_INT128__) #ifndef mul_u64_u32_add_u64_shr diff --git a/kernel/time/vsyscall.c b/kernel/time/vsyscall.c index aa59919b8f2c23..f43dd3f4744b50 100644 --- a/kernel/time/vsyscall.c +++ b/kernel/time/vsyscall.c @@ -41,14 +41,12 @@ static inline void update_vdso_time_data(struct vdso_time_data *vdata, struct ti nsec = tk->tkr_mono.xtime_nsec; nsec += ((u64)tk->wall_to_monotonic.tv_nsec << tk->tkr_mono.shift); - while (nsec >= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift)) { - nsec -= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift); - vdso_ts->sec++; - } - vdso_ts->nsec = nsec; + vdso_ts->sec += __iter_div64_u64_rem(nsec, (u64)NSEC_PER_SEC << tk->tkr_mono.shift, + &vdso_ts->nsec); /* Copy MONOTONIC time for BOOTTIME */ sec = vdso_ts->sec; + nsec = vdso_ts->nsec; /* Add the boot offset */ sec += tk->monotonic_to_boot.tv_sec; nsec += (u64)tk->monotonic_to_boot.tv_nsec << tk->tkr_mono.shift; @@ -56,12 +54,8 @@ static inline void update_vdso_time_data(struct vdso_time_data *vdata, struct ti /* CLOCK_BOOTTIME */ vdso_ts = &vc[CS_HRES_COARSE].basetime[CLOCK_BOOTTIME]; vdso_ts->sec = sec; - - while (nsec >= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift)) { - nsec -= (((u64)NSEC_PER_SEC) << tk->tkr_mono.shift); - vdso_ts->sec++; - } - vdso_ts->nsec = nsec; + vdso_ts->sec += __iter_div64_u64_rem(nsec, (u64)NSEC_PER_SEC << tk->tkr_mono.shift, + &vdso_ts->nsec); /* CLOCK_MONOTONIC_RAW */ vdso_ts = &vc[CS_RAW].basetime[CLOCK_MONOTONIC_RAW]; From 4e6aa9d672875aaa20cfb64ffb6fcd79e5165522 Mon Sep 17 00:00:00 2001 From: Zhan Xusheng Date: Wed, 16 Sep 2026 10:32:51 +0800 Subject: [PATCH 0192/1012] vdso/vsyscall: Keep the CLOCK_AUX base scaled MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The vDSO basetime of a clock is stored in the scaled nanoseconds of tkr_mono, so that the reader can floor the base and the cycle delta together in vdso_calc_ns(). vdso_time_update_aux() instead shifts the base down to nanoseconds, adds the offset, and shifts it back up, which zeroes the fractional nanoseconds of xtime_nsec. The reader then floors the base and the delta separately: ktime_get_aux(): base + ((delta * mult + xtime_nsec) >> shift) vdso: base + (xtime_nsec >> shift) + ((delta * mult) >> shift) Since floor(a) + floor(b) <= floor(a + b), the vDSO reports 0 or 1 ns below the syscall for the same clock. It is not a monotonicity problem: across an update the step is floor(a + d) - floor(a) - floor(d), which is 0 or 1, never negative. Add the offset in scaled nanoseconds as the other high resolution clocks do, and normalise with __iter_div64_u64_rem() so that the stored base stays below one second and the userspace fast-path does not iterate more in __iter_div_u64_rem(). Only the sub-second field changes. (a + (b << shift)) >> shift is exactly (a >> shift) + b, so the seconds carried out of the normalisation are the same as before; what the old form dropped was the low shift bits of the remainder. monotonic_to_aux.tv_nsec is a normalised timespec64 fraction, so it stays below NSEC_PER_SEC even for a negative offset, and the sum stays below 2 * (NSEC_PER_SEC << shift). The largest shift clocks_calc_mult_shift() can pick is 32, which makes that 8.6e18 against a u64 limit of 1.8e19. Fixes: 380b84e168e5 ("vdso/vsyscall: Update auxiliary clock data in the datapage") Signed-off-by: Zhan Xusheng Signed-off-by: Thomas Gleixner Reviewed-by: Thomas Weißschuh Link: https://patch.msgid.link/20260916023252.418473-4-zhanxusheng@xiaomi.com --- kernel/time/vsyscall.c | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/kernel/time/vsyscall.c b/kernel/time/vsyscall.c index f43dd3f4744b50..0e4b499328c082 100644 --- a/kernel/time/vsyscall.c +++ b/kernel/time/vsyscall.c @@ -155,11 +155,11 @@ void vdso_time_update_aux(struct timekeeper *tk) vdso_ts->sec = tk->xtime_sec + tk->monotonic_to_aux.tv_sec; - nsec = tk->tkr_mono.xtime_nsec >> tk->tkr_mono.shift; - nsec += tk->monotonic_to_aux.tv_nsec; - vdso_ts->sec += __iter_div_u64_rem(nsec, NSEC_PER_SEC, &nsec); - nsec = nsec << tk->tkr_mono.shift; - vdso_ts->nsec = nsec; + nsec = tk->tkr_mono.xtime_nsec; + nsec += (u64)tk->monotonic_to_aux.tv_nsec << tk->tkr_mono.shift; + vdso_ts->sec += __iter_div64_u64_rem(nsec, + (u64)NSEC_PER_SEC << tk->tkr_mono.shift, + &vdso_ts->nsec); } __arch_update_vdso_clock(vc); From ca46a07c4b68246c4e400ce3649ab7228a878d8f Mon Sep 17 00:00:00 2001 From: Zhan Xusheng Date: Wed, 16 Sep 2026 10:32:52 +0800 Subject: [PATCH 0193/1012] vdso/gettimeofday: Assert that the clockid fits into the u32 bitmask MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit __cvdso_clock_gettime_common() and __cvdso_clock_getres_common() convert the clockid into a bitmask and match it against VDSO_HRES, VDSO_COARSE, VDSO_RAW and VDSO_AUX: msk = 1U << clock; vdso_clockid_valid() rejects anything above CLOCK_AUX_LAST beforehand, and CLOCK_AUX_LAST is 23, so the shift count is in range. Nothing records that dependency though. Raising MAX_AUX_CLOCKS beyond 16 moves CLOCK_AUX_LAST to 32 and makes the shift undefined. Add a BUILD_BUG_ON() at both conversion sites. The condition is on a function parameter rather than a constant, so it relies on the compiler deriving the range from the vdso_clockid_valid() bail-out above it. gcc 13 and clang 18 both do: the x86 vdso64 and vdso32 builds stay clean, and raising MAX_AUX_CLOCKS to 17 trips the assert. Suggested-by: Thomas Weißschuh Signed-off-by: Zhan Xusheng Signed-off-by: Thomas Gleixner Reviewed-by: Thomas Weißschuh Link: https://patch.msgid.link/20260916023252.418473-5-zhanxusheng@xiaomi.com --- lib/vdso/gettimeofday.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/lib/vdso/gettimeofday.c b/lib/vdso/gettimeofday.c index f7a591aba59f11..ef4dcc61448916 100644 --- a/lib/vdso/gettimeofday.c +++ b/lib/vdso/gettimeofday.c @@ -285,6 +285,7 @@ __cvdso_clock_gettime_common(const struct vdso_time_data *vd, clockid_t clock, * Convert the clockid to a bitmask and use it to check which * clocks are handled in the VDSO directly. */ + BUILD_BUG_ON(clock >= BITS_PER_TYPE(msk)); msk = 1U << clock; if (likely(msk & VDSO_HRES)) vc = &vc[CS_HRES_COARSE]; @@ -438,6 +439,7 @@ bool __cvdso_clock_getres_common(const struct vdso_time_data *vd, clockid_t cloc * Convert the clockid to a bitmask and use it to check which * clocks are handled in the VDSO directly. */ + BUILD_BUG_ON(clock >= BITS_PER_TYPE(msk)); msk = 1U << clock; if (msk & (VDSO_HRES | VDSO_RAW)) { /* From 4174f803c128778ceeb382f1258dbb60e1f16c9a Mon Sep 17 00:00:00 2001 From: Alan Previn Date: Wed, 23 Sep 2026 11:36:02 -0700 Subject: [PATCH 0194/1012] drm/xe/gsc: Define GSC for NVL-S NVL-S is identified by GSC major version 108. The compatibility version is 0.0. Signed-off-by: Alan Previn Reviewed-by: Julia Filipchuk Link: https://patch.msgid.link/20260923183600.1186943-5-alan.previn.teres.alexis@intel.com Signed-off-by: Rodrigo Vivi --- drivers/gpu/drm/xe/xe_uc_fw.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/gpu/drm/xe/xe_uc_fw.c b/drivers/gpu/drm/xe/xe_uc_fw.c index a8e6f18cc9b4aa..f61910431cf25e 100644 --- a/drivers/gpu/drm/xe/xe_uc_fw.c +++ b/drivers/gpu/drm/xe/xe_uc_fw.c @@ -141,6 +141,7 @@ struct fw_blobs_by_type { /* for the GSC FW we match the compatibility version and not the release one */ #define XE_GSC_FIRMWARE_DEFS(fw_def, major_ver) \ + fw_def(NOVALAKE_S, GT_TYPE_ANY, major_ver(xe, gsc, nvl, 108, 0, 0)) \ fw_def(PANTHERLAKE, GT_TYPE_ANY, major_ver(xe, gsc, ptl, 105, 1, 0)) \ fw_def(LUNARLAKE, GT_TYPE_ANY, major_ver(xe, gsc, lnl, 104, 1, 0)) \ fw_def(METEORLAKE, GT_TYPE_ANY, major_ver(i915, gsc, mtl, 102, 1, 0)) From 97abba17af9820915f04e3243d585eebc7bdeabb Mon Sep 17 00:00:00 2001 From: Alan Previn Date: Wed, 23 Sep 2026 11:36:03 -0700 Subject: [PATCH 0195/1012] drm/xe/nvl-s: Enable PXP for NVL-S Now that the GSC FW is defined, we can enable PXP for NVL-S. The feature will only be turned on if the binary is found on disk. Signed-off-by: Alan Previn Reviewed-by: Julia Filipchuk Link: https://patch.msgid.link/20260923183600.1186943-6-alan.previn.teres.alexis@intel.com Signed-off-by: Rodrigo Vivi --- drivers/gpu/drm/xe/xe_pci.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/gpu/drm/xe/xe_pci.c b/drivers/gpu/drm/xe/xe_pci.c index 3719427dd17b89..e44b74b9aa27af 100644 --- a/drivers/gpu/drm/xe/xe_pci.c +++ b/drivers/gpu/drm/xe/xe_pci.c @@ -461,6 +461,7 @@ static const struct xe_device_desc nvls_desc = { .has_flat_ccs = 1, .has_pre_prod_wa = 1, .has_sriov = true, + .has_pxp = true, .max_gt_per_tile = 2, MULTI_LRC_MASK, .va_bits = 48, From 3fc93d311249d5ad969a520dd3aae73bc007d431 Mon Sep 17 00:00:00 2001 From: Daniel Charles Date: Tue, 29 Sep 2026 13:08:39 -0700 Subject: [PATCH 0196/1012] drm/xe/xe3p: Force non-compressible memory reads to 256B overfetches Add a tuning entry to set bit 14 (NONCOMPMEMRD256BOVRFETCHEN) of L3SQCREG2 for Xe3p and later platforms. BSpec: 72161, 59927 Signed-off-by: Daniel Charles Reviewed-by: Matt Roper Link: https://patch.msgid.link/20260929200839.843116-1-daniel.charles@intel.com Signed-off-by: Matt Roper --- drivers/gpu/drm/xe/regs/xe_gt_regs.h | 1 + drivers/gpu/drm/xe/xe_tuning.c | 4 ++++ 2 files changed, 5 insertions(+) diff --git a/drivers/gpu/drm/xe/regs/xe_gt_regs.h b/drivers/gpu/drm/xe/regs/xe_gt_regs.h index 926b14e32c40fb..650f117c3e135b 100644 --- a/drivers/gpu/drm/xe/regs/xe_gt_regs.h +++ b/drivers/gpu/drm/xe/regs/xe_gt_regs.h @@ -468,6 +468,7 @@ #define L3_SQ_DISABLE_COAMA_2WAY_COH REG_BIT(30) #define L3_SQ_DISABLE_COAMA REG_BIT(22) #define COMPMEMRD256BOVRFETCHEN REG_BIT(20) +#define NONCOMPMEMRD256BOVRFETCHEN REG_BIT(14) #define L3SQCREG3 XE_REG_MCR(0xb108) #define COMPPWOVERFETCHEN REG_BIT(28) diff --git a/drivers/gpu/drm/xe/xe_tuning.c b/drivers/gpu/drm/xe/xe_tuning.c index bcec40ca2d35f3..a9bd3f8969a545 100644 --- a/drivers/gpu/drm/xe/xe_tuning.c +++ b/drivers/gpu/drm/xe/xe_tuning.c @@ -100,6 +100,10 @@ VISIBLE_IF_KUNIT const struct xe_rtp_table_sr gt_tunings = XE_RTP_TABLE_SR( XE_RTP_ACTIONS(FIELD_SET(GAMSTLB_CTRL, BANK_HASH_MODE, BANK_HASH_4KB_MODE)) }, + { XE_RTP_NAME("Tuning: Force All Non-Compressible Memory Reads to be 256B Overfetches"), + XE_RTP_RULES(GRAPHICS_VERSION_RANGE(3510, XE_RTP_END_VERSION_UNDEFINED)), + XE_RTP_ACTIONS(SET(L3SQCREG2, NONCOMPMEMRD256BOVRFETCHEN)) + }, ); EXPORT_SYMBOL_IF_KUNIT(gt_tunings); From 6d996c3b974c501f347e8bad1644afd1180b2795 Mon Sep 17 00:00:00 2001 From: Eric Biggers Date: Tue, 29 Sep 2026 15:27:52 -0700 Subject: [PATCH 0197/1012] crypto: aes - Fix undesired override of some optimized AES modes The new library APIs for AES encryption modes were wired up to the traditional crypto API via crypto/aes.c. However, for now the kernel is still in a transitional state where various architectures still have architecture-optimized implementations of AES modes in arch/*/crypto/, wired up to the traditional crypto API only. Because of that, the crypto/aes.c algorithms were given a cra_priority of only 110 to prevent them from overriding arch/*/crypto/ in the traditional crypto API. However, because of how the traditional crypto API works, the cra_priority trick doesn't work in cases where the relevant algorithm isn't directly implemented by arch/*/crypto/ but rather is provided by a template instance using other code in arch/*/crypto/. For example, x86 doesn't have its own "ccm(aes)" but rather relies on the "ccm" template constructing it from the x86-optimized "ctr(aes)". The existence of the library-based "ccm(aes)" prevents that, even though its priority is lower than what the template would produce. Thus, "ccm(aes)" ends up using the slower single-block AES code. Therefore, skip wiring up the relevant library-based code to the traditional crypto API on architectures where this problem can occur, as determined by what exists in arch/*/crypto/ for each architecture. This is ugly, but it's also temporary: these conditions will go away as architecture-optimized implementations of AES modes are migrated into the library. But until then, we need to prevent performance regressions by ensuring that the optimized code continues to be used. Fixes: 20df21a482aa ("crypto: aes - Add CBC and CBC-CTS support using library") Fixes: 8ca62072faa1 ("crypto: aes - Add GCM support using library") Fixes: f70ad727d1d6 ("crypto: aes - Add CCM support using library") Fixes: 94efa0c9fb36 ("crypto: aes - Add XTS support using library") Closes: https://github.com/sparclinux/issues/issues/106 Link: https://patch.msgid.link/20260929222752.36427-1-ebiggers@kernel.org Signed-off-by: Eric Biggers --- crypto/aes.c | 51 +++++++++++++++++++++++++++++++++++++++++++++++---- 1 file changed, 47 insertions(+), 4 deletions(-) diff --git a/crypto/aes.c b/crypto/aes.c index 94791f481e9842..51eee78396ead5 100644 --- a/crypto/aes.c +++ b/crypto/aes.c @@ -637,7 +637,18 @@ static struct skcipher_alg skcipher_algs[] = { .decrypt = crypto_aes_cbc_decrypt, }, #endif -#if IS_ENABLED(CONFIG_CRYPTO_CTS) + /* + * Don't register library-based "cts(cbc(aes))" on architectures where + * it might block a "better" implementation from being instantiated via + * the "cts" template. These exclusions are temporary and will go away + * as the arch-optimized AES code is migrated into the library. + */ +#if IS_ENABLED(CONFIG_CRYPTO_CTS) && \ + !(IS_ENABLED(CONFIG_ARM) || \ + IS_ENABLED(CONFIG_ARM64) || \ + IS_ENABLED(CONFIG_PPC) || \ + IS_ENABLED(CONFIG_S390) || \ + IS_ENABLED(CONFIG_SPARC)) { .base.cra_name = "cts(cbc(aes))", .base.cra_driver_name = "cts-cbc-aes-lib", @@ -687,7 +698,13 @@ static struct skcipher_alg skcipher_algs[] = { .decrypt = crypto_aes_xctr_crypt, }, #endif -#if IS_ENABLED(CONFIG_CRYPTO_XTS) + /* + * Don't register library-based "xts(aes)" on architectures where it + * might block a "better" implementation from being instantiated via the + * "xts" template. This exclusion is temporary and will go away when + * the library AES-XTS is optimized for SPARC. + */ +#if IS_ENABLED(CONFIG_CRYPTO_XTS) && !IS_ENABLED(CONFIG_SPARC) { .base.cra_name = "xts(aes)", .base.cra_driver_name = "xts-aes-lib", @@ -980,7 +997,20 @@ static __maybe_unused int crypto_aes_ccm_decrypt(struct aead_request *req) } static struct aead_alg aead_algs[] = { -#if IS_ENABLED(CONFIG_CRYPTO_GCM) + /* + * Don't register library-based "gcm(aes)" and "rfc4106(gcm(aes))" on + * architectures where they might block a "better" implementation from + * being instantiated via the "gcm" and "rfc4106" templates. These + * exclusions are temporary and will go away as the arch-optimized AES + * code is migrated into the library. + */ +#if IS_ENABLED(CONFIG_CRYPTO_GCM) && \ + !(IS_ENABLED(CONFIG_ARM) || \ + IS_ENABLED(CONFIG_ARM64) || \ + IS_ENABLED(CONFIG_PPC) || \ + IS_ENABLED(CONFIG_RISCV) || \ + IS_ENABLED(CONFIG_S390) || \ + IS_ENABLED(CONFIG_SPARC)) { .base.cra_name = "gcm(aes)", .base.cra_driver_name = "gcm-aes-lib", @@ -1012,7 +1042,20 @@ static struct aead_alg aead_algs[] = { .chunksize = AES_BLOCK_SIZE, }, #endif /* CONFIG_CRYPTO_GCM */ -#if IS_ENABLED(CONFIG_CRYPTO_CCM) + /* + * Don't register library-based "ccm(aes)" on architectures where it + * might block a "better" implementation from being instantiated via the + * "ccm" template. These exclusions are temporary and will go away as + * the arch-optimized AES code is migrated into the library. + */ +#if IS_ENABLED(CONFIG_CRYPTO_CCM) && \ + !(IS_ENABLED(CONFIG_ARM) || \ + IS_ENABLED(CONFIG_ARM64) || \ + IS_ENABLED(CONFIG_PPC) || \ + IS_ENABLED(CONFIG_RISCV) || \ + IS_ENABLED(CONFIG_S390) || \ + IS_ENABLED(CONFIG_SPARC) || \ + IS_ENABLED(CONFIG_X86)) { .base.cra_name = "ccm(aes)", .base.cra_driver_name = "ccm-aes-lib", From 1846f0bb064ee264234260040c90ac4410c8c3c2 Mon Sep 17 00:00:00 2001 From: "Uladzislau Rezki (Sony)" Date: Sat, 5 Sep 2026 17:27:17 +0200 Subject: [PATCH 0198/1012] mm/vmalloc: use dedicated unbound workqueues for vmap drain drain_vmap_area_work() function can take >10ms to complete when there are many accumulated vmap areas in a system with high CPU count, causing workqueue watchdog warnings when run via schedule_work(): workqueue: drain_vmap_area_work hogged CPU for >10000us Move the top-level drain work to a dedicated WQ_UNBOUND workqueue so the scheduler can run this background work on any available CPU, improving responsiveness. Use the WQ_MEM_RECLAIM to ensure forward progress under memory pressure. Move purge helpers to separate WQ_UNBOUND | WQ_MEM_RECLAIM workqueue. This allows drain_vmap_work to wait for helpers completion without creating dependency on the same rescuer thread and avoid a potential parent/child deadlock. Simplify purge helper scheduling by removing cpumask-based iteration to iterating directly over vmap nodes checking work_queued state. Link: https://lore.kernel.org/20260905152717.11711-1-urezki@gmail.com Fixes: 72210662c5a2 ("mm: vmalloc: offload free_vmap_area_lock lock") Signed-off-by: Uladzislau Rezki (Sony) Signed-off-by: Andrew Morton Reported-by: Li RongQing Closes: https://lore.kernel.org/all/20260319074307.2325-1-lirongqing@baidu.com/ Reviewed-by: Baoquan He Reviewed-by: Ye Liu Cc: Dev Jain Cc: --- mm/vmalloc.c | 79 ++++++++++++++++++++++++++++++++++------------------ 1 file changed, 52 insertions(+), 27 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index bea9f76ed7e742..89c327a6ce7d9f 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -972,6 +972,7 @@ static struct vmap_node { struct list_head purge_list; struct work_struct purge_work; unsigned long nr_purged; + bool work_queued; } single; /* @@ -1090,6 +1091,8 @@ static void reclaim_and_purge_vmap_areas(void); static BLOCKING_NOTIFIER_HEAD(vmap_notify_list); static void drain_vmap_area_work(struct work_struct *work); static DECLARE_WORK(drain_vmap_work, drain_vmap_area_work); +static struct workqueue_struct *drain_vmap_helpers_wq; +static struct workqueue_struct *drain_vmap_wq; static __cacheline_aligned_in_smp atomic_long_t vmap_lazy_nr; @@ -2351,6 +2354,16 @@ static void purge_vmap_node(struct work_struct *work) reclaim_list_global(&local_list); } +static bool +schedule_drain_vmap_work(struct workqueue_struct *wq, + struct work_struct *work) +{ + if (wq) + return queue_work(wq, work); + + return false; +} + /* * Purges all lazily-freed vmap areas. */ @@ -2358,19 +2371,12 @@ static bool __purge_vmap_area_lazy(unsigned long start, unsigned long end, bool full_pool_decay) { unsigned long nr_purged_areas = 0; + unsigned int nr_purge_nodes = 0; unsigned int nr_purge_helpers; - static cpumask_t purge_nodes; - unsigned int nr_purge_nodes; struct vmap_node *vn; - int i; lockdep_assert_held(&vmap_purge_lock); - /* - * Use cpumask to mark which node has to be processed. - */ - purge_nodes = CPU_MASK_NONE; - for_each_vmap_node(vn) { INIT_LIST_HEAD(&vn->purge_list); vn->skip_populate = full_pool_decay; @@ -2390,10 +2396,9 @@ static bool __purge_vmap_area_lazy(unsigned long start, unsigned long end, end = max(end, list_last_entry(&vn->purge_list, struct vmap_area, list)->va_end); - cpumask_set_cpu(node_to_id(vn), &purge_nodes); + nr_purge_nodes++; } - nr_purge_nodes = cpumask_weight(&purge_nodes); if (nr_purge_nodes > 0) { flush_tlb_kernel_range(start, end); @@ -2401,29 +2406,31 @@ static bool __purge_vmap_area_lazy(unsigned long start, unsigned long end, nr_purge_helpers = atomic_long_read(&vmap_lazy_nr) / lazy_max_pages(); nr_purge_helpers = clamp(nr_purge_helpers, 1U, nr_purge_nodes) - 1; - for_each_cpu(i, &purge_nodes) { - vn = &vmap_nodes[i]; + for_each_vmap_node(vn) { + vn->work_queued = false; + + if (list_empty(&vn->purge_list)) + continue; if (nr_purge_helpers > 0) { INIT_WORK(&vn->purge_work, purge_vmap_node); + vn->work_queued = schedule_drain_vmap_work( + READ_ONCE(drain_vmap_helpers_wq), &vn->purge_work); - if (cpumask_test_cpu(i, cpu_online_mask)) - schedule_work_on(i, &vn->purge_work); - else - schedule_work(&vn->purge_work); - - nr_purge_helpers--; - } else { - vn->purge_work.func = NULL; - purge_vmap_node(&vn->purge_work); - nr_purged_areas += vn->nr_purged; + if (vn->work_queued) { + nr_purge_helpers--; + continue; + } } - } - for_each_cpu(i, &purge_nodes) { - vn = &vmap_nodes[i]; + /* Sync path. Process locally. */ + purge_vmap_node(&vn->purge_work); + nr_purged_areas += vn->nr_purged; + } - if (vn->purge_work.func) { + /* Wait for completion if queued any. */ + for_each_vmap_node(vn) { + if (vn->work_queued) { flush_work(&vn->purge_work); nr_purged_areas += vn->nr_purged; } @@ -2487,7 +2494,8 @@ static void free_vmap_area_noflush(struct vmap_area *va) /* After this point, we may free va at any time */ if (unlikely(nr_lazy > nr_lazy_max)) - schedule_work(&drain_vmap_work); + schedule_drain_vmap_work(READ_ONCE(drain_vmap_wq), + &drain_vmap_work); } /* @@ -5587,3 +5595,20 @@ void __init vmalloc_init(void) vmap_node_shrinker->scan_objects = vmap_node_shrink_scan; shrinker_register(vmap_node_shrinker); } + +static int __init vmalloc_init_workqueue(void) +{ + struct workqueue_struct *drain_wq, *helpers_wq; + unsigned int flags = WQ_UNBOUND | WQ_MEM_RECLAIM; + + drain_wq = alloc_workqueue("vmap_drain", flags, 0); + WARN_ON_ONCE(drain_wq == NULL); + WRITE_ONCE(drain_vmap_wq, drain_wq); + + helpers_wq = alloc_workqueue("vmap_drain_helpers", flags, 0); + WARN_ON_ONCE(helpers_wq == NULL); + WRITE_ONCE(drain_vmap_helpers_wq, helpers_wq); + + return 0; +} +early_initcall(vmalloc_init_workqueue); From c228de5a6b5c396042d56c3b92a621039135bfef Mon Sep 17 00:00:00 2001 From: Krystian Kaniewski Date: Fri, 4 Sep 2026 14:12:59 +0200 Subject: [PATCH 0199/1012] xarray: fix index jumping backwards in xas_find() A bug in the XArray iterator xas_find() causes the iterator's index (xas->xa_index) to jump backwards when iterating over a multi-index entry (like a THP) that resides in a non-leaf node and is concurrently split. When iterating over a multi-index entry in a non-leaf node, xas_load() sets xas->xa_offset to the base offset of the entry, but leaves xas->xa_index at the requested index. When the caller subsequently wants to advance to the next entry, xas_find() is called. xas_find() attempts to synchronize xas->xa_offset with xas->xa_index before advancing. However, the fixup logic was incorrectly restricted to leaf nodes (!xas->xa_node->shift). Because the THP resides in a non-leaf node, the fixup is skipped. As a result, xas_find() simply increments xas->xa_offset and recalculates xas->xa_index based on this new offset. This causes xas->xa_index to jump backwards. If the THP was concurrently split, the entry at the new offset is a node pointer, so xas_find() descends into it and returns the folio at the backwards index. The caller (filemap_map_pages()) then calculates the PTE pointer based on this backwards index, resulting in an invalid memory access such as an out-of-bounds read or use-after-free on a page-table page freed via tlb_remove_table_rcu(). A userspace access that faults in a file-backed mapping can trigger this path. When the index moves backwards, filemap_map_pages() can calculate a PTE outside the page locked for fault-around and dereference a freed page-table page, resulting in a KASAN-detected use-after-free read. To fix this, check if xas->xa_offset matches get_offset(xas->xa_index, xas->xa_node). If it does not and the node is a non-leaf node, set xas->xa_offset to get_offset(xas->xa_index, xas->xa_node) before advancing. Also add test cases in test_xarray to verify xas_find() behavior when iterating over and splitting multi-index entries. Link: https://lore.kernel.org/20260904121301.200049-1-krystianmkaniewski@gmail.com Fixes: b803b42823d0 ("xarray: Add XArray iterators") Reported-by: syzbot+b72767277f29b6407083@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=b72767277f29b6407083 Closes: https://syzkaller.appspot.com/ai_job?id=a01c56bd-74d0-411c-afb4-ee6f0cb6cb61 Signed-off-by: Krystian Kaniewski Signed-off-by: Andrew Morton Assisted-by: Gemini:gemini-3.7-flash syzbot Cc: Matthew Wilcox (Oracle) Cc: Mike Rapoport Cc: --- lib/test_xarray.c | 70 +++++++++++++++++++++++++++++++++++++++++++++++ lib/xarray.c | 8 ++++-- 2 files changed, 75 insertions(+), 3 deletions(-) diff --git a/lib/test_xarray.c b/lib/test_xarray.c index 5ca0aefee9aa5f..615f7730a5bd4a 100644 --- a/lib/test_xarray.c +++ b/lib/test_xarray.c @@ -1247,6 +1247,75 @@ static noinline void check_multi_find_3(struct xarray *xa) } } +static noinline void check_multi_find_4(struct xarray *xa) +{ +#ifdef CONFIG_XARRAY_MULTI + unsigned int order = XA_CHUNK_SHIFT + 1; + unsigned long next = 1UL << order; + unsigned long start = next - 1; + XA_STATE(xas, xa, start); + XA_STATE_ORDER(split, xa, 0, 0); + void *entry; + unsigned long i; + + /* Multi-index entry in two slots of a non-leaf node. */ + xa_store_order(xa, 0, order, xa_mk_index(0), GFP_KERNEL); + XA_BUG_ON(xa, xa_store_index(xa, next, GFP_KERNEL) != NULL); + + rcu_read_lock(); + entry = xas_find(&xas, ULONG_MAX); + XA_BUG_ON(xa, entry != xa_mk_index(0)); + XA_BUG_ON(xa, xas.xa_index != start); + + entry = xas_find(&xas, ULONG_MAX); + XA_BUG_ON(xa, entry != xa_mk_index(next)); + XA_BUG_ON(xa, xas.xa_index != next); + + entry = xas_find(&xas, ULONG_MAX); + XA_BUG_ON(xa, entry != NULL); + rcu_read_unlock(); + + xa_erase_index(xa, next); + xa_erase_index(xa, 0); + XA_BUG_ON(xa, !xa_empty(xa)); + + /* Split the multi-index entry after a lookup begins inside it. */ + xa_store_order(xa, 0, order, xa_mk_index(0), GFP_KERNEL); + XA_BUG_ON(xa, xa_store_index(xa, next, GFP_KERNEL) != NULL); + + xas_set(&xas, start); + rcu_read_lock(); + entry = xas_find(&xas, ULONG_MAX); + XA_BUG_ON(xa, entry != xa_mk_index(0)); + XA_BUG_ON(xa, xas.xa_index != start); + rcu_read_unlock(); + + xas_split_alloc(&split, xa_mk_index(0), order, GFP_KERNEL); + if (xas_error(&split)) { + XA_BUG_ON(xa, true); + goto out; + } + xas_lock(&split); + xas_split(&split, xa_mk_index(0), order); + for (i = 0; i < next; i++) + __xa_store(xa, i, xa_mk_index(i), 0); + xas_unlock(&split); + + rcu_read_lock(); + entry = xas_find(&xas, ULONG_MAX); + XA_BUG_ON(xa, entry != xa_mk_index(next)); + XA_BUG_ON(xa, xas.xa_index != next); + + entry = xas_find(&xas, ULONG_MAX); + XA_BUG_ON(xa, entry != NULL); + rcu_read_unlock(); + +out: + xa_destroy(xa); + XA_BUG_ON(xa, !xa_empty(xa)); +#endif +} + static noinline void check_find_1(struct xarray *xa) { unsigned long i, j, k; @@ -1370,6 +1439,7 @@ static noinline void check_find(struct xarray *xa) check_multi_find_1(xa, i); check_multi_find_2(xa); check_multi_find_3(xa); + check_multi_find_4(xa); } /* See find_swap_entry() in mm/shmem.c */ diff --git a/lib/xarray.c b/lib/xarray.c index bfe7bef80f34eb..3913c8d6486bb1 100644 --- a/lib/xarray.c +++ b/lib/xarray.c @@ -1409,9 +1409,11 @@ void *xas_find(struct xa_state *xas, unsigned long max) entry = xas_load(xas); if (entry || xas_not_node(xas->xa_node)) return entry; - } else if (!xas->xa_node->shift && - xas->xa_offset != (xas->xa_index & XA_CHUNK_MASK)) { - xas->xa_offset = ((xas->xa_index - 1) & XA_CHUNK_MASK) + 1; + } else if (xas->xa_offset != get_offset(xas->xa_index, xas->xa_node)) { + if (!xas->xa_node->shift) + xas->xa_offset = ((xas->xa_index - 1) & XA_CHUNK_MASK) + 1; + else + xas->xa_offset = get_offset(xas->xa_index, xas->xa_node); } xas_next_offset(xas); From 300d1eeeb7aa321d346054b4ba9867eb914c657a Mon Sep 17 00:00:00 2001 From: Zhao Li Date: Tue, 28 Apr 2026 19:30:38 +0800 Subject: [PATCH 0200/1012] mm/hugetlb: fix max-only subpool accounting on alloc_hugetlb_folio failure Failed hugetlbfs page allocations can permanently consume the mount's size= quota without allocating a huge page. Repeated failures can make the filesystem appear full and cause later huge-page faults or allocations to fail with SIGBUS/allocation failure despite available huge pages and unused real filesystem capacity. alloc_hugetlb_folio() calls hugepage_subpool_get_pages() when map_chg is set. For a subpool with max_hpages != -1, that bumps used_hpages regardless of whether it returns gbl_chg = 0 (rsv slot consumed) or gbl_chg > 0 (used_hpages slot only). If the allocation later fails before a folio is returned, the unwind must undo the used_hpages bump. The old cleanup only ran for !gbl_chg, leaking used_hpages on the gbl_chg > 0 path. For gbl_chg > 0 on max-only subpools (max_hpages != -1, min_hpages == -1), hugepage_subpool_get_pages() took only a speculative used_hpages slot. Drop that slot directly under spool->lock. In that configuration hugepage_subpool_put_pages() cannot restore rsv_hpages, so the direct decrement is the exact inverse and is race-free against concurrent puts. This matches the used_hpages-only part of hugetlb_reserve_pages()'s out_put_pages cleanup, but restricts it to the max-only case where no rsv_hpages restoration is possible. Mounts with min_hpages != -1 are left unchanged for now. v2's approach (hugepage_subpool_put_pages() + h->resv_huge_pages++ to back a restored rsv_hpages slot) double-counts global backing under concurrent free_huge_folio() and creates phantom reservations under concurrent hugetlb_unreserve_pages(). Safe cleanup of that quadrant needs a coordinated fix across multiple call sites. Reproduced on size=20M hugetlbfs with the faulting task in a hugetlb cgroup whose limit is exceeded. Vanilla leaks 6/8 hugepages of subpool quota; this patch leaks 0/8. Verified under QEMU. Link: https://lore.kernel.org/20260428113037.88766-2-enderaoelyther@gmail.com Fixes: a833a693a490 ("mm: hugetlb: fix incorrect fallback for subpool") Signed-off-by: Zhao Li Signed-off-by: Andrew Morton Tested-by: Ackerley Tng Cc: David Hildenbrand Cc: Ma Wupeng Cc: Muchun Song Cc: Oscar Salvador Cc: # v6.15+ --- mm/hugetlb.c | 25 ++++++++++++++++++------- 1 file changed, 18 insertions(+), 7 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index cea25773a6c953..49ffbcb54f8c08 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3070,13 +3070,24 @@ struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma, return folio; out_subpool_put: - /* - * put page to subpool iff the quota of subpool's rsv_hpages is used - * during hugepage_subpool_get_pages. - */ - if (map_chg && !gbl_chg) { - gbl_reserve = hugepage_subpool_put_pages(spool, 1); - hugetlb_acct_memory(h, -gbl_reserve); + if (map_chg) { + if (!gbl_chg) { + /* Full inverse when subpool_get_pages() consumed rsv_hpages. */ + gbl_reserve = hugepage_subpool_put_pages(spool, 1); + hugetlb_acct_memory(h, -gbl_reserve); + } else if (gbl_chg > 0 && spool && spool->min_hpages == -1 && + spool->max_hpages != -1) { + unsigned long flags; + + /* + * For max-only subpools, subpool_get_pages() took only a + * speculative used_hpages slot. Drop that slot directly. + */ + spin_lock_irqsave(&spool->lock, flags); + if (spool->used_hpages > 0) + spool->used_hpages--; + unlock_or_release_subpool(spool, flags); + } } out_end_reservation: From f6a2c780c578ab180b4568ac548ef14c7651e96f Mon Sep 17 00:00:00 2001 From: Salvatore Dipietro Date: Fri, 4 Sep 2026 11:56:28 +0000 Subject: [PATCH 0201/1012] mm/page_alloc: avoid direct compaction for costly __GFP_NORETRY allocations Commit 5d8edfb900d5 ("iomap: Copy larger chunks from userspace") introduced high-order folio allocations in the iomap buffered write path. When memory is fragmented, each failed costly-order allocation enters __alloc_pages_slowpath() which runs direct compaction and drain_all_pages(), causing a 0.38x throughput drop on PostgreSQL pgbench (simple-update) with 1024 clients on a 96-vCPU arm64 system. The root issue is that direct compaction is too expensive for hot allocation paths that have fallbacks to smaller allocations. __filemap_get_folio_mpol() already marks higher-order allocations with __GFP_NORETRY | __GFP_NOWARN, signalling that the caller can handle failure. However, the page allocator still attempts full direct compaction for costly orders with __GFP_NORETRY, which is unnecessarily aggressive when the caller will simply retry at a lower order. For costly-order allocations with __GFP_NORETRY, clear __GFP_DIRECT_RECLAIM at the very start of the slowpath, before can_direct_reclaim, can_compact and the nofail checks are evaluated. This makes the entire slowpath treat the request as non-blocking: no direct reclaim, no direct compaction and no drain_all_pages() IPI across every CPU. kswapd (and in turn kcompactd) is still woken further down for background defragmentation, so compaction keeps working for long-term system health while being removed from the latency-critical direct allocation path. Allocations that also request __GFP_THISNODE are exempted. That flag pairing identifies the local-node-first THP attempt issued by alloc_pages_mpol() (mempolicy.c), which relies on direct compaction to form transparent huge pages. Test environment: Hardware: AWS EC2 m8g.24xlarge (96 vCPU, arm64) 12x 1TB IO2 32000 IOPS RAID0 XFS OS: AL2023 Kernel: v7.3-rc1 Database: PostgreSQL 18.4 Workload: pgbench simple-update, 1024 clients, 96 threads, 1200s Results (average of 3 runs, TPS): Config Avg TPS % vs Baseline baseline (no patch) 59,408 - With this patch 155,409 +161.6% Link: https://lore.kernel.org/20260904115629.3993331-1-dipiets@amazon.it Link: https://lore.kernel.org/all/20260403193535.9970-1-dipiets@amazon.it/T/#t [v1] Link: https://lore.kernel.org/linux-mm/20260420161404.642-1-dipiets@amazon.it/T/#u [v2] Link: https://lore.kernel.org/all/20260710143437.12379-1-dipiets@amazon.it/T/#u [v3] Fixes: 5d8edfb900d5 ("iomap: Copy larger chunks from userspace") Signed-off-by: Salvatore Dipietro Signed-off-by: Andrew Morton Acked-by: Vlastimil Babka (SUSE) Acked-by: Zi Yan Reviewed-by: Johannes Weiner Reviewed-by: Christoph Hellwig Cc: David Hildenbrand Cc: Michal Hocko Cc: Matthew Wilcox Cc: Dave Chinner Cc: Ritesh Harjani Cc: --- mm/page_alloc.c | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 12fac9084c483d..542c2ec3106134 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -4784,10 +4784,10 @@ static inline struct page * __alloc_pages_slowpath(gfp_t gfp_mask, unsigned int order, struct alloc_context *ac) { - bool can_direct_reclaim = gfp_mask & __GFP_DIRECT_RECLAIM; - bool can_compact = can_direct_reclaim && gfp_compaction_allowed(gfp_mask); - bool nofail = gfp_mask & __GFP_NOFAIL; const bool costly_order = order > PAGE_ALLOC_COSTLY_ORDER; + bool can_direct_reclaim; + bool can_compact; + bool nofail; struct page *page = NULL; unsigned int alloc_flags; unsigned long did_some_progress; @@ -4802,6 +4802,18 @@ __alloc_pages_slowpath(gfp_t gfp_mask, unsigned int order, bool can_retry_reserves = true; unsigned long alloc_start_time = jiffies; + /* + * Costly __GFP_NORETRY callers have a cheap fallback, so don't stall + * them in reclaim or compaction. __GFP_THISNODE callers are exempt. + */ + if (costly_order && (gfp_mask & __GFP_NORETRY) && + !(gfp_mask & __GFP_THISNODE)) + gfp_mask &= ~__GFP_DIRECT_RECLAIM; + + can_direct_reclaim = gfp_mask & __GFP_DIRECT_RECLAIM; + can_compact = can_direct_reclaim && gfp_compaction_allowed(gfp_mask); + nofail = gfp_mask & __GFP_NOFAIL; + if (unlikely(nofail)) { /* * Also we don't support __GFP_NOFAIL without __GFP_DIRECT_RECLAIM, From 98a6b57d33e65e99903f1531fd7eb4d9a5bf8f83 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sun, 20 Sep 2026 15:13:10 +0100 Subject: [PATCH 0202/1012] mm/mremap: fix locked_vm leak from MREMAP_DONTUNMAP self-merge Patch series "mm/mremap: fix two issues with MREMAP_DONTUNMAP". The MREMAP_DONTUNMAP feature is highly unusual in that it permits mremap() operations that keep the original VMA in place. Historically this has led to a lot of bugs where non-obvious interactions occur between existing mremap() operations and the original VMA. Commit 397432cab17b ("mm/mremap: account mm->locked_vm correctly for MREMAP_DONTUNMAP") fixed an accidentally introduced bug around mm->locked_vm accounting, but this wasn't the only issue. And thus history repeats itself, as it turns out that mm->locked_vm accounting is broken by MREMAP_DONTUNMAP yet again by two further cases, and has been broken ever since the feature was introduced. Both relate to the fact that VMA_LOCKED_BIT is cleared on the source VMA (it has to be as all page tables are moved): 1. If an unfaulted VMA_LOCKONFAULT_BIT anonymous VMA self-merges it clears the VMA_LOCKED_BIT flag and permanently leaks mm->locked_vm pages. 2. If a partial mremap() is performed on a locked VMA there is a leak equal to the number of pages not copied. (Both for MREMAP_DONTUNMAP operations only) Both issues can be fixed by treating the source range as distinct from the destination range, which is the definition of what MREMAP_DONTUNMAP does so is appropriate. In case 1, simply disallow the self-merge, keeping adjacent source and destination VMAs distinct. In case 2, split the source range ahead of time, so accounting is always correct. Both changes were tested locally and confirmed to fix the issues. For the purposes of a backport, the fixes are kept distinct, a follow-up series can add self-tests. This patch (of 2): The MREMAP_DONTUNMAP feature is highly unusual in that it permits mremap() operations that keep the original VMA in place. Historically this has led to a lot of bugs where non-obvious interactions occur between existing mremap() operations and the original VMA. Fix another of these - self-merge. Self-merge occurs when a VMA is moved in front of or behind itself and the attributes of the VMA permit such a merge. Practically this can only happen for unfaulted anonymous VMAs due to the page offset equality requirement for merge: |------------| | | | v |...........||-----------||...........| | || unfaulted || | |...........||-----------||...........| ^ | | | |------------| This becomes problematic if the VMA is configured by the user to mlock-on-fault, i.e. the VMA_LOCKED_BIT, VMA_LOCKONFAULT_BIT VMA flags are set. MREMAP_DONTUNMAP clears mlock flags for the source VMA and maintains them for the destination VMA. Self-merge makes this impossible (there is only one VMA) and incorrectly clears the destination VMA's mlock flags. This causes a leak in mm->locked_vm as clearing this flag does not decrement the counter and the VMA no longer has VMA_LOCKED_BIT set so it is not decremented on unmap. Resolve this by simply disallowing a self-merge in this case - the source and destination VMAs are kept distinct and then are able to have distinct mlock() flags. Update dontunmap_complete() to make the now-redundant self-merge check a VM_WARN_ON_ONCE() instead to guard against future regressions. Also update the VMA userland tests to reflect the change. Link: https://lore.kernel.org/20260920-fix-dontunmap-partial-self-merge-v1-0-6ffb556f8f8b@kernel.org Link: https://lore.kernel.org/20260920-fix-dontunmap-partial-self-merge-v1-1-6ffb556f8f8b@kernel.org Fixes: e346b3813067 ("mm/mremap: add MREMAP_DONTUNMAP to mremap()") Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Jose A. Perez de Azpillaga Reviewed-by: Pedro Falcato Acked-by: Kiryl Shutsemau (Meta) Cc: Brian Geffon Cc: Jann Horn Cc: Liam Howlett Cc: Minchan Kim Cc: "Vlastimil Babka (SUSE)" Cc: --- mm/mremap.c | 8 ++++++-- mm/vma.c | 17 ++++++++++++++++- mm/vma.h | 2 +- tools/testing/vma/tests/vma.c | 10 +++++----- 4 files changed, 28 insertions(+), 9 deletions(-) diff --git a/mm/mremap.c b/mm/mremap.c index 7c368440fafe24..058dd66fe1d762 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -1275,7 +1275,8 @@ static int copy_vma_and_data(struct vma_remap_struct *vrm, PAGETABLE_MOVE(pmc, NULL, NULL, vrm->addr, vrm->new_addr, vrm->old_len); new_vma = copy_vma(&vma, vrm->new_addr, vrm->new_len, new_pgoff, - new_anon_pgoff, &pmc.need_rmap_locks); + new_anon_pgoff, &pmc.need_rmap_locks, + vrm->flags & MREMAP_DONTUNMAP); if (!new_vma) { vrm_uncharge(vrm); *new_vma_ptr = NULL; @@ -1335,6 +1336,9 @@ static void dontunmap_complete(struct vma_remap_struct *vrm, unsigned long old_start = vma->vm_start; unsigned long old_end = vma->vm_end; + /* Self-merge is disallowed. */ + VM_WARN_ON_ONCE(new_vma == vma); + /* We always clear VMA_LOCKED[ONFAULT]_BIT on the old VMA. */ vma_clear_flags_mask(vma, VMA_LOCKED_MASK); @@ -1342,7 +1346,7 @@ static void dontunmap_complete(struct vma_remap_struct *vrm, * anon_vma links of the old vma is no longer needed after its page * table has been moved. */ - if (new_vma != vma && start == old_start && end == old_end) { + if (start == old_start && end == old_end) { const pgoff_t pgoff_unfaulted = vma->vm_start >> PAGE_SHIFT; unlink_anon_vmas(vma); diff --git a/mm/vma.c b/mm/vma.c index 6cde67883fb04a..d8308c319daa7e 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -1943,7 +1943,7 @@ static int vma_link(struct mm_struct *mm, struct vm_area_struct *vma) */ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, unsigned long addr, unsigned long len, pgoff_t pgoff, - pgoff_t anon_pgoff, bool *need_rmap_locks) + pgoff_t anon_pgoff, bool *need_rmap_locks, bool keep_source) { struct vm_area_struct *vma = *vmap; unsigned long old_vma_start = vma->vm_start; @@ -1981,6 +1981,21 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, vmg.pgoff = pgoff; vmg.anon_pgoff = anon_pgoff; vmg.next = vma_iter_next_rewind(&vmi, NULL); + + /* + * If the original VMA is kept (MREMAP_DONTUNMAP), the source and + * destination VMA must be treated distinctly. + * + * A merge violates this, so in this case disallow a self-merge. + */ + if (can_self_merge && keep_source) { + if (vmg.prev == vma) + vmg.prev = NULL; + if (vmg.next == vma) + vmg.next = NULL; + can_self_merge = false; + } + new_vma = vma_merge_copied_range(&vmg); if (new_vma) { diff --git a/mm/vma.h b/mm/vma.h index 024fabe63560a0..39514fac72fc70 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -535,7 +535,7 @@ void unlink_file_vma_batch_add(struct unlink_vma_file_batch *vb, struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, unsigned long addr, unsigned long len, pgoff_t pgoff, - pgoff_t anon_pgoff, bool *need_rmap_locks); + pgoff_t anon_pgoff, bool *need_rmap_locks, bool keep_source); struct anon_vma *find_mergeable_anon_vma(struct vm_area_struct *vma); diff --git a/tools/testing/vma/tests/vma.c b/tools/testing/vma/tests/vma.c index c8ef7b8cd46be7..e973e0a6d1a829 100644 --- a/tools/testing/vma/tests/vma.c +++ b/tools/testing/vma/tests/vma.c @@ -40,7 +40,7 @@ static bool test_copy_vma(void) vma = alloc_and_link_vma(&mm, 0x1000, 0x2000, 1, vma_flags); vma_set_anonymous(vma); vma_orig = vma; - vma_new = copy_vma(&vma, 0x2000, 0x1000, 1, 1, &need_locks); + vma_new = copy_vma(&vma, 0x2000, 0x1000, 1, 1, &need_locks, false); ASSERT_EQ(vma_new, vma_orig); ASSERT_EQ(vma, vma_orig); ASSERT_EQ(vma_new->vm_start, 0x1000); @@ -53,7 +53,7 @@ static bool test_copy_vma(void) vma = alloc_and_link_vma(&mm, 0x2000, 0x3000, 2, vma_flags); vma_set_anonymous(vma); vma_orig = vma; - vma_new = copy_vma(&vma, 0x1000, 0x1000, 2, 2, &need_locks); + vma_new = copy_vma(&vma, 0x1000, 0x1000, 2, 2, &need_locks, false); ASSERT_EQ(vma_new, vma_orig); ASSERT_EQ(vma, vma_orig); ASSERT_EQ(vma_new->vm_start, 0x1000); @@ -71,7 +71,7 @@ static bool test_copy_vma(void) vma = alloc_and_link_vma(&mm, 0x3000, 0x4000, 3, vma_flags); vma_set_anonymous(vma); vma_orig = vma; - vma_new = copy_vma(&vma, 0x2000, 0x1000, 3, 3, &need_locks); + vma_new = copy_vma(&vma, 0x2000, 0x1000, 3, 3, &need_locks, false); ASSERT_NE(vma_new, vma_orig); ASSERT_EQ(vma_new, vma); ASSERT_EQ(vma_new->vm_start, 0x1000); @@ -82,7 +82,7 @@ static bool test_copy_vma(void) /* Move backwards and do not merge. */ vma = alloc_and_link_vma(&mm, 0x3000, 0x5000, 3, vma_flags); - vma_new = copy_vma(&vma, 0, 0x2000, 0, 3, &need_locks); + vma_new = copy_vma(&vma, 0, 0x2000, 0, 3, &need_locks, false); ASSERT_NE(vma_new, vma); ASSERT_EQ(vma_new->vm_start, 0); ASSERT_EQ(vma_new->vm_end, 0x2000); @@ -95,7 +95,7 @@ static bool test_copy_vma(void) vma = alloc_and_link_vma(&mm, 0, 0x2000, 0, vma_flags); vma_next = alloc_and_link_vma(&mm, 0x6000, 0x8000, 6, vma_flags); - vma_new = copy_vma(&vma, 0x4000, 0x2000, 4, 4, &need_locks); + vma_new = copy_vma(&vma, 0x4000, 0x2000, 4, 4, &need_locks, false); vma_assert_attached(vma_new); ASSERT_EQ(vma_new, vma_next); From 45b45bec90b2a661707c86f500407bde0c797810 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sun, 20 Sep 2026 15:13:11 +0100 Subject: [PATCH 0203/1012] mm/mremap: fix locked_vm leak by splitting VMA for MREMAP_DONTUNMAP The MREMAP_DONTUNMAP feature is highly unusual in that it permits mremap() operations that keep the original VMA in place. Historically this has led to a lot of bugs where non-obvious interactions occur between existing mremap() operations and the original VMA. Fix another of these - partial copies. The long-standing mremap() partial VMA logic has the baked-in assumption that the originating VMA is unmapped and thus moved. However MREMAP_DONTUNMAP defeats this by performing a partial copy instead since it keeps the source VMA around. An mremap(..., MREMAP_DONTUNMAP) operation disallows resizing of the VMA, but the operation can be performed partially: |-----------------| | | | v <------> <------> .new_sz. new_sz |--.------.--| |------| | .source. | | dest | |--.------.--| |------| <------------> old_sz The page tables in the specified range are moved, but the original VMA is kept intact. This interacts poorly with mlock()'d VMAs, as the VMA_LOCKED_BIT flag is cleared for the entire source VMA and set for the entire destination VMA. This results in an mm->locked_vm leak as the change is therefore not accounted correctly. The clear solution here is to make the portion of the source VMA which is mremap()'d distinct from the rest of it, a.k.a. split it. Therefore resolve this issue by splitting it ahead of the rest of the mremap() operation. In order to make this change re-expose split_vma() in vma.h for CONFIG_MMU (nommu doesn't compile mremap.c and uses a static helper instead). A quick search of how MREMAP_DONTUNMAP is used in the wild suggests that the partial case is either unused or rarely used, so this should not result in unreasonable VMA proliferation. Since this makes every mremap() MREMAP_DONTUNMAP operation operate across an entire, distinct, VMA, also eliminate now-redundant code checking for this in dontunmap_complete(). Finally, update the sys_map_count check to account for this case. Link: https://lore.kernel.org/20260920-fix-dontunmap-partial-self-merge-v1-2-6ffb556f8f8b@kernel.org Fixes: e346b3813067 ("mm/mremap: add MREMAP_DONTUNMAP to mremap()") Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Jose A. Perez de Azpillaga Reviewed-by: Pedro Falcato Acked-by: Kiryl Shutsemau (Meta) Cc: Brian Geffon Cc: Jann Horn Cc: Liam Howlett Cc: Minchan Kim Cc: "Vlastimil Babka (SUSE)" Cc: --- mm/mremap.c | 81 +++++++++++++++++++++++++++++++++-------------------- mm/vma.c | 2 +- mm/vma.h | 5 ++++ 3 files changed, 56 insertions(+), 32 deletions(-) diff --git a/mm/mremap.c b/mm/mremap.c index 058dd66fe1d762..a06a9bf2de1d70 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -1037,6 +1037,7 @@ static void vrm_stat_account(struct vma_remap_struct *vrm, } static bool __check_map_count_against_split(struct mm_struct *mm, + bool is_dontunmap, bool before_unmaps) { const int sys_map_count = get_sysctl_max_map_count(); @@ -1088,26 +1089,38 @@ static bool __check_map_count_against_split(struct mm_struct *mm, * Therefore we must check to ensure we have headroom of 2 additional * VMAs. */ - return map_count + 2 <= sys_map_count; + map_count += 2; + + /* + * If MREMAP_DONTUNMAP is set and a partial operation is performed, + * the VMA is split ahead of time and the -1 observed above doesn't + * apply. + */ + if (is_dontunmap) + map_count++; + + return map_count <= sys_map_count; } /* Do we violate the map count limit if we split VMAs when moving the VMA? */ -static bool check_map_count_against_split(void) +static bool check_map_count_against_split(struct vma_remap_struct *vrm) { return __check_map_count_against_split(current->mm, + vrm->flags & MREMAP_DONTUNMAP, /*before_unmaps=*/false); } /* Do we violate the map count limit if we split VMAs prior to early unmaps? */ -static bool check_map_count_against_split_early(void) +static bool check_map_count_against_split_early(struct vma_remap_struct *vrm) { return __check_map_count_against_split(current->mm, + vrm->flags & MREMAP_DONTUNMAP, /*before_unmaps=*/true); } /* - * Perform checks before attempting to write a VMA prior to it being - * moved. + * Perform checks and preparation before attempting to write a VMA prior to it + * being moved. */ static unsigned long prep_move_vma(struct vma_remap_struct *vrm) { @@ -1116,19 +1129,17 @@ static unsigned long prep_move_vma(struct vma_remap_struct *vrm) unsigned long old_addr = vrm->addr; unsigned long old_len = vrm->old_len; vm_flags_t dummy = vma->vm_flags; + const bool split_before = vma->vm_start != old_addr; + const bool split_after = vma->vm_end != old_addr + old_len; - /* - * We'd prefer to avoid failure later on in do_munmap: we copy a VMA, - * which may not merge, then (if MREMAP_DONTUNMAP is not set) unmap the - * source, which may split, causing a net increase of 2 mappings. - */ - if (!check_map_count_against_split()) + /* Avoid failure later on. */ + if (!check_map_count_against_split(vrm)) return -ENOMEM; if (vma->vm_ops && vma->vm_ops->may_split) { - if (vma->vm_start != old_addr) + if (split_before) err = vma->vm_ops->may_split(vma, old_addr); - if (!err && vma->vm_end != old_addr + old_len) + if (!err && split_after) err = vma->vm_ops->may_split(vma, old_addr + old_len); if (err) return err; @@ -1146,6 +1157,21 @@ static unsigned long prep_move_vma(struct vma_remap_struct *vrm) if (err) return err; + /* + * To account mlock()'d pages correctly in the MREMAP_DONTUNMAP + * case perform any split ahead of time. + */ + if (vrm->flags & MREMAP_DONTUNMAP) { + VMA_ITERATOR(vmi, vma->vm_mm, old_addr); + + if (split_before) + err = split_vma(&vmi, vma, old_addr, 1); + if (!err && split_after) + err = split_vma(&vmi, vma, old_addr + old_len, 0); + vrm->vmi_needs_invalidate = true; + return err; + } + return 0; } @@ -1330,11 +1356,8 @@ static int copy_vma_and_data(struct vma_remap_struct *vrm, static void dontunmap_complete(struct vma_remap_struct *vrm, struct vm_area_struct *new_vma) { - unsigned long start = vrm->addr; - unsigned long end = vrm->addr + vrm->old_len; struct vm_area_struct *vma = vrm->vma; - unsigned long old_start = vma->vm_start; - unsigned long old_end = vma->vm_end; + const pgoff_t pgoff_unfaulted = vma->vm_start >> PAGE_SHIFT; /* Self-merge is disallowed. */ VM_WARN_ON_ONCE(new_vma == vma); @@ -1346,19 +1369,15 @@ static void dontunmap_complete(struct vma_remap_struct *vrm, * anon_vma links of the old vma is no longer needed after its page * table has been moved. */ - if (start == old_start && end == old_end) { - const pgoff_t pgoff_unfaulted = vma->vm_start >> PAGE_SHIFT; - - unlink_anon_vmas(vma); - /* - * The VMA is now unfaulted and it is an invariant that - * unfaulted anonymous VMAs have page offset equal to - * vma->vm_start >> PAGE_SHIFT. - */ - vma_set_anon_pgoff(vma, pgoff_unfaulted); - if (vma_is_anonymous(vma) && !vma->vm_file) - vma_set_pgoff(vma, pgoff_unfaulted); - } + unlink_anon_vmas(vma); + /* + * The VMA is now unfaulted and it is an invariant that + * unfaulted anonymous VMAs have page offset equal to + * vma->vm_start >> PAGE_SHIFT. + */ + vma_set_anon_pgoff(vma, pgoff_unfaulted); + if (vma_is_anonymous(vma) && !vma->vm_file) + vma_set_pgoff(vma, pgoff_unfaulted); } static unsigned long move_vma(struct vma_remap_struct *vrm) @@ -2006,7 +2025,7 @@ static unsigned long do_mremap(struct vma_remap_struct *vrm) return -EINTR; vrm->mmap_locked = true; - if (!check_map_count_against_split_early()) { + if (!check_map_count_against_split_early(vrm)) { mmap_write_unlock(mm); return -ENOMEM; } diff --git a/mm/vma.c b/mm/vma.c index d8308c319daa7e..f3f230cbd4595d 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -634,7 +634,7 @@ __split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, * Split a vma into two pieces at address 'addr', a new vma is allocated * either for the first part or the tail. */ -static int split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, +int split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, unsigned long addr, int new_below) { if (vma->vm_mm->map_count >= get_sysctl_max_map_count()) diff --git a/mm/vma.h b/mm/vma.h index 39514fac72fc70..36973abaa015a8 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -556,6 +556,11 @@ int do_brk_flags(struct vma_iterator *vmi, struct vm_area_struct *brkvma, unsigned long unmapped_area(struct vm_unmapped_area_info *info); unsigned long unmapped_area_topdown(struct vm_unmapped_area_info *info); +#ifdef CONFIG_MMU +int split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, + unsigned long addr, int new_below); +#endif + static inline bool vma_wants_manual_pte_write_upgrade(struct vm_area_struct *vma) { /* From 37f484ec42eeb3afc3db71e7234e54ee326516d6 Mon Sep 17 00:00:00 2001 From: John Garry Date: Wed, 23 Sep 2026 08:53:24 +0100 Subject: [PATCH 0204/1012] mailmap: update addresses for John Garry Point any employment addresses at my personal dev address. Link: https://lore.kernel.org/20260923075324.1382927-1-john.garry@linux.dev Signed-off-by: John Garry Signed-off-by: Andrew Morton --- .mailmap | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/.mailmap b/.mailmap index 389b94a0124e31..38b72611b3e995 100644 --- a/.mailmap +++ b/.mailmap @@ -459,7 +459,8 @@ Johan Hovold Johan Hovold John Crispin John Fastabend -John Garry +John Garry +John Garry John Keeping John Moon John Paul Adrian Glaubitz From c131bc1737a892115e12719819dcd48055753338 Mon Sep 17 00:00:00 2001 From: Mikhail Gavrilov Date: Fri, 25 Sep 2026 10:06:47 +0500 Subject: [PATCH 0205/1012] mm: don't schedule deferred kernel page table freeing while booting Booting with a boot-time function tracer and a filter, for example ftrace=function ftrace_filter=pud_free_pmd_page panics on 7.3-rc4 as soon as the tracer starts: [ 23.531178] Starting tracer 'function' [ 23.675800] Oops: general protection fault, probably for non-canonical address 0xdffffc0000000038: 0000 [#1] SMP KASAN NOPTI [ 23.819917] KASAN: null-ptr-deref in range [0x00000000000001c0-0x00000000000001c7] [ 23.964025] CPU: 0 UID: 0 PID: 0 Comm: swapper Not tainted 7.3.0-rc4-fe2ec83746e5-with-fixes-v2+ #195 PREEMPT(undef) [ 24.252248] RIP: 0010:__queue_work+0xab/0xf00 [ 25.981629] Call Trace: [ 26.125727] [ 26.413912] ? pagetable_free_kernel+0x20/0x120 [ 26.990283] queue_work_on+0x97/0xf0 [ 27.134382] __cpa_collapse_large_pages+0x501/0x6f0 [ 27.566662] cpa_flush+0x394/0x620 [ 27.998953] change_page_attr_set_clr+0x321/0x4a0 [ 29.151729] set_memory_rox+0xa2/0xf0 [ 29.584018] create_trampoline+0x431/0x6f0 ... [ 44.343347] Kernel panic - not syncing: Attempted to kill the idle task! The boot-time tracer is started from early_trace_init(), which runs before workqueue_init_early(). Making its trampoline read-only splits a large page, and CPA collapses it again right away. The split table has been a kernel page table since commit 9e4a3ec3411b ("x86/mm/pat: Allocate split page tables as kernel page tables"), so the collapse frees it through pagetable_free_kernel(), which queues work on system_percpu_wq - still NULL at that point. That commit is correct in itself; it only lets CPA reach pagetable_free_kernel() before the workqueue that function relies on exists. Keep putting the table on the list, but don't schedule the work while the system is still booting. A core_initcall schedules it once to free whatever was queued by then. Link: https://lore.kernel.org/20260925050647.86913-1-mikhail.v.gavrilov@gmail.com Link: https://lore.kernel.org/20260924064321.23787-1-mikhail.v.gavrilov@gmail.com Fixes: 9e4a3ec3411b ("x86/mm/pat: Allocate split page tables as kernel page tables") Signed-off-by: Mikhail Gavrilov Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Suggested-by: Lorenzo Stoakes (ARM) Reviewed-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Jose A. Perez de Azpillaga Tested-by: Jose A. Perez de Azpillaga Cc: Dave Hansen Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Vishal Moola (Oracle) Cc: Ingo Molnar Cc: Lu Baolu Cc: Jason Gunthorpe Cc: Steven Rostedt Cc: --- mm/pgtable-generic.c | 20 +++++++++++++++++++- 1 file changed, 19 insertions(+), 1 deletion(-) diff --git a/mm/pgtable-generic.c b/mm/pgtable-generic.c index b91b1a98029c7f..cd227fc05d2d8f 100644 --- a/mm/pgtable-generic.c +++ b/mm/pgtable-generic.c @@ -438,12 +438,30 @@ static void kernel_pgtable_work_func(struct work_struct *work) __pagetable_free(pt); } +static void schedule_kernel_pgtable_free(void) +{ + schedule_work(&kernel_pgtable_work.work); +} + void pagetable_free_kernel(struct ptdesc *pt) { spin_lock(&kernel_pgtable_work.lock); list_add(&pt->pt_list, &kernel_pgtable_work.list); spin_unlock(&kernel_pgtable_work.lock); - schedule_work(&kernel_pgtable_work.work); + /* + * The workqueue may not exist yet while the system is booting. + * kernel_pgtable_drain_early() schedules the work once it does. + */ + if (system_state != SYSTEM_BOOTING) + schedule_kernel_pgtable_free(); +} + +static int __init kernel_pgtable_drain_early(void) +{ + /* Free the kernel page tables queued while booting. */ + schedule_kernel_pgtable_free(); + return 0; } +core_initcall(kernel_pgtable_drain_early); #endif From 9d5de5af971393f6fe64cbdd523cd2369a91a43c Mon Sep 17 00:00:00 2001 From: Carlos Llamas Date: Sun, 27 Sep 2026 16:24:18 +0000 Subject: [PATCH 0206/1012] selftests/mm: cleanup -Wformat issues in hugetlb-mmap Commit ae571cd6015c ("selftests/mm: hugetlb-mmap: add setup of HugeTLB pages") and commit 9c5a65f374f8 ("selftests/mm: merge map_hugetlb into hugepage-mmap") added logs of 'hugepage_size' which has a size_t type. However, the incorrect format specifier '%lu' was used which triggers -Wformat warnings when building for 32-bit: hugetlb-mmap.c:125:55: warning: format specifies type 'unsigned long' but the argument has type 'size_t' (aka 'unsigned int') [-Wformat] 125 | ksft_print_msg("Default size hugepages (%lu kB)\n", hugepage_size >> 10); | ~~~ ^~~~~~~~~~~~~~~~~~~ | %zu hugetlb-mmap.c:134:47: warning: format specifies type 'unsigned long' but the argument has type 'size_t' (aka 'unsigned int') [-Wformat] 134 | ksft_exit_skip("Not enough %lu Kb pages\n", hugepage_size >> 10); | ~~~ ^~~~~~~~~~~~~~~~~~~ | %zu Fix this by switching to the expected '%zu' format specifier. Link: https://lore.kernel.org/20260927162419.820609-1-cmllamas@google.com Fixes: ae571cd6015c ("selftests/mm: hugetlb-mmap: add setup of HugeTLB pages") Fixes: 9c5a65f374f8 ("selftests/mm: merge map_hugetlb into hugepage-mmap") Signed-off-by: Carlos Llamas Signed-off-by: Andrew Morton Reviewed-by: Sarthak Sharma Reviewed-by: SJ Park Acked-by: Lorenzo Stoakes (ARM) Cc: Mike Rapoport Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan Cc: --- tools/testing/selftests/mm/hugetlb-mmap.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tools/testing/selftests/mm/hugetlb-mmap.c b/tools/testing/selftests/mm/hugetlb-mmap.c index 0f2aad1b7dbd6f..2edbc992e6bcf0 100644 --- a/tools/testing/selftests/mm/hugetlb-mmap.c +++ b/tools/testing/selftests/mm/hugetlb-mmap.c @@ -122,7 +122,7 @@ int main(int argc, char **argv) hugepage_size = default_huge_page_size(); if (!hugepage_size) ksft_exit_skip("Could not detect default hugetlb page size."); - ksft_print_msg("Default size hugepages (%lu kB)\n", hugepage_size >> 10); + ksft_print_msg("Default size hugepages (%zu kB)\n", hugepage_size >> 10); } /* munmap will fail if the length is not page aligned */ @@ -131,7 +131,7 @@ int main(int argc, char **argv) hugetlb_set_nr_pages(hugepage_size, nr); if (hugetlb_free_pages(hugepage_size) < nr) - ksft_exit_skip("Not enough %lu Kb pages\n", hugepage_size >> 10); + ksft_exit_skip("Not enough %zu Kb pages\n", hugepage_size >> 10); ksft_set_plan(2); ksft_print_msg("Mapping %lu Mbytes\n", (unsigned long)length >> 20); From eb6c69dd890d9634f177c3296bc443cf40b19665 Mon Sep 17 00:00:00 2001 From: Donggeun Yoo Date: Sat, 26 Sep 2026 21:41:44 +0900 Subject: [PATCH 0207/1012] userfaultfd: clear the inherited uffd bit in move_swap_pte() Patch series "userfaultfd: clear the inherited uffd bit in move_swap_pte()", v3. UFFDIO_MOVE on a swapped-out page installs the source PTE at the destination unchanged, so a uffd bit set on a write-protected or RWP-protected source lands in a destination VMA that was never registered for either, and nothing clears it afterwards. Patch 1 clears the bit, then re-arms it if the destination is RWP-registered, which is what the present-page and zeropage move paths already do. Patch 2 adds the tests that catch it. This patch (of 2): UFFDIO_MOVE on a swapped-out page installs the source PTE at the destination unchanged, so a uffd bit set on a write-protected or RWP-protected source lands in a destination VMA that was never registered for either. Nothing clears it there, and userspace sees: - /proc//pagemap reports the page as uffd-tracked (bit 57), both while it is swapped out and after it is faulted back in. - MADV_COLLAPSE fails with EINVAL on a range containing it, because the collapse scan bails on a swap entry with the uffd bit set. - With CONFIG_PAGE_TABLE_CHECK, faulting the page in warns. do_swap_page() carries the bit into the present PTE and, since the destination isn't WP-registered, also makes it writable: WARNING: mm/page_table_check.c:202 at __page_table_check_ptes_set+0x185/0x1e0 Call Trace: set_ptes+0x67/0xc0 do_swap_page+0x990/0xfe0 __handle_mm_fault+0x7d0/0xeb0 handle_mm_fault+0x9c/0x250 do_user_addr_fault+0x207/0x650 exc_page_fault+0x65/0x150 asm_exc_page_fault+0x26/0x30 Reaching it takes UFFDIO_MOVE out of a WP- or RWP-protected area into one that isn't, on a page that is swapped out at the time. It hasn't been seen in practice. A resident page doesn't carry the bit, because move_present_ptes() builds the destination PTE from dst_vma->vm_page_prot. Clear it on the moved swap entry as well, then re-arm it if the destination is RWP-registered. The WP case has been there since v6.8, where UFFDIO_MOVE was added. Link: https://lore.kernel.org/20260926124145.2878520-1-donggeunyoo.kernel@gmail.com Link: https://lore.kernel.org/20260926124145.2878520-2-donggeunyoo.kernel@gmail.com Fixes: adef440691ba ("userfaultfd: UFFDIO_MOVE uABI") Signed-off-by: Donggeun Yoo Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Mike Rapoport Cc: Peter Xu Cc: Suren Baghdasaryan Cc: Andrea Arcangeli Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Michal Hocko Cc: Shuah Khan Cc: Kiryl Shutsemau Cc: --- mm/userfaultfd.c | 1 + 1 file changed, 1 insertion(+) diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index 74f04c323c50fb..f39f109f17989e 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -1449,6 +1449,7 @@ static int move_swap_pte(struct mm_struct *mm, struct vm_area_struct *dst_vma, orig_src_pte = ptep_get_and_clear(mm, src_addr, src_pte); if (pgtable_supports_soft_dirty()) orig_src_pte = pte_swp_mksoft_dirty(orig_src_pte); + orig_src_pte = pte_swp_clear_uffd(orig_src_pte); /* Re-arm RWP on the moved swap entry if dst_vma is RWP-registered. */ if (userfaultfd_rwp(dst_vma)) orig_src_pte = pte_swp_mkuffd(orig_src_pte); From bab1062cce19f0ce5ccf5782d3d588ffdbed0427 Mon Sep 17 00:00:00 2001 From: Donggeun Yoo Date: Sat, 26 Sep 2026 21:41:45 +0900 Subject: [PATCH 0208/1012] selftests/mm: add tests for UFFDIO_MOVE of a uffd-protected swap entry Move a swapped-out page out of a write-protected or RWP-protected area into a destination registered for missing faults only, and read pagemap bit 57 at the destination. The destination was never protected, so the bit must be clear. Link: https://lore.kernel.org/20260926124145.2878520-3-donggeunyoo.kernel@gmail.com Signed-off-by: Donggeun Yoo Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Mike Rapoport Cc: Peter Xu Cc: Suren Baghdasaryan Cc: Andrea Arcangeli Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Michal Hocko Cc: Shuah Khan Cc: Kiryl Shutsemau --- tools/testing/selftests/mm/uffd-unit-tests.c | 84 ++++++++++++++++++++ 1 file changed, 84 insertions(+) diff --git a/tools/testing/selftests/mm/uffd-unit-tests.c b/tools/testing/selftests/mm/uffd-unit-tests.c index ef9b3956bdcfd9..580178630eded8 100644 --- a/tools/testing/selftests/mm/uffd-unit-tests.c +++ b/tools/testing/selftests/mm/uffd-unit-tests.c @@ -2037,6 +2037,75 @@ static void uffd_move_pmd_split_test(uffd_global_test_opts_t *gopts, uffd_test_a uffd_move_pmd_handle_fault); } +/* + * Moving a swapped-out page out of a write-protected or RWP-protected area + * must not carry the uffd bit into a destination registered for missing + * faults only: such a bit is never cleared afterwards, so pagemap keeps + * reporting the page as uffd-tracked. + * + * Needs a swap device; skipped if MADV_PAGEOUT cannot evict the page. + */ +static void uffd_move_swap_test_common(uffd_global_test_opts_t *gopts, + bool rwp) +{ + unsigned long page_size = gopts->page_size; + struct uffdio_move move = { }; + int pagemap_fd; + + if (rwp) { + if (uffd_register_rwp(gopts->uffd, gopts->area_src, page_size)) + err("register src failure"); + } else if (uffd_register(gopts->uffd, gopts->area_src, page_size, + false, true, false)) { + err("register src failure"); + } + if (uffd_register(gopts->uffd, gopts->area_dst, page_size, + true, false, false)) + err("register dst failure"); + + if (rwp) + rwprotect_range(gopts->uffd, (unsigned long)gopts->area_src, + page_size, true); + else + wp_range(gopts->uffd, (unsigned long)gopts->area_src, + page_size, true); + + pagemap_fd = pagemap_open(); + if (madvise(gopts->area_src, page_size, MADV_PAGEOUT)) + err("MADV_PAGEOUT"); + if (!pagemap_is_swapped(pagemap_fd, gopts->area_src)) { + uffd_test_skip("MADV_PAGEOUT did not swap the page; is swap enabled?"); + goto out; + } + + move.dst = (unsigned long)gopts->area_dst; + move.src = (unsigned long)gopts->area_src; + move.len = page_size; + if (ioctl(gopts->uffd, UFFDIO_MOVE, &move)) + err("UFFDIO_MOVE"); + + if (pagemap_get_entry(pagemap_fd, gopts->area_dst) & PM_UFFD_WP) + uffd_test_fail("uffd bit moved into an area registered for missing faults only"); + else + uffd_test_pass(); +out: + close(pagemap_fd); + uffd_unregister(gopts->uffd, gopts->area_src, page_size); + uffd_unregister(gopts->uffd, gopts->area_dst, page_size); +} + +static void uffd_move_swap_wp_test(uffd_global_test_opts_t *gopts, + uffd_test_args_t *targs) +{ + uffd_move_swap_test_common(gopts, false); +} + +static void uffd_move_swap_rwp_test(uffd_global_test_opts_t *gopts, + uffd_test_args_t *targs) +{ + uffd_move_swap_test_common(gopts, true); +} + static bool uffdio_verify_results(const char *name, int ret, int error, long result) { @@ -2359,6 +2428,21 @@ uffd_test_case_t uffd_tests[] = { .uffd_feature_required = UFFD_FEATURE_MOVE, .test_case_ops = &uffd_move_test_pmd_case_ops, }, + { + .name = "move-swap-wp", + .uffd_fn = uffd_move_swap_wp_test, + .mem_targets = MEM_ANON, + .uffd_feature_required = UFFD_FEATURE_MOVE | + UFFD_FEATURE_PAGEFAULT_FLAG_WP, + .test_case_ops = &uffd_move_test_case_ops, + }, + { + .name = "move-swap-rwp", + .uffd_fn = uffd_move_swap_rwp_test, + .mem_targets = MEM_ANON, + .uffd_feature_required = UFFD_FEATURE_MOVE | UFFD_FEATURE_RWP, + .test_case_ops = &uffd_move_test_case_ops, + }, { .name = "wp-fork", .uffd_fn = uffd_wp_fork_test, From 4e913faf4498715acd01e6569fffb026fde3e274 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 24 Sep 2026 15:48:24 +0100 Subject: [PATCH 0209/1012] drivers/char/mem: mmap readonly MAP_SHARED-/dev/zero correctly Rather surprisingly, opening /dev/zero read-only then mmap()'ing it MAP_SHARED gets you true anonymous memory (albeit in a VMA with non-NULL vma->vm_file). This is a by-product of MAP_PRIVATE-/dev/zero being how anonymous memory was mapped in Linux's distant past. It happens because mmap_zero_prepare() gates on VMA_SHARED_BIT and when mapping a read-only file MAP_SHARED, do_mmap() clears VMA_SHARED_BIT and VMA_MAYWRITE_BIT. The gating is incorrect - the (poorly named) VMA_MAYSHARE_BIT flag exists explicitly to tell you if something was originally mapped MAP_SHARED. So the fix is simple - gate on this instead. This isn't exactly a common use case, but it's unexpected behaviour which now causes an assert if CONFIG_DEBUG_VM is set. While this bug has existed since the dawn of time for linux (or at least since 2.6.12), it hasn't caused issues in the past, so while it's incorrect behaviour, it doesn't seem necessary to backport that far. The mapping is now accounted at mmap time and can fail with -ENOMEM under strict overcommit, and read faults allocate folios. However this is normal behaviour for a read-only shmem mapping. Commit 93c0c8dc87f6 ("mm/rmap: use anon pgoff to track MAP_PRIVATE file-backed anon folios") is the first patch at which the debug assert fires, so target that instead. Link: https://lore.kernel.org/20260924-fix-dev-zero-readonly-shared-v1-1-153c2111e323@kernel.org Fixes: 93c0c8dc87f6 ("mm/rmap: use anon pgoff to track MAP_PRIVATE file-backed anon folios") Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reported-by: Closes: https://lore.kernel.org/linux-mm/6ab4ae75.80e1c6cc.1e8e5f.000d.GAE@google.com/ Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Lance Yang --- drivers/char/mem.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/char/mem.c b/drivers/char/mem.c index 63253d1de5d70b..5b93c92c2cf194 100644 --- a/drivers/char/mem.c +++ b/drivers/char/mem.c @@ -503,7 +503,7 @@ static int mmap_zero_prepare(struct vm_area_desc *desc) #ifndef CONFIG_MMU return -ENOSYS; #endif - if (vma_desc_test(desc, VMA_SHARED_BIT)) + if (vma_desc_test(desc, VMA_MAYSHARE_BIT)) return shmem_zero_setup_desc(desc); /* From d173d2c96f3eeafca83751dc9c4e4cd47e08bb7c Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Tue, 29 Sep 2026 18:45:51 +0100 Subject: [PATCH 0210/1012] mm: page_alloc: make defrag_mode retries follow the promoted order Since commit 7e8756d7ad22 ("mm: page_alloc: fix non-movable reclaim storm in defrag_mode"), direct reclaim and compaction for non-movable requests under defrag_mode run at pageblock_order, to produce the whole blocks that ALLOC_NOFRAGMENT needs. The retry decisions that follow still use the request order. An order-0 request can therefore retry indefinitely without ever reaching the ALLOC_NOFRAGMENT fallback: - Reclaim at pageblock_order gives up after one pass as soon as a zone looks compaction_ready(), and do_try_to_free_pages() then returns 1 even though nothing was reclaimed. It returns before the retry that would reclaim memory.low-protected cgroups, so when most memory is protected, the pass that did run finds next to nothing. - Compaction at pageblock_order fails or is deferred. - should_reclaim_retry() takes the reported progress as progress for the order-0 request and resets no_progress_loops. The request retries. Order 1-3 requests loop the same way, and should_compact_retry() also checks their pageblock_order compaction result against the request order. On a production host (64G, defrag_mode, memory.low covering most of the workload), 95% of direct reclaim runs were order-9 runs that returned 1 with nothing reclaimed, at up to 60k runs per second. Across ~200M should_reclaim_retry() calls in a day, no_progress_loops never left 0. The spinning allocations were SLUB slab refills for inode and dentry caches. The time spent registers as memory pressure, and pressure-based OOM killing takes down both workloads and system services. Treat promoted requests like costly orders: - Reclaim progress does not reset no_progress_loops for them. - should_compact_retry() checks the compaction result at the promoted order. It does not retry COMPACT_SKIPPED, since the request can fall back, and it does not escalate compaction to COMPACT_PRIO_SYNC_FULL. When the fallback is taken, reset the retry counters, so that the fallback attempt gets a full retry budget before the OOM killer is considered. In a VM reproducer (32G, defrag_mode, inode churn under memory.low): before after should_reclaim_retry() calls 63M 293k peak memory pressure (PSI some avg10) 99% 12% File creation runs 5.7x faster. Link: https://lore.kernel.org/20260929174553.175333-1-kirill@shutemov.name Fixes: 7e8756d7ad22 ("mm: page_alloc: fix non-movable reclaim storm in defrag_mode") Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Vlastimil Babka Cc: Johannes Weiner Cc: David Hildenbrand Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Brendan Jackman Cc: Zi Yan Cc: Shakeel Butt Cc: Usama Arif Cc: Harry Yoo --- mm/page_alloc.c | 85 ++++++++++++++++++++++++++++++++----------------- 1 file changed, 56 insertions(+), 29 deletions(-) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 542c2ec3106134..31eeaceac6beda 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -4127,6 +4127,31 @@ __alloc_pages_may_oom(gfp_t gfp_mask, unsigned int order, return page; } +/* + * If fallbacks are not permitted (defrag_mode), we either need to + * reclaim space in a block of matching type, or clear out an entire + * block to allow __rmqueue_claim() to convert. + * + * Reclaim by itself is primarily freeing space in movable blocks, + * since that's where the LRU pages live. So this works for movable + * requests, but not for others. + * + * For those, promote the order of reclaim and compaction to help make + * blocks, instead of spinning in reclaim alone unproductively. Retry + * decisions based on the outcome of that work - reclaim progress and + * compaction results - must account for the promotion as well, see + * should_reclaim_retry() and should_compact_retry(). + */ +static inline unsigned int nofrag_promote_order(unsigned int order, + unsigned int alloc_flags, + const struct alloc_context *ac) +{ + if ((alloc_flags & ALLOC_NOFRAGMENT) && ac->migratetype != MIGRATE_MOVABLE) + return max(order, pageblock_order); + + return order; +} + /* * Maximum number of compaction retries with a progress before OOM * killer is consider as the only way to move forward. @@ -4149,22 +4174,7 @@ __alloc_pages_direct_compact(gfp_t gfp_mask, unsigned int order, .order = order, .page = NULL, }; - int compact_order = order; - - /* - * If fallbacks are not permitted (defrag_mode), we either - * need to reclaim space in a block of matching type, or clear - * out an entire block to allow __rmqueue_claim() to convert. - * - * Reclaim by itself is primarily freeing space in movable - * blocks, since that's where the LRU pages live. So this - * works for movable requests, but not for others. - * - * For those, promote the order to help make blocks, instead - * of spinning in reclaim alone unproductively. - */ - if ((alloc_flags & ALLOC_NOFRAGMENT) && ac->migratetype != MIGRATE_MOVABLE) - compact_order = max(order, pageblock_order); + unsigned int compact_order = nofrag_promote_order(order, alloc_flags, ac); if (!compact_order) return NULL; @@ -4256,8 +4266,11 @@ should_compact_retry(gfp_t gfp_mask, struct alloc_context *ac, int order, bool ret = false; int retries = *compaction_retries; enum compact_priority priority = *compact_priority; + unsigned int compact_order; - if (!order) + /* Check the compaction result at the order compaction ran at */ + compact_order = nofrag_promote_order(order, alloc_flags, ac); + if (!compact_order) return false; if (fatal_signal_pending(current)) @@ -4266,10 +4279,14 @@ should_compact_retry(gfp_t gfp_mask, struct alloc_context *ac, int order, /* * Compaction was skipped due to a lack of free order-0 * migration targets. Continue if reclaim can help. + * + * Promoted requests have exhausted their reclaim retries at + * this point, and they can fall back instead. */ if (compact_result == COMPACT_SKIPPED) { - ret = compaction_zonelist_suitable(ac, order, alloc_flags, - gfp_mask); + if (compact_order == order) + ret = compaction_zonelist_suitable(ac, order, alloc_flags, + gfp_mask); goto out; } @@ -4287,7 +4304,7 @@ should_compact_retry(gfp_t gfp_mask, struct alloc_context *ac, int order, * need much more detailed feedback from compaction to * make a better decision. */ - if (order > PAGE_ALLOC_COSTLY_ORDER) + if (compact_order > PAGE_ALLOC_COSTLY_ORDER) max_retries /= 4; if (++(*compaction_retries) <= max_retries) { @@ -4299,7 +4316,7 @@ should_compact_retry(gfp_t gfp_mask, struct alloc_context *ac, int order, /* * Compaction failed. Retry with increasing priority. */ - min_priority = (order > PAGE_ALLOC_COSTLY_ORDER) ? + min_priority = (compact_order > PAGE_ALLOC_COSTLY_ORDER) ? MIN_COMPACT_COSTLY_PRIORITY : MIN_COMPACT_PRIORITY; if (*compact_priority > min_priority) { @@ -4468,11 +4485,7 @@ __alloc_pages_direct_reclaim(gfp_t gfp_mask, unsigned int order, struct page *page = NULL; unsigned long pflags; bool drained = false; - int reclaim_order = order; - - /* Match the slowpath compaction promotion in __alloc_pages_direct_compact */ - if ((alloc_flags & ALLOC_NOFRAGMENT) && ac->migratetype != MIGRATE_MOVABLE) - reclaim_order = max(order, pageblock_order); + unsigned int reclaim_order = nofrag_promote_order(order, alloc_flags, ac); psi_memstall_enter(&pflags); *did_some_progress = __perform_reclaim(gfp_mask, reclaim_order, ac); @@ -4648,9 +4661,17 @@ should_reclaim_retry(gfp_t gfp_mask, unsigned order, /* * Costly allocations might have made a progress but this doesn't mean * their order will become available due to high fragmentation so - * always increment the no progress counter for them + * always increment the no progress counter for them. + * + * The same goes for requests whose reclaim is promoted to make whole + * blocks. At that order, reclaim also reports progress when it backs + * off for compaction without freeing anything. + * + * The watermark check below stays at the request order: it asks + * whether the request itself could succeed after reclaim. */ - if (did_some_progress && order <= PAGE_ALLOC_COSTLY_ORDER) + if (did_some_progress && order <= PAGE_ALLOC_COSTLY_ORDER && + nofrag_promote_order(order, alloc_flags, ac) == order) *no_progress_loops = 0; else (*no_progress_loops)++; @@ -5024,9 +5045,15 @@ __alloc_pages_slowpath(gfp_t gfp_mask, unsigned int order, &compaction_retries)) goto retry; - /* Reclaim/compaction failed to prevent the fallback */ + /* + * Reclaim/compaction failed to prevent the fallback. The retry + * budget was spent on making blocks, not on the request itself; + * give the fallback a fresh one before considering OOM. + */ if (defrag_mode && (alloc_flags & ALLOC_NOFRAGMENT)) { alloc_flags &= ~ALLOC_NOFRAGMENT; + no_progress_loops = 0; + compaction_retries = 0; goto retry; } From 991676a81d2c0ea4ed5349bd62934b7fe48abcc9 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Wed, 19 Aug 2026 17:51:43 +0800 Subject: [PATCH 0211/1012] mm: use a folio in the softleaf_is_device_private path Use the folio APIs in the device_private migration path of do_swap_page(), replacing four calls to compound_head() with two page_folio() calls. The second one re-fetches the folio from vmf->page after migrate_to_ram(), which might have split the folio. Link: https://lore.kernel.org/20260819095144.45660-1-hongfu.li@linux.dev Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Link: https://lore.kernel.org/all/e20678ed-3fa1-4677-a1d7-e2af481e8302@kernel.org/ Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Anshuman Khandual Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/memory.c | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 8b0c2c735d3de7..9cbce5c90bffde 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4926,18 +4926,21 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) goto unlock; /* - * Get a page reference while we know the page can't be - * freed. + * Get a folio reference while we know the folio can't + * be freed. */ - if (trylock_page(vmf->page)) { + folio = page_folio(vmf->page); + if (folio_trylock(folio)) { struct dev_pagemap *pgmap; - get_page(vmf->page); + folio_get(folio); pte_unmap_unlock(vmf->pte, vmf->ptl); pgmap = page_pgmap(vmf->page); ret = pgmap->ops->migrate_to_ram(vmf); - unlock_page(vmf->page); - put_page(vmf->page); + /* migrate_to_ram() might have split the folio. */ + folio = page_folio(vmf->page); + folio_unlock(folio); + folio_put(folio); } else { pte_unmap(vmf->pte); softleaf_entry_wait_on_locked(entry, vmf->ptl); From 08142ee1275024d1f5b9ec5ad12394cf1989c016 Mon Sep 17 00:00:00 2001 From: Qi Xi Date: Wed, 19 Aug 2026 16:20:52 +0800 Subject: [PATCH 0212/1012] mm: drop stale MAX_ORDER references The treewide rename in commit 5e0a760b4441 ("mm, treewide: rename MAX_ORDER to MAX_PAGE_ORDER") left a few spots still using the old name: - two comments in include/net/mana/mana.h and mm/page_alloc.c; - the gdb helper scripts/gdb/linux/mm.py, where self.MAX_ORDER is a local mirror of the kernel's MAX_ORDER define. Rename the leftover instances to MAX_PAGE_ORDER so the tree is consistent. No functional changes. Link: https://lore.kernel.org/20260819082052.3338603-1-xiqi2@huawei.com Signed-off-by: Qi Xi Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Jan Kiszka Cc: Johannes Weiner Cc: Kefeng Wang Cc: Kieran Bingham Cc: Konstantin Taranov Cc: Long Li Cc: Michal Hocko Cc: Nanyong Sun Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/net/mana/mana.h | 4 ++-- mm/page_alloc.c | 2 +- scripts/gdb/linux/mm.py | 8 ++++---- 3 files changed, 7 insertions(+), 7 deletions(-) diff --git a/include/net/mana/mana.h b/include/net/mana/mana.h index 83b7eff4646ead..e8fda092be37fb 100644 --- a/include/net/mana/mana.h +++ b/include/net/mana/mana.h @@ -47,8 +47,8 @@ enum mana_priv_flag_bits { #define COMP_ENTRY_SIZE 64 /* This Max value for RX buffers is derived from __alloc_page()'s max page - * allocation calculation. It allows maximum 2^(MAX_ORDER -1) pages. RX buffer - * size beyond this value gets rejected by __alloc_page() call. + * allocation calculation. It allows maximum 2^MAX_PAGE_ORDER pages. RX + * buffer size beyond this value gets rejected by __alloc_page() call. */ #define MAX_RX_BUFFERS_PER_QUEUE 8192 #define DEF_RX_BUFFERS_PER_QUEUE 1024 diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 31eeaceac6beda..104a07031f30a1 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -8017,7 +8017,7 @@ static bool cond_accept_memory(struct zone *zone, unsigned int order, /* * Watermarks have not been initialized yet. * - * Accepting one MAX_ORDER page to ensure progress. + * Accepting one MAX_PAGE_ORDER page to ensure progress. */ if (!wmark) return try_to_accept_memory_one(zone); diff --git a/scripts/gdb/linux/mm.py b/scripts/gdb/linux/mm.py index dffadccbb01d24..28d33624c38bd6 100644 --- a/scripts/gdb/linux/mm.py +++ b/scripts/gdb/linux/mm.py @@ -56,7 +56,7 @@ def __init__(self): self.MAX_PHYSMEM_BITS = 46 self.SECTION_SIZE_BITS = 27 - self.MAX_ORDER = 10 + self.MAX_PAGE_ORDER = 10 self.SECTIONS_SHIFT = self.MAX_PHYSMEM_BITS - self.SECTION_SIZE_BITS self.NR_MEM_SECTIONS = 1 << self.SECTIONS_SHIFT @@ -233,11 +233,11 @@ def __init__(self): self.SECTIONS_SHIFT = self.MAX_PHYSMEM_BITS - self.SECTION_SIZE_BITS if str(constants.LX_CONFIG_ARCH_FORCE_MAX_ORDER).isdigit(): - self.MAX_ORDER = constants.LX_CONFIG_ARCH_FORCE_MAX_ORDER + self.MAX_PAGE_ORDER = constants.LX_CONFIG_ARCH_FORCE_MAX_ORDER else: - self.MAX_ORDER = 10 + self.MAX_PAGE_ORDER = 10 - self.MAX_ORDER_NR_PAGES = 1 << (self.MAX_ORDER) + self.MAX_ORDER_NR_PAGES = 1 << (self.MAX_PAGE_ORDER) self.PFN_SECTION_SHIFT = self.SECTION_SIZE_BITS - self.PAGE_SHIFT self.NR_MEM_SECTIONS = 1 << self.SECTIONS_SHIFT self.PAGES_PER_SECTION = 1 << self.PFN_SECTION_SHIFT From db95bf12a0b07020093350979725421eab123704 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Wed, 26 Aug 2026 20:44:09 +0800 Subject: [PATCH 0213/1012] mm/vmscan: drop the combined limit gate in __node_reclaim() __node_reclaim() is called from two paths: node_reclaim() and user_proactive_reclaim(). node_reclaim() already bails out early unless node_pagecache_reclaimable() is over pgdat->min_unmapped_pages or the reclaimable slab is over pgdat->min_slab_pages. The identical check inside __node_reclaim() that guards the shrink_node() loop is therefore redundant for this path. user_proactive_reclaim() is proactive reclaim driven by userspace and should not be gated by the per-node min_unmapped_pages / min_slab_pages limits at all [1]. With the gate in place, a proactive request is silently turned into a no-op whenever the node happens to sit below both thresholds. Drop the gate in __node_reclaim() and always run the shrink_node() loop. The node_reclaim() path is unchanged, since its caller has already applied the same test; the proactive path is no longer wrongly gated. Link: https://lore.kernel.org/20260826124409.35569-1-ridong.chen@linux.dev Link: https://sashiko.dev/#/patchset/20260723045718.2052070-1-ridong.chen@linux.dev [1] Fixes: b980077899ea ("mm: introduce per-node proactive reclaim interface") Acked-by: Johannes Weiner Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Shakeel Butt Acked-by: Davidlohr Bueso Assisted-by: Claude:claude-opus-4-8 Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Roman Gushchin Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 13 +++---------- 1 file changed, 3 insertions(+), 10 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index f11491ee9ed5c1..6dff207ad8c612 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -7851,16 +7851,9 @@ static unsigned long __node_reclaim(struct pglist_data *pgdat, noreclaim_flag = memalloc_noreclaim_save(); set_task_reclaim_state(p, &sc->reclaim_state); - if (node_pagecache_reclaimable(pgdat) > pgdat->min_unmapped_pages || - node_page_state_pages(pgdat, NR_SLAB_RECLAIMABLE_B) > pgdat->min_slab_pages) { - /* - * Free memory by calling shrink node with increasing - * priorities until we have enough memory freed. - */ - do { - shrink_node(pgdat, sc); - } while (sc->nr_reclaimed < nr_pages && --sc->priority >= 0); - } + do { + shrink_node(pgdat, sc); + } while (sc->nr_reclaimed < nr_pages && --sc->priority >= 0); set_task_reclaim_state(p, NULL); memalloc_noreclaim_restore(noreclaim_flag); From 1a76906294c65f685b842b305664e131384588dc Mon Sep 17 00:00:00 2001 From: JonasZhou Date: Tue, 25 Aug 2026 18:46:59 +0800 Subject: [PATCH 0214/1012] mm/vmalloc: avoid false sharing with drain_vmap_work free_vmap_area_noflush() queues drain_vmap_work after the number of lazily freed pages exceeds lazy_max_pages(). Until the worker purges those pages, concurrent frees keep calling schedule_work(). Even if the work is already pending, queue_work_on() performs a locked test_and_set_bit() on the pending bit in the work item. On the tested x86-64 build, drain_vmap_work and vmap_nodes occupy the same 64-byte cache line. The work item starts at offset 0 and the vmap_nodes pointer at offset 32. The latter is read by vmap allocation and free paths, so updates to the work item invalidate a cache line read by all CPUs. Put drain_vmap_work in the cacheline-aligned data section. Tests were run on Linux 7.2. On a two-socket Intel Xeon Silver 4208 system using 16 workers, the runtimes of vmalloc.fix_align, vmalloc.fix_size, and vmalloc.no_block_alloc decreased by 12.61%, 5.78%, and 6.87%, respectively. HITM samples for the affected cache line and total HITM samples decreased by 96.55% and 13.36%, respectively. Link: https://lore.kernel.org/20260825104659.100134-1-jonaszhou-oc@zhaoxin.com Signed-off-by: JonasZhou Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Cc: --- mm/vmalloc.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 89c327a6ce7d9f..b879260d31a577 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -1090,7 +1090,12 @@ RB_DECLARE_CALLBACKS_MAX(static, free_vmap_area_rb_augment_cb, static void reclaim_and_purge_vmap_areas(void); static BLOCKING_NOTIFIER_HEAD(vmap_notify_list); static void drain_vmap_area_work(struct work_struct *work); -static DECLARE_WORK(drain_vmap_work, drain_vmap_area_work); +/* + * Keep the work item, whose pending bit is updated by freeing CPUs, + * away from vmap metadata read by allocation and free paths. + */ +static __cacheline_aligned_in_smp +DECLARE_WORK(drain_vmap_work, drain_vmap_area_work); static struct workqueue_struct *drain_vmap_helpers_wq; static struct workqueue_struct *drain_vmap_wq; From 28e8cace79b9c4308e72f35648901d5d9e5fe734 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Tue, 25 Aug 2026 10:10:13 +0800 Subject: [PATCH 0215/1012] mm/hugetlb: fix resv_huge_pages double decrement in memfd error path alloc_hugetlb_folio_reserve() decrements h->resv_huge_pages when dequeuing a folio, but unlike the use_global_reservation handling in hugetlb_alloc_folio(), it does not set HPageRestoreReserve on the folio. Its sole caller memfd_alloc_folio() pre-allocates a reservation via hugetlb_reserve_pages() before allocating. When hugetlb_add_to_page_cache() fails, folio_put() drops the folio without HPageRestoreReserve set, so free_huge_folio() does not restore the reservation. The subsequent hugetlb_unreserve_pages() on the err_unresv path decrements the counter a second time, leaving resv_huge_pages off by one for every failed allocation. Set HPageRestoreReserve when consuming the reservation in alloc_hugetlb_folio_reserve(). On the error path, free_huge_folio() then restores the reservation before hugetlb_unreserve_pages() releases it. The success path is unaffected, as hugetlb_add_to_page_cache() clears the flag once the folio is added to the page cache. Link: https://lore.kernel.org/20260825021013.25672-1-hongfu.li@linux.dev Fixes: 26a8ea80929c ("mm/hugetlb: fix memfd_pin_folios resv_huge_pages leak") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Cc: David Hildenbrand Cc: Oscar Salvador Cc: Steven Sistare Cc: Vivek Kasireddy --- mm/hugetlb.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 49ffbcb54f8c08..3af46101f1bbce 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -2187,8 +2187,10 @@ struct folio *alloc_hugetlb_folio_reserve(struct hstate *h, int preferred_nid, folio = dequeue_hugetlb_folio_nodemask(h, gfp_mask, preferred_nid, nmask); - if (folio) + if (folio) { + folio_set_hugetlb_restore_reserve(folio); h->resv_huge_pages--; + } spin_unlock_irq(&hugetlb_lock); return folio; From f1cab9cd613e9ff087f1fc2d40fc0402cd46198f Mon Sep 17 00:00:00 2001 From: Hemanth Selam Date: Tue, 25 Aug 2026 21:47:15 +0530 Subject: [PATCH 0216/1012] selftests/mm: remove the local PKEY_UNRESTRICTED fallback pkey-helpers.h defines PKEY_UNRESTRICTED itself when the macro is not already known, a stopgap from when the generic definition was still under review. It has been merged since, commit 6d61527d931b ("mm/pkey: Add PKEY_UNRESTRICTED macro"), so the guard is never taken and the FIXME can be honoured. The definition comes from tools/include/uapi/asm-generic/mman-common.h via TOOLS_INCLUDES, which commit e076eaca5906 ("selftests: break the dependency upon local header files") added so that the mm selftests build without "make headers". It is reached through the that the system includes. Building the pkey tests with KHDR_INCLUDES pointing at an empty directory confirms that; emptying TOOLS_INCLUDES as well is what makes the macro go missing. No functional change intended. Link: https://lore.kernel.org/20260825161715.2807297-1-hemanth.selam@gmail.com Signed-off-by: Hemanth Selam Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: SJ Park Assisted-by: Cursor:claude-opus-5 Cc: Kevin Brodsky Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Yury Khrustalev --- tools/testing/selftests/mm/pkey-helpers.h | 7 ------- 1 file changed, 7 deletions(-) diff --git a/tools/testing/selftests/mm/pkey-helpers.h b/tools/testing/selftests/mm/pkey-helpers.h index 46a8a1878dc1fd..9b949951743dd6 100644 --- a/tools/testing/selftests/mm/pkey-helpers.h +++ b/tools/testing/selftests/mm/pkey-helpers.h @@ -114,13 +114,6 @@ void record_pkey_malloc(void *ptr, long size, int prot); #define PKEY_MASK (PKEY_DISABLE_ACCESS | PKEY_DISABLE_WRITE) #endif -/* - * FIXME: Remove once the generic PKEY_UNRESTRICTED definition is merged. - */ -#ifndef PKEY_UNRESTRICTED -#define PKEY_UNRESTRICTED 0x0 -#endif - #ifndef set_pkey_bits static inline u64 set_pkey_bits(u64 reg, int pkey, u64 flags) { From 21d4bd35e6a0c80ba224db462b28b3f88ac2a54b Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Fri, 21 Aug 2026 10:16:06 +0800 Subject: [PATCH 0217/1012] mm/mglru: preserve inactive placement when enabling MGLRU When the LRU is switched to MGLRU (echo y > /sys/kernel/mm/lru_gen/ enabled), fill_evictable() re-inserts every folio via lru_gen_add_folio(..., false). With reclaiming hardcoded to false, an inactive anonymous folio (no PG_active, not in the swapcache) takes the "gen = MIN_NR_GENS" branch in lru_gen_folio_seq() and is seeded at seq = max_seq - 1, which lru_gen_is_active() treats as active. Its inactive placement is lost and NR_INACTIVE_ANON is folded into NR_ACTIVE_ANON. Pass reclaiming=!active so a folio from an inactive list is seeded into an older generation. Folios from the active list carry PG_active and hit the first branch either way, so they are unchanged. Reclaiming also selects the insertion end in lru_gen_add_folio(): list_add_tail() for inactive folios, list_add() for active ones. Both the legacy LRU and a MGLRU generation keep the hottest folios at the head and the coldest at the tail, and reclaim takes from the tail. To preserve that order the folio must be taken from the end matching the insertion end, so take inactive folios from the head and active folios from the tail; otherwise hot/cold would be reversed within the generation. Tested on x86_64, next-20260812, 2G VM + 1G swap, ~1.5G anon pushed onto the inactive list before enabling MGLRU: Active(anon) Inactive(anon) before switch (legacy) 2952 1548792 kB after `echo y`, unpatched 1552052 0 kB after `echo y`, patched 15144 1536636 kB Inactive file folios stay inactive either way (NR_INACTIVE_FILE is preserved). Link: https://lore.kernel.org/20260821021606.877330-1-ridong.chen@linux.dev Fixes: 354ed5974429 ("mm: multi-gen LRU: kill switch") Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Suggested-by: Barry Song Acked-by: Barry Song Assisted-by: Claude:claude-opus-4-8 Cc: Axel Rasmussen Cc: David Hildenbrand Cc: Jan Alexander Steffens (heftig) Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oleksandr Natalenko Cc: Shakeel Butt Cc: Steven Barrett Cc: Suleiman Souhlal Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao --- mm/vmscan.c | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 6dff207ad8c612..fdd13299a04a93 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -5307,7 +5307,17 @@ static bool fill_evictable(struct lruvec *lruvec) while (!list_empty(head)) { bool success; - struct folio *folio = lru_to_folio(head); + struct folio *folio; + + /* + * lru_gen_add_folio() uses list_add_tail() rather + * than list_add() when reclaiming is true. Match + * its ordering to avoid cold/hot inversion. + */ + if (active) + folio = lru_to_folio(head); + else + folio = list_first_entry(head, struct folio, lru); VM_WARN_ON_ONCE_FOLIO(folio_test_unevictable(folio), folio); VM_WARN_ON_ONCE_FOLIO(folio_test_active(folio) != active, folio); @@ -5315,7 +5325,11 @@ static bool fill_evictable(struct lruvec *lruvec) VM_WARN_ON_ONCE_FOLIO(folio_lru_gen(folio) != -1, folio); lruvec_del_folio(lruvec, folio); - success = lru_gen_add_folio(lruvec, folio, false); + /* + * Borrow reclaiming=true to place inactive folios in + * the older gens. + */ + success = lru_gen_add_folio(lruvec, folio, !active); VM_WARN_ON_ONCE(!success); if (!--remaining) From 4a82bcc4519df98580fb8f7eb3fffcfe0c9f2dbe Mon Sep 17 00:00:00 2001 From: Chengfeng Ye Date: Mon, 24 Aug 2026 19:24:33 +0800 Subject: [PATCH 0218/1012] mm/ksm: mark migration stores with WRITE_ONCE() ksm_get_folio() deliberately samples stable_node->kpfn and folio->mapping without taking the folio lock because the KSM folio may be migrated concurrently. folio_migrate_ksm() updates the same state using plain assignments. The reader can load the old kpfn, then the migrator can store the new kpfn, execute smp_wmb(), and clear the old folio's mapping before the reader checks that mapping. Thus the initial kpfn load can overlap its update and the subsequent mapping load can overlap the clear, with no common lock. This leaves marked READ_ONCE() accesses racing with plain stores. The kernel reported: BUG: KCSAN: data-race in folio_migrate_ksm / ksm_get_folio read (marked) to 0xffff8ce401421330 of 8 bytes by task 48 on cpu 3: ksm_get_folio+0x7f/0x2a0 ksm_scan_thread+0x1635/0x3330 kthread+0x1af/0x1f0 write to 0xffff8ce401421330 of 8 bytes by task 102 on cpu 1: folio_migrate_ksm+0x6a/0xd0 folio_migrate_flags+0x193/0x420 __migrate_folio.isra.0+0x162/0x1a0 migrate_folio+0x4c/0x70 move_to_new_folio+0xd6/0x170 Use WRITE_ONCE() for both stores to pair them with the existing lockless reads. This preserves the existing smp_wmb()/smp_rmb() migration protocol and control flow while preventing compiler transformations of the shared accesses. Link: https://lore.kernel.org/20260824112433.191301-1-nicoyip.dev@gmail.com Fixes: c8d6553b9580 ("ksm: make KSM page migration possible") Signed-off-by: Chengfeng Ye Signed-off-by: Andrew Morton Acked-by: Xu Xin Acked-by: David Hildenbrand (Arm) Cc: Hugh Dickins Cc: --- mm/ksm.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/mm/ksm.c b/mm/ksm.c index 49d48d1e099801..fe42c17490b89c 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -1115,7 +1115,8 @@ static inline void folio_set_stable_node(struct folio *folio, struct ksm_stable_node *stable_node) { VM_WARN_ON_FOLIO(folio_test_anon(folio) && PageAnonExclusive(&folio->page), folio); - folio->mapping = (void *)((unsigned long)stable_node | FOLIO_MAPPING_KSM); + WRITE_ONCE(folio->mapping, + (void *)((unsigned long)stable_node | FOLIO_MAPPING_KSM)); } #ifdef CONFIG_SYSFS @@ -3316,7 +3317,7 @@ void folio_migrate_ksm(struct folio *newfolio, struct folio *folio) stable_node = folio_stable_node(folio); if (stable_node) { VM_BUG_ON_FOLIO(stable_node->kpfn != folio_pfn(folio), folio); - stable_node->kpfn = folio_pfn(newfolio); + WRITE_ONCE(stable_node->kpfn, folio_pfn(newfolio)); /* * newfolio->mapping was set in advance; now we need smp_wmb() * to make sure that the new stable_node->kpfn is visible From e85c1f16f4213b56c747a34937413823a5a20ed3 Mon Sep 17 00:00:00 2001 From: Kaitao Cheng Date: Mon, 24 Aug 2026 23:16:55 +0800 Subject: [PATCH 0219/1012] mm/hugetlb: use hugetlb_vmemmap_optimizable() in boolean contexts The two boot-time sites in hugetlb_hstate_alloc_pages_onenode() and hugetlb_pages_alloc_boot_node() only need to know whether HVO is applicable, not the exact optimizable size. Switch them from hugetlb_vmemmap_optimizable_size() to hugetlb_vmemmap_optimizable() to make intent explicit. No functional change intended. Link: https://lore.kernel.org/20260824151655.30840-1-kaitao.cheng@linux.dev Signed-off-by: Kaitao Cheng Signed-off-by: Andrew Morton Reviewed-by: Joshua Hahn Reviewed-by: Muchun Song Cc: David Hildenbrand Cc: Oscar Salvador --- mm/hugetlb.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 3af46101f1bbce..d73108393e4b76 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3438,7 +3438,7 @@ static void __init hugetlb_hstate_alloc_pages_onenode(struct hstate *h, int nid) folio = only_alloc_fresh_hugetlb_folio(h, gfp_mask, nid, &node_states[N_MEMORY], NULL); if (!folio && !list_empty(&folio_list) && - hugetlb_vmemmap_optimizable_size(h)) { + hugetlb_vmemmap_optimizable(h)) { prep_and_add_allocated_folios(h, &folio_list); INIT_LIST_HEAD(&folio_list); folio = only_alloc_fresh_hugetlb_folio(h, gfp_mask, nid, @@ -3507,7 +3507,7 @@ static void __init hugetlb_pages_alloc_boot_node(unsigned long start, unsigned l for (i = 0; i < num; ++i) { struct folio *folio; - if (hugetlb_vmemmap_optimizable_size(h) && + if (hugetlb_vmemmap_optimizable(h) && (si_mem_available() == 0) && !list_empty(&folio_list)) { prep_and_add_allocated_folios(h, &folio_list); INIT_LIST_HEAD(&folio_list); From cf0b6868c8f56e18ac494e121a225b2f0d3f2cac Mon Sep 17 00:00:00 2001 From: Jinjiang Tu Date: Mon, 24 Aug 2026 14:10:09 +0800 Subject: [PATCH 0220/1012] docs: ksm: fix typos in sysfs knob names Patch series "docs/ksm: fix advisor documentation and comment", v3. This series fixes two problems left in the KSM advisor documentation and code comment: - Patch 1 fixes two typos in sysfs knob names ("adivsor_max_cpu" and "adivsor_max_pages_to_scan") in ksm.rst that don't match the actual knob names. - Patch 2 fixes the description of advisor_min_pages_to_scan: it is described as the lower limit of pages_to_scan, but is actually only used as it's initial value, and the runtime pages_to_scan can drop below advisor_min_pages_to_scan. This patch (of 2): The sysfs knob names in mm/ksm.c are "advisor_max_cpu" and "advisor_max_pages_to_scan", but the ksm.rst documentation spelled both as "adivsor_*", fix the two typos. Link: https://lore.kernel.org/20260824061010.3343959-1-tujinjiang@huawei.com Link: https://lore.kernel.org/20260824061010.3343959-2-tujinjiang@huawei.com Signed-off-by: Jinjiang Tu Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Randy Dunlap Cc: Chengming Zhou Cc: Jonathan Corbet Cc: Kefeng Wang Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Nanyong Sun Cc: Stefan Roesch Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: xu xin --- Documentation/admin-guide/mm/ksm.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Documentation/admin-guide/mm/ksm.rst b/Documentation/admin-guide/mm/ksm.rst index ad8e7a41f3b5d2..c9f533b10f6f9d 100644 --- a/Documentation/admin-guide/mm/ksm.rst +++ b/Documentation/admin-guide/mm/ksm.rst @@ -174,7 +174,7 @@ advisor_mode The section about ``advisor`` explains in detail how the scan time advisor works. -adivsor_max_cpu +advisor_max_cpu specifies the upper limit of the cpu percent usage of the ksmd background thread. The default is 70. @@ -186,7 +186,7 @@ advisor_min_pages_to_scan specifies the lower limit of the ``pages_to_scan`` parameter of the scan time advisor. The default is 500. -adivsor_max_pages_to_scan +advisor_max_pages_to_scan specifies the upper limit of the ``pages_to_scan`` parameter of the scan time advisor. The default is 30000. From 8a2df6e3abbdd1175517a419285010ad923717da Mon Sep 17 00:00:00 2001 From: Jinjiang Tu Date: Mon, 24 Aug 2026 14:10:10 +0800 Subject: [PATCH 0221/1012] mm/ksm: fix advisor_min_pages_to_scan description Both Documentation/admin-guide/mm/ksm.rst and the comment next to the variable definition in mm/ksm.c describe advisor_min_pages_to_scan as a lower limit of the pages_to_scan parameter, but that is not how the scan-time advisor actually uses it. commit 4e5fa4f5eff6 ("mm/ksm: add ksm advisor") only uses it to initialize ksm_thread_pages_to_scan when the scan-time advisor is enabled. ksm_thread_pages_to_scan is adjusted by scan_time_advisor() after a full scan finishes. ksm_thread_pages_to_scan could be increased or decreased depend on the real scan time is longer or shorter than the target scan time. The min value of ksm_thread_pages_to_scan is only limited by KSM_ADVISOR_MIN_CPU, so ksm_thread_pages_to_scan could be smaller than ksm_advisor_min_pages_to_scan. The semantics of advisor_min_pages_to_scan was updated in the v2 patchset [1], but the documentation wasn't updated. Update the documentation and comment to match the semantics of advisor_min_pages_to_scan. Link: https://lore.kernel.org/linux-mm/20231028000945.2428830-2-shr@devkernel.io/ [1] Link: https://lore.kernel.org/20260824061010.3343959-3-tujinjiang@huawei.com Signed-off-by: Jinjiang Tu Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Chengming Zhou Cc: Jonathan Corbet Cc: Kefeng Wang Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Nanyong Sun Cc: Stefan Roesch Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: xu xin Cc: Randy Dunlap --- Documentation/admin-guide/mm/ksm.rst | 4 ++-- mm/ksm.c | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/Documentation/admin-guide/mm/ksm.rst b/Documentation/admin-guide/mm/ksm.rst index c9f533b10f6f9d..c329ca747b8c47 100644 --- a/Documentation/admin-guide/mm/ksm.rst +++ b/Documentation/admin-guide/mm/ksm.rst @@ -183,8 +183,8 @@ advisor_target_scan_time pages. The default value is 200 seconds. advisor_min_pages_to_scan - specifies the lower limit of the ``pages_to_scan`` parameter of the - scan time advisor. The default is 500. + specifies the initial value of the ``pages_to_scan`` parameter of + the scan time advisor. The default is 500. advisor_max_pages_to_scan specifies the upper limit of the ``pages_to_scan`` parameter of the diff --git a/mm/ksm.c b/mm/ksm.c index fe42c17490b89c..624f37975e1295 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -348,7 +348,7 @@ static enum ksm_advisor_type ksm_advisor; * Only called through the sysfs control interface: */ -/* At least scan this many pages per batch. */ +/* Initial number of pages to scan per batch. */ static unsigned long ksm_advisor_min_pages_to_scan = 500; static void set_advisor_defaults(void) From ce9a404c67cbe1a3db5937ed0a2390ec642e8d58 Mon Sep 17 00:00:00 2001 From: Anshuman Date: Wed, 26 Aug 2026 11:43:00 +0530 Subject: [PATCH 0222/1012] selftests/mm: fix line buffer leak in mremap_test is_range_mapped() is_range_mapped() uses getline() to read /proc/self/maps line by line, but never frees the buffer it allocates. Every exit path (parse failure, match found, or reaching EOF) returns without calling free(line), leaking the buffer on each call. The function is called multiple times in this test, so the leak accumulates across calls. Free line before returning. Link: https://lore.kernel.org/20260826061300.14038-1-anshumantewari123@gmail.com Signed-off-by: Anshuman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: SJ Park Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/mremap_test.c | 1 + 1 file changed, 1 insertion(+) diff --git a/tools/testing/selftests/mm/mremap_test.c b/tools/testing/selftests/mm/mremap_test.c index 131d9d6db86790..779ef2d5f1b9f7 100644 --- a/tools/testing/selftests/mm/mremap_test.c +++ b/tools/testing/selftests/mm/mremap_test.c @@ -156,6 +156,7 @@ static bool is_range_mapped(FILE *maps_fp, unsigned long start, } } + free(line); return success; } From f38cd1244c93610fc90dce98d0d6e6a170c2b0f6 Mon Sep 17 00:00:00 2001 From: Jiayuan Chen Date: Thu, 27 Aug 2026 10:54:56 +0800 Subject: [PATCH 0223/1012] mm/memcontrol: fix data-race on reading jiffies_64 KCSAN reported a data-race between tick_do_update_jiffies64() updating jiffies_64 and mem_cgroup_flush_stats_ratelimited() reading it directly. Unlike jiffies, jiffies_64 is not volatile, so raw reads are plain accesses and can even be torn on 32-bit. Use get_jiffies_64() instead, and fix the same pattern in mem_cgroup_flush_foreign(). Link: https://lore.kernel.org/20260827025457.116191-1-jiayuan.chen@linux.dev Fixes: 508bed884767 ("mm: memcg: change flush_next_time to flush_last_time") Fixes: 97b27821b485 ("writeback, memcg: Implement foreign dirty flushing") Signed-off-by: Jiayuan Chen Signed-off-by: Andrew Morton Reported-by: syzbot+ced4d9a8cadb5ef3adae@syzkaller.appspotmail.com Acked-by: Muchun Song Acked-by: Johannes Weiner Acked-by: Michal Hocko Cc: Chris Li Cc: Jan Kara Cc: Jens Axboe Cc: Roman Gushchin Cc: Shakeel Butt Cc: Tejun Heo Cc: --- mm/memcontrol.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 856a7d07586ccc..d399710e9799ab 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -757,7 +757,7 @@ static void __mem_cgroup_flush_stats(struct mem_cgroup *memcg, bool force) return; if (mem_cgroup_is_root(memcg)) - WRITE_ONCE(flush_last_time, jiffies_64); + WRITE_ONCE(flush_last_time, get_jiffies_64()); css_rstat_flush(&memcg->css); } @@ -785,7 +785,7 @@ void mem_cgroup_flush_stats(struct mem_cgroup *memcg) void mem_cgroup_flush_stats_ratelimited(struct mem_cgroup *memcg) { /* Only flush if the periodic flusher is one full cycle late */ - if (time_after64(jiffies_64, READ_ONCE(flush_last_time) + 2*FLUSH_TIME)) + if (time_after64(get_jiffies_64(), READ_ONCE(flush_last_time) + 2 * FLUSH_TIME)) mem_cgroup_flush_stats(memcg); } @@ -3946,7 +3946,7 @@ void mem_cgroup_flush_foreign(struct bdi_writeback *wb) { struct mem_cgroup *memcg = mem_cgroup_from_css(wb->memcg_css); unsigned long intv = msecs_to_jiffies(dirty_expire_interval * 10); - u64 now = jiffies_64; + u64 now = get_jiffies_64(); int i; for (i = 0; i < MEMCG_CGWB_FRN_CNT; i++) { From 9d1494a65fcd6b18abe78b0ba1fe339e22372a70 Mon Sep 17 00:00:00 2001 From: Hui Zhu Date: Thu, 27 Aug 2026 15:05:46 +0800 Subject: [PATCH 0224/1012] mm/vmstat: annotate data race for per-cpu pageset fields zoneinfo_show_print() reads pcp->count, pcp->high, pcp->batch, pcp->high_min, pcp->high_max and the per-cpu stat_threshold while holding only zone->lock, which does not synchronize these fields. The writers are the page allocation and free fast paths under pcp->lock, decay_pcp_high() which updates pcp->high without any lock, pageset_update() which writes batch/high_min/high_max with WRITE_ONCE(), and refresh_zone_stat_thresholds() which writes stat_threshold locklessly. The race is benign: the values are only printed to /proc/zoneinfo, they are naturally aligned integers, and pageset_update() already documents that users of batch/high_min/high_max must cope with the fields changing asynchronously. Annotate the reads with data_race(), following commit af1c31acc853 ("mm/vmstat: annotate data race for zone->free_area[order].nr_free"). Found by KCSAN testing on an older kernel; the same race still exists on mainline. No functional change intended. BUG: KCSAN: data-race in zoneinfo_show_print+0x355/0x520 root/klinux/mm/vmstat.c:1774 race at unknown origin, with read to 0xffff8e1835410808 of 4 bytes by task 22653 on cpu 12: zoneinfo_show_print+0x355/0x520 root/klinux/mm/vmstat.c:1774 walk_zones_in_node root/klinux/mm/vmstat.c:1496 [inline] zoneinfo_show+0x41/0x70 root/klinux/mm/vmstat.c:1806 seq_read_iter+0x30c/0x970 root/klinux/fs/seq_file.c:230 proc_reg_read_iter+0x10c/0x170 root/klinux/fs/proc/inode.c:305 copy_splice_read+0x2a1/0x4e0 root/klinux/fs/splice.c:365 do_splice_read root/klinux/fs/splice.c:985 [inline] do_splice_read+0x139/0x1a0 root/klinux/fs/splice.c:959 splice_direct_to_actor+0x16b/0x540 root/klinux/fs/splice.c:1089 do_splice_direct_actor root/klinux/fs/splice.c:1207 [inline] do_splice_direct+0x10a/0x180 root/klinux/fs/splice.c:1233 do_sendfile+0x6ea/0x7e0 root/klinux/fs/read_write.c:1363 __do_sys_sendfile64 root/klinux/fs/read_write.c:1424 [inline] __se_sys_sendfile64 root/klinux/fs/read_write.c:1410 [inline] __x64_sys_sendfile64+0x117/0x130 root/klinux/fs/read_write.c:1410 x64_sys_call+0x1cc7/0x1ee0 root/klinux/./arch/x86/include/generated/asm/syscalls_64.h:41 do_syscall_x64 root/klinux/arch/x86/entry/common.c:46 [inline] do_syscall_64+0x75/0x2c0 root/klinux/arch/x86/entry/common.c:76 entry_SYSCALL_64_after_hwframe+0x76/0xe0 value changed: 0x000001b2 -> 0x000001b1 Reported by Kernel Concurrency Sanitizer on: CPU: 12 PID: 22653 Comm: syz-executor.12 Not tainted 6.6.140+ #672 Hardware name: QEMU Standard PC (i440FX + PIIX, 1996), BIOS rel-1.16.3-0-ga6ed6b701f0a-prebuilt.qemu.org 04/01/2014 Link: https://lore.kernel.org/20260827070546.1336383-1-hui.zhu@linux.dev Signed-off-by: Hui Zhu Signed-off-by: Andrew Morton Acked-by: Vlastimil Babka (SUSE) Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan --- mm/vmstat.c | 17 +++++++++++------ 1 file changed, 11 insertions(+), 6 deletions(-) diff --git a/mm/vmstat.c b/mm/vmstat.c index cb57714539fb5e..a3e809c57f295a 100644 --- a/mm/vmstat.c +++ b/mm/vmstat.c @@ -1837,6 +1837,11 @@ static void zoneinfo_show_print(struct seq_file *m, pg_data_t *pgdat, struct per_cpu_zonestat __maybe_unused *pzstats; pcp = per_cpu_ptr(zone->per_cpu_pageset, i); + /* + * Access to the per-cpu pageset fields is lockless as they + * are used only for printing purposes. Use data_race to + * avoid KCSAN warning. + */ seq_printf(m, "\n cpu: %i" "\n count: %i" @@ -1845,15 +1850,15 @@ static void zoneinfo_show_print(struct seq_file *m, pg_data_t *pgdat, "\n high_min: %i" "\n high_max: %i", i, - pcp->count, - pcp->high, - pcp->batch, - pcp->high_min, - pcp->high_max); + data_race(pcp->count), + data_race(pcp->high), + data_race(pcp->batch), + data_race(pcp->high_min), + data_race(pcp->high_max)); #ifdef CONFIG_SMP pzstats = per_cpu_ptr(zone->per_cpu_zonestats, i); seq_printf(m, "\n vm stats threshold: %d", - pzstats->stat_threshold); + data_race(pzstats->stat_threshold)); #endif } seq_printf(m, From c80a55d9dbe559fe506d4db9bf61f1ff68d16edf Mon Sep 17 00:00:00 2001 From: Hao Li Date: Thu, 27 Aug 2026 15:18:06 +0800 Subject: [PATCH 0225/1012] mm: remove unused anon_vma_trylock_write() The last user of anon_vma_trylock_write() was removed by cc22b9978509, leaving the helper unused. Remove it. Link: https://lore.kernel.org/20260827071845.17636-1-hao.li@linux.dev Signed-off-by: Hao Li Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Reviewed-by: SJ Park Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/internal.h | 5 ----- 1 file changed, 5 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 38b1165212c941..da833cafcd599d 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -295,11 +295,6 @@ static inline void anon_vma_lock_write(struct anon_vma *anon_vma) down_write(&anon_vma->root->rwsem); } -static inline int anon_vma_trylock_write(struct anon_vma *anon_vma) -{ - return down_write_trylock(&anon_vma->root->rwsem); -} - static inline void anon_vma_unlock_write(struct anon_vma *anon_vma) { up_write(&anon_vma->root->rwsem); From 0d73f9473e05500cd35397ab2efe8852d32761e1 Mon Sep 17 00:00:00 2001 From: Yue Haibing Date: Thu, 27 Aug 2026 16:27:22 +0800 Subject: [PATCH 0226/1012] mm/swap: remove unused declaration swapcache_clear() Commit c246d236b18b ("mm/shmem: never bypass the swap cache for SWP_SYNCHRONOUS_IO") removed the implementations but leave this. Link: https://lore.kernel.org/20260827082722.1809702-1-yuehaibing@huawei.com Signed-off-by: Yue Haibing Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Acked-by: Kairui Song Acked-by: Nick Huang Reviewed-by: Barry Song Reviewed-by: SJ Park Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham --- mm/swap.h | 1 - 1 file changed, 1 deletion(-) diff --git a/mm/swap.h b/mm/swap.h index 90a551a88df63b..fddba7a87500a4 100644 --- a/mm/swap.h +++ b/mm/swap.h @@ -324,7 +324,6 @@ void __swap_cache_replace_folio(struct swap_cluster_info *ci, struct folio *old, struct folio *new); void show_swap_cache_info(void); -void swapcache_clear(struct swap_info_struct *si, swp_entry_t entry, int nr); struct folio *read_swap_cache_async(struct swap_io_ctx *ctx, swp_entry_t entry, gfp_t gfp_mask, struct vm_area_struct *vma, unsigned long addr); struct folio *swap_cluster_readahead(swp_entry_t entry, gfp_t flag, From 0ffc59bf949ba38aa3a516891ccc882d09318a9d Mon Sep 17 00:00:00 2001 From: Tao Cui Date: Wed, 26 Aug 2026 10:17:53 +0800 Subject: [PATCH 0227/1012] docs: cgroup: document empty-write behavior of memory limit knobs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A maintenance script on a cluster wrote an unset variable into memory.max of a workload cgroup; the variable expanded to an empty string, the write succeeded, and the workload in the cgroup was OOM-killed. Nothing pointed back at the write, so it took quite some time to trace the OOM kills to that script. The memory controller documentation does not say what an empty write does; the cpuset controller documents its empty-value semantics. The actual behavior is that the empty string is accepted as 0. Reproduced on a k8s cluster (v1.29, cgroup v2, two-container pod, 384M limit): # LIMIT= # echo "$LIMIT" > $CG/memory.max # echo $? 0 m6demo 0/2 OOMKilled 0 Memory cgroup out of memory: Killed process 339529 (sleep) ... anon-rss:32kB State it where the interface files are introduced, alongside the existing notes on units and page rounding. Link: https://lore.kernel.org/all/aoVUlFdZYLFn_gvJ@tiehlicka/ Link: https://lore.kernel.org/20260826021753.197871-1-cui.tao@linux.dev Signed-off-by: Tao Cui Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Shakeel Butt Cc: Johannes Weiner Cc: Michal Koutný Cc: Muchun Song Cc: Roman Gushchin Cc: Tejun Heo --- Documentation/admin-guide/cgroup-v2.rst | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/Documentation/admin-guide/cgroup-v2.rst b/Documentation/admin-guide/cgroup-v2.rst index 86a2a0099178ea..8d2603751c51a1 100644 --- a/Documentation/admin-guide/cgroup-v2.rst +++ b/Documentation/admin-guide/cgroup-v2.rst @@ -1321,6 +1321,10 @@ All memory amounts are in bytes. If a value which is not aligned to PAGE_SIZE is written, the value may be rounded up to the closest PAGE_SIZE multiple when read back. +For the limit files described below, an empty or all-whitespace +write is accepted and sets the limit to 0. To disable a limit, +write "max"; to set it to zero explicitly, write "0". + memory.current A read-only single value file which exists on non-root cgroups. From c4f222147c1879615ffd39fb224df0decbd5dcfb Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Tue, 25 Aug 2026 13:20:59 +0200 Subject: [PATCH 0228/1012] selftests/mm: khugepaged: remove str_dup() usage We don't check str_dup() return value and never free it. While both things are irrelevant in practice, let's just clean it up by working on argv[0] directly and avoiding the str_dup(). Nobody after us needs these parts of the argv[0] string anyway. This patch is inspired by previous work from Anshuman Tewari [1]. Link: https://lore.kernel.org/r/20260821114416.12255-1-anshumantewari123@gmail.com [1] Link: https://lore.kernel.org/20260825-remove_str_dup-v1-1-0ba2121a820c@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Anshuman Tewari Reviewed-by: Zi Yan Reviewed-by: Lance Yang Reviewed-by: Dev Jain Acked-by: Usama Arif Reviewed-by: Baolin Wang Reviewed-by: Barry Song Reviewed-by: Andrew Morton --- tools/testing/selftests/mm/khugepaged.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 1d2d6bd72fd2af..83d27d069c4139 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1227,7 +1227,7 @@ static void parse_test_type(int argc, char **argv) return; } - buf = strdup(argv[0]); + buf = argv[0]; token = strsep(&buf, ":"); if (!strcmp(token, "all")) { From c62e0fe70294d68b58f2cc92adacbd495b8663b4 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Tue, 25 Aug 2026 20:01:53 +0800 Subject: [PATCH 0229/1012] mm/memcontrol: remove unused memcg parameter in calculate_high_delay() The memcg argument of calculate_high_delay() is never referenced in its function body. The delay calculation only depends on nr_pages and max_overage, and both callers have already obtained max_overage from the same memcg. Drop this unused parameter and update the two call sites inside __mem_cgroup_handle_over_high(). Link: https://lore.kernel.org/20260825120153.1405-1-hongfu.li@linux.dev Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Acked-by: Michal Hocko Reviewed-by: Gregory Price (Meta) Reviewed-by: SJ Park Acked-by: Shakeel Butt --- mm/memcontrol.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index d399710e9799ab..1709ac96bbdec5 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2515,8 +2515,7 @@ static u64 swap_find_max_overage(struct mem_cgroup *memcg) * Get the number of jiffies that we should penalise a mischievous cgroup which * is exceeding its memory.high by checking both it and its ancestors. */ -static unsigned long calculate_high_delay(struct mem_cgroup *memcg, - unsigned int nr_pages, +static unsigned long calculate_high_delay(unsigned int nr_pages, u64 max_overage) { unsigned long penalty_jiffies; @@ -2594,10 +2593,10 @@ void __mem_cgroup_handle_over_high(gfp_t gfp_mask) * memory.high is breached and reclaim is unable to keep up. Throttle * allocators proactively to slow down excessive growth. */ - penalty_jiffies = calculate_high_delay(memcg, nr_pages, + penalty_jiffies = calculate_high_delay(nr_pages, mem_find_max_overage(memcg)); - penalty_jiffies += calculate_high_delay(memcg, nr_pages, + penalty_jiffies += calculate_high_delay(nr_pages, swap_find_max_overage(memcg)); /* From 3bcfc24a638a6175dc3a233bb69ef0714bcbe546 Mon Sep 17 00:00:00 2001 From: Qi Xi Date: Tue, 25 Aug 2026 20:05:48 +0800 Subject: [PATCH 0230/1012] mm/page_isolation: fix UBSAN shift-out-of-bounds warning Patch series "mm/page_isolation: fix UBSAN shift-out-of-bounds in isolate_single_pageblock", v3. Patch 1 fixes a UBSAN shift-out-of-bounds warning in the PageBuddy branch of isolate_single_pageblock() triggered by concurrent buddy allocation. Patch 2 addresses the same class of issue in the PageCompound branch, where racy compound_order() and compound_head() reads could lead to out-of-range shifts or incorrect page skipping. This patch (of 2): A contig-range allocation racing with buddy allocation on the adjacent pageblock can trigger: UBSAN: shift-out-of-bounds in mm/page_isolation.c:393:15 shift exponent -749042176 is negative Call trace: isolate_single_pageblock start_isolate_page_range alloc_contig_frozen_range_noprof alloc_contig_range_noprof isolate_single_pageblock() first calls set_migratetype_isolate() with zone->lock held, which marks the pageblock MIGRATE_ISOLATE and moves any free page straddling the boundary out of the way. Once the lock is dropped, it scans the MAX_ORDER_NR_PAGES-aligned window [start_pfn, boundary_pfn) locklessly, only to skip the free pages already handled above and to detect in-use pages straddling the boundary. Since this scan only reads page state to decide how far to skip and returns -EBUSY on a straddling in-use page, it does not take the lock. The window also covers the adjacent pageblock, whose free pages stay on the normal movable/CMA freelist and can be allocated concurrently. So after the scan observes PageBuddy(page), another CPU can allocate the page, leaving a stale value in page->private that makes "1 << order" shift out of range. Use buddy_order_unsafe() with READ_ONCE to read the order, and validate it is within MAX_PAGE_ORDER before shifting to prevent UBSAN warnings. Since pageblock_isolate_and_move_free_pages() already handles free pages straddling boundary_pfn under zone->lock, bail out with -EBUSY instead of VM_WARN_ON_ONCE() when a PageBuddy page appears to cross the boundary during the lockless scan. Link: https://lore.kernel.org/20260825120549.966271-2-xiqi2@huawei.com Fixes: b2c9e2fbba32 ("mm: make alloc_contig_range work at pageblock granularity") Signed-off-by: Qi Xi Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Johannes Weiner Cc: Kefeng Wang Cc: Michal Hocko Cc: Nanyong Sun Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Brendan Jackman Cc: --- mm/page_isolation.c | 18 ++++++++++++------ 1 file changed, 12 insertions(+), 6 deletions(-) diff --git a/mm/page_isolation.c b/mm/page_isolation.c index e5dfc7bf494460..e4ee998c00f3bc 100644 --- a/mm/page_isolation.c +++ b/mm/page_isolation.c @@ -388,13 +388,19 @@ static int isolate_single_pageblock(unsigned long boundary_pfn, } if (PageBuddy(page)) { - int order = buddy_order(page); + unsigned int order = buddy_order_unsafe(page); - /* pageblock_isolate_and_move_free_pages() handled this */ - VM_WARN_ON_ONCE(pfn + (1 << order) > boundary_pfn); - - pfn += 1UL << order; - continue; + /* buddy_order_unsafe() is racy. Validate the order before shifting. */ + if (order <= MAX_PAGE_ORDER && + /* + * pageblock_isolate_and_move_free_pages() splits + * cross-boundary PageBuddy, verify it. + */ + pfn + (1UL << order) <= boundary_pfn) { + pfn += 1UL << order; + continue; + } + goto failed; } /* From ba7aa3dc7fab5840bfaba25923af0ef57d833644 Mon Sep 17 00:00:00 2001 From: Qi Xi Date: Tue, 25 Aug 2026 20:05:49 +0800 Subject: [PATCH 0231/1012] mm/page_isolation: guard compound_order() against racing The PageCompound branch reads compound_head() without holding a reference. A racing split or free can cause compound_head() to return a stale pointer, and compound_nr() reads the order from that stale head, leading to out-of-range shifts and making the skip distance meaningless. Read the order explicitly with compound_order() and validate it is within MAX_FOLIO_ORDER before shifting. Also verify the derived head_pfn against the legitimate pfn: the head must not be past pfn, must be aligned to nr_pages, and pfn must fall within the compound page. Bail out with -EBUSY if any check fails. Link: https://lore.kernel.org/20260825120549.966271-3-xiqi2@huawei.com Fixes: b2c9e2fbba32 ("mm: make alloc_contig_range work at pageblock granularity") Signed-off-by: Qi Xi Signed-off-by: Andrew Morton Suggested-by: Zi Yan Reviewed-by: Zi Yan Cc: Brendan Jackman Cc: Johannes Weiner Cc: Kefeng Wang Cc: Michal Hocko Cc: Nanyong Sun Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: --- mm/page_isolation.c | 22 ++++++++++++++++++++-- 1 file changed, 20 insertions(+), 2 deletions(-) diff --git a/mm/page_isolation.c b/mm/page_isolation.c index e4ee998c00f3bc..5aaad037384e44 100644 --- a/mm/page_isolation.c +++ b/mm/page_isolation.c @@ -419,10 +419,28 @@ static int isolate_single_pageblock(unsigned long boundary_pfn, if (PageCompound(page)) { struct page *head = compound_head(page); unsigned long head_pfn = page_to_pfn(head); - unsigned long nr_pages = compound_nr(head); + unsigned int order = compound_order(head); + unsigned long nr_pages; + + /* compound_order() is racy. Cap it at MAX_FOLIO_ORDER. */ + if (order > MAX_FOLIO_ORDER) + goto failed; + + nr_pages = 1UL << order; + + /* + * compound_head() is also racy, so the derived head_pfn + * needs additional checks to make sure it is valid. + * Otherwise, just fail the check. pfn comes from + * __first_valid_page() as a legitimate PFN, so use it to + * check head_pfn. + */ + if (head_pfn > pfn || !IS_ALIGNED(head_pfn, nr_pages) || + pfn - head_pfn >= nr_pages) + goto failed; if (head_pfn + nr_pages <= boundary_pfn || - PageHuge(page)) { + PageHuge(head)) { pfn = head_pfn + nr_pages; continue; } From 6a50937c01acb1a25ba3b7e3d12fc495cb4aef43 Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Tue, 25 Aug 2026 20:30:23 +0800 Subject: [PATCH 0232/1012] selftests/mm: fix incorrect skip output in pkey_sighandler_tests When pkeys is not supported, ksft_exit_skip() runs with ksft_plan already set, which takes the "ok N # SKIP" branch intended for skipping a single test case. The result is a TAP plan of 5 but only one result line. $ ./pkey_sighandler_tests TAP version 13 1..5 ok 1 # SKIP pkeys not supported # 1 skipped test(s) detected. Consider enabling relevant config options to improve coverage. # Planned tests != run tests (5 != 1) # Totals: pass:0 fail:0 xfail:0 xpass:0 skip:1 error:0 Move ksft_set_plan() after the skip check so ksft_exit_skip() takes the "1..0 # SKIP" branch, the correct TAP representation for skipping an entire test file. $ ./pkey_sighandler_tests TAP version 13 1..0 # SKIP pkeys not supported Link: https://lore.kernel.org/20260825123023.64418-1-zenghui.yu@linux.dev Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/pkey_sighandler_tests.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/pkey_sighandler_tests.c b/tools/testing/selftests/mm/pkey_sighandler_tests.c index c218d0510a2a42..74bf79a5399dac 100644 --- a/tools/testing/selftests/mm/pkey_sighandler_tests.c +++ b/tools/testing/selftests/mm/pkey_sighandler_tests.c @@ -543,11 +543,12 @@ static void (*pkey_tests[])(void) = { int main(int argc, char *argv[]) { ksft_print_header(); - ksft_set_plan(ARRAY_SIZE(pkey_tests)); if (!is_pkeys_supported()) ksft_exit_skip("pkeys not supported\n"); + ksft_set_plan(ARRAY_SIZE(pkey_tests)); + for (test_nr = 0; test_nr < ARRAY_SIZE(pkey_tests); test_nr++) { tracing_on(); (*pkey_tests[test_nr])(); From ed1ba92ebe80920965277015024b04b28bfa6933 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:40 +0800 Subject: [PATCH 0233/1012] mm/sparse: relax struct mem_section size constraints Patch series "mm: Introduce section-based vmemmap optimization for HugeTLB", v6. This series is split out from the earlier, larger series "mm: Generalize HVO for HugeTLB and device DAX" [1]. While the parent series generalizes vmemmap optimization across HugeTLB and device DAX, this subset addresses a single, self-contained step: making the generic sparse-vmemmap code section-based optimization aware and switching HugeTLB bootmem pages to this path. HugeTLB vmemmap optimization currently has its own early boot setup path. It pre-populates optimized vmemmap mappings before the normal sparse-vmemmap population code runs, and sparsemem carries SPARSEMEM_VMEMMAP_PREINIT only to support that special case. That makes the HugeTLB vmemmap optimization path harder to share with other users of sparse-vmemmap optimization and leaves a fair amount of HugeTLB-specific boot-time state in the generic memory initialization flow. This series introduces section-based vmemmap optimization support in the sparse-vmemmap code and switches HugeTLB bootmem pages over to it. Instead of having HugeTLB pre-populate optimized vmemmap mappings itself, HugeTLB now records the compound page order in the corresponding memory sections. The generic sparse-vmemmap population path can then allocate or reuse shared tail vmemmap pages based on section metadata. The patches are organized as follows: - patches 1-2 prepare sparsemem and vmemmap optimization metadata - patches 3-8 teach the common sparse-vmemmap paths to use that state - patches 9-10 switch HugeTLB bootmem optimization to the section-based path - patches 11-17 clean up sparsemem and HugeTLB bootmem code that is no longer needed after the conversion This is intended to be the second smaller step toward the broader HVO generalization. The device DAX conversion and the wider HVO consolidation are left for follow-up series. This patch (of 17): struct mem_section is currently forced to a power-of-2 size so the section-to-root lookup can use a mask instead of a modulo. That requirement makes future extensions harder than necessary: adding a small field can require configuration-dependent padding or layout checks just to preserve the lookup scheme. Keep the lookup correct for any struct mem_section size by using a plain modulo instead. Do not leave the layout entirely unconstrained, though. Keep struct mem_section double-word aligned so modest size changes, such as adding another word-sized field on 64-bit systems, still keep a compact and efficient layout. If future fields grow the structure beyond that sweet spot, the lookup remains correct; only the exact layout efficiency changes. Link: https://lore.kernel.org/20260910063256.64386-2-songmuchun@bytedance.com Link: https://lore.kernel.org/all/20260513130542.35604-1-songmuchun@bytedance.com/ [1] Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Qi Zheng --- include/linux/mmzone.h | 11 +++-------- mm/sparse.c | 2 -- scripts/gdb/linux/mm.py | 6 ++---- 3 files changed, 5 insertions(+), 14 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 94f9c3ff541604..0a2428714108c4 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2027,13 +2027,9 @@ struct mem_section { * section. (see page_ext.h about this.) */ struct page_ext *page_ext; - unsigned long pad; #endif - /* - * WARNING: mem_section must be a power-of-2 in size for the - * calculation and use of SECTION_ROOT_MASK to make sense. - */ -}; +/* Sacrifice minor padding space for efficient lookup. */ +} __aligned(2 * sizeof(unsigned long)); #ifdef CONFIG_SPARSEMEM_EXTREME #define SECTIONS_PER_ROOT (PAGE_SIZE / sizeof (struct mem_section)) @@ -2043,7 +2039,6 @@ struct mem_section { #define SECTION_NR_TO_ROOT(sec) ((sec) / SECTIONS_PER_ROOT) #define NR_SECTION_ROOTS DIV_ROUND_UP(NR_MEM_SECTIONS, SECTIONS_PER_ROOT) -#define SECTION_ROOT_MASK (SECTIONS_PER_ROOT - 1) #ifdef CONFIG_SPARSEMEM_EXTREME extern struct mem_section **mem_section; @@ -2067,7 +2062,7 @@ static inline struct mem_section *__nr_to_section(unsigned long nr) if (!mem_section || !mem_section[root]) return NULL; #endif - return &mem_section[root][nr & SECTION_ROOT_MASK]; + return &mem_section[root][nr % SECTIONS_PER_ROOT]; } /* diff --git a/mm/sparse.c b/mm/sparse.c index 7c15406e77f5d2..c84b4c7b8c7069 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -322,8 +322,6 @@ void __init sparse_init(void) unsigned long pnum_end, pnum_begin, map_count = 1; int nid_begin; - /* see include/linux/mmzone.h 'struct mem_section' definition */ - BUILD_BUG_ON(!is_power_of_2(sizeof(struct mem_section))); memblocks_present(); if (compound_info_has_mask()) { diff --git a/scripts/gdb/linux/mm.py b/scripts/gdb/linux/mm.py index 28d33624c38bd6..193a88d763abf7 100644 --- a/scripts/gdb/linux/mm.py +++ b/scripts/gdb/linux/mm.py @@ -70,7 +70,6 @@ def __init__(self): self.SECTIONS_PER_ROOT = 1 self.NR_SECTION_ROOTS = DIV_ROUND_UP(self.NR_MEM_SECTIONS, self.SECTIONS_PER_ROOT) - self.SECTION_ROOT_MASK = self.SECTIONS_PER_ROOT - 1 try: self.SECTION_HAS_MEM_MAP = 1 << int(gdb.parse_and_eval('SECTION_HAS_MEM_MAP_BIT')) @@ -100,7 +99,7 @@ def SECTION_NR_TO_ROOT(self, sec): def __nr_to_section(self, nr): root = self.SECTION_NR_TO_ROOT(nr) mem_section = gdb.parse_and_eval("mem_section") - return mem_section[root][nr & self.SECTION_ROOT_MASK] + return mem_section[root][nr % self.SECTIONS_PER_ROOT] def pfn_to_section_nr(self, pfn): return pfn >> self.PFN_SECTION_SHIFT @@ -249,7 +248,6 @@ def __init__(self): self.SECTIONS_PER_ROOT = 1 self.NR_SECTION_ROOTS = DIV_ROUND_UP(self.NR_MEM_SECTIONS, self.SECTIONS_PER_ROOT) - self.SECTION_ROOT_MASK = self.SECTIONS_PER_ROOT - 1 self.SUBSECTION_SHIFT = 21 self.SEBSECTION_SIZE = 1 << self.SUBSECTION_SHIFT self.PFN_SUBSECTION_SHIFT = self.SUBSECTION_SHIFT - self.PAGE_SHIFT @@ -304,7 +302,7 @@ def SECTION_NR_TO_ROOT(self, sec): def __nr_to_section(self, nr): root = self.SECTION_NR_TO_ROOT(nr) mem_section = gdb.parse_and_eval("mem_section") - return mem_section[root][nr & self.SECTION_ROOT_MASK] + return mem_section[root][nr % self.SECTIONS_PER_ROOT] def pfn_to_section_nr(self, pfn): return pfn >> self.PFN_SECTION_SHIFT From e294be8fca6a166c4217d1f452184a05841ce91b Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:41 +0800 Subject: [PATCH 0234/1012] mm/sparse-vmemmap: rename HVO order macros The macros VMEMMAP_TAIL_MIN_ORDER and NR_VMEMMAP_TAILS describe the order range where HVO can be applied, but their names tie that range to the tail-page cache implementation. Rename them with a VMEMMAP_OPTIMIZATION prefix and use the new names in the HVO paths. This makes the code describe the optimization requirements rather than the tail-page cache implementation detail. No functional change intended. Link: https://lore.kernel.org/20260910063256.64386-3-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mmzone.h | 17 +++++++++-------- mm/hugetlb.c | 4 ++-- mm/hugetlb_vmemmap.c | 2 +- mm/sparse-vmemmap.c | 4 ++-- 4 files changed, 14 insertions(+), 13 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 0a2428714108c4..5fb9b37819d550 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -107,13 +107,14 @@ is_power_of_2(sizeof(struct page)) ? \ MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) -/* - * vmemmap optimization (like HVO) is only possible for page orders that fill - * two or more pages with struct pages. - */ -#define VMEMMAP_TAIL_MIN_ORDER (ilog2(2 * PAGE_SIZE / sizeof(struct page))) -#define __NR_VMEMMAP_TAILS (MAX_FOLIO_ORDER - VMEMMAP_TAIL_MIN_ORDER + 1) -#define NR_VMEMMAP_TAILS (__NR_VMEMMAP_TAILS > 0 ? __NR_VMEMMAP_TAILS : 0) +/* The number of struct pages covered by the retained vmemmap pages with HVO enabled. */ +#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES (PAGE_SIZE / sizeof(struct page)) +#define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) + +#define __VMEMMAP_OPTIMIZATION_NR_ORDERS \ + (MAX_FOLIO_ORDER - VMEMMAP_OPTIMIZATION_MIN_ORDER + 1) +#define VMEMMAP_OPTIMIZATION_NR_ORDERS \ + (__VMEMMAP_OPTIMIZATION_NR_ORDERS > 0 ? __VMEMMAP_OPTIMIZATION_NR_ORDERS : 0) enum migratetype { MIGRATE_UNMOVABLE, @@ -1158,7 +1159,7 @@ struct zone { atomic_long_t vm_stat[NR_VM_ZONE_STAT_ITEMS]; atomic_long_t vm_numa_event[NR_VM_NUMA_EVENT_ITEMS]; #ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP - struct page *vmemmap_tails[NR_VMEMMAP_TAILS]; + struct page *vmemmap_tails[VMEMMAP_OPTIMIZATION_NR_ORDERS]; #endif } ____cacheline_internodealigned_in_smp; diff --git a/mm/hugetlb.c b/mm/hugetlb.c index d73108393e4b76..f1c92e80c34672 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3368,7 +3368,7 @@ void __init hugetlb_bootmem_struct_page_init(void) struct zone *zone; for_each_zone(zone) { - for (int i = 0; i < NR_VMEMMAP_TAILS; i++) { + for (int i = 0; i < VMEMMAP_OPTIMIZATION_NR_ORDERS; i++) { struct page *tail, *p; unsigned int order; @@ -3376,7 +3376,7 @@ void __init hugetlb_bootmem_struct_page_init(void) if (!tail) continue; - order = i + VMEMMAP_TAIL_MIN_ORDER; + order = i + VMEMMAP_OPTIMIZATION_MIN_ORDER; p = page_to_virt(tail); /* * prep_and_add_bootmem_folios() can access pageblock diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index 917db0984143c2..ae8fdaa4211891 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -494,7 +494,7 @@ static bool vmemmap_should_optimize_folio(const struct hstate *h, struct folio * static struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) { - const unsigned int idx = order - VMEMMAP_TAIL_MIN_ORDER; + const unsigned int idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; struct page *tail, *p; int node = zone_to_nid(zone); diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 5a2469fb1838c7..aa6a4a2fae9886 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -329,12 +329,12 @@ static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone * unsigned int idx; int node = zone_to_nid(zone); - if (WARN_ON_ONCE(order < VMEMMAP_TAIL_MIN_ORDER)) + if (WARN_ON_ONCE(order < VMEMMAP_OPTIMIZATION_MIN_ORDER)) return NULL; if (WARN_ON_ONCE(order > MAX_FOLIO_ORDER)) return NULL; - idx = order - VMEMMAP_TAIL_MIN_ORDER; + idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; tail = zone->vmemmap_tails[idx]; if (tail) return tail; From ae392b37043d81fbf35889e4ab4969c59b7391f5 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:42 +0800 Subject: [PATCH 0235/1012] mm/mm_init: skip initializing shared vmemmap tail pages memmap_init_range() initializes every struct page in the target range. For compound pages with vmemmap optimization, the tail struct pages are backed by a shared vmemmap page. Initializing those tail struct pages would overwrite the shared vmemmap page contents, requiring users such as HugeTLB to restore the metadata afterwards. Track the compound page order for HVO-backed sections and use that metadata to detect struct pages that fall into the shared tail vmemmap range. Skip those shared tail pages in memmap_init_range(), then initialize pageblock migratetypes for the processed range with a helper after the per-page initialization loop. Keep direct mem_section access inside sparse helpers. Expose pfn_to_section_compound_order() for callers that only need the order associated with a PFN. This lets memmap_init_range() skip shared tail vmemmap pages without exposing __pfn_to_section() to !SPARSEMEM builds. This is a preparatory change for consolidating handling across users of vmemmap optimization, and it also avoids redundant initialization of shared tail vmemmap pages during early boot. That early-boot benefit appears only once HugeTLB is switched to this common handling, since HugeTLB is the early-boot user that creates those shared tail vmemmap pages. Link: https://lore.kernel.org/20260910063256.64386-4-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mmzone.h | 8 ++++++++ mm/mm_init.c | 40 +++++++++++++++++++++++----------------- mm/sparse.h | 33 +++++++++++++++++++++++++++++++++ 3 files changed, 64 insertions(+), 17 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 5fb9b37819d550..738bf3d272563b 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2022,6 +2022,14 @@ struct mem_section { unsigned long section_mem_map; struct mem_section_usage *usage; +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP + /* + * Normally, sections hold regular (order-0) pages. However, for + * sections with HVO enabled, this tracks the compound page order + * to enable deduplication of redundant vmemmap pages. + */ + unsigned int compound_page_order; +#endif #ifdef CONFIG_PAGE_EXTENSION /* * If SPARSEMEM, pgdat doesn't have page_ext pointer. We use diff --git a/mm/mm_init.c b/mm/mm_init.c index 1533aebafb6885..6655fe696e0f01 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -29,6 +29,7 @@ #include #include #include +#include #include #include #include @@ -677,21 +678,19 @@ static inline void fixup_hashdist(void) static inline void fixup_hashdist(void) {} #endif /* CONFIG_NUMA */ -#if defined(CONFIG_ZONE_DEVICE) || defined(CONFIG_DEFERRED_STRUCT_PAGE_INIT) static __meminit void pageblock_migratetype_init_range(unsigned long pfn, - unsigned long nr_pages, int migratetype, bool atomic) + unsigned long nr_pages, int migratetype, bool isolate, bool atomic) { const unsigned long end = pfn + nr_pages; for (pfn = pageblock_align(pfn); pfn < end; pfn += pageblock_nr_pages) { enum migratetype mt = kho_scratch_migratetype(pfn, migratetype); - init_pageblock_migratetype(pfn_to_page(pfn), mt, false); - if (!atomic && IS_ALIGNED(pfn, PAGES_PER_SECTION)) + init_pageblock_migratetype(pfn_to_page(pfn), mt, isolate); + if (!atomic && IS_ALIGNED(pfn, PFN_DOWN(SZ_1G))) cond_resched(); } } -#endif #ifdef CONFIG_DEFERRED_STRUCT_PAGE_INIT static inline void pgdat_set_deferred_range(pg_data_t *pgdat) @@ -886,6 +885,17 @@ void __meminit memmap_init_range(unsigned long size, int nid, unsigned long zone } } + /* + * Vmemmap-optimizable PFNs are backed by shared tail struct pages, + * which have already been initialized during vmemmap population. + */ + if (vmemmap_optimizable_pfn(pfn)) { + const unsigned int order = pfn_to_section_compound_order(pfn); + + pfn = min(ALIGN(pfn, 1UL << order), end_pfn); + continue; + } + page = pfn_to_page(pfn); __init_single_page(page, pfn, zone, nid); if (context == MEMINIT_HOTPLUG) { @@ -897,19 +907,13 @@ void __meminit memmap_init_range(unsigned long size, int nid, unsigned long zone __SetPageOffline(page); } - /* - * Usually, we want to mark the pageblock MIGRATE_MOVABLE, - * such that unmovable allocations won't be scattered all - * over the place during system boot. - */ - if (pageblock_aligned(pfn)) { - enum migratetype mt = kho_scratch_migratetype(pfn, migratetype); - - init_pageblock_migratetype(page, mt, isolate_pageblock); + if (pageblock_aligned(pfn)) cond_resched(); - } pfn++; } + + pageblock_migratetype_init_range(start_pfn, pfn - start_pfn, migratetype, + isolate_pageblock, /* atomic */ false); } static void __init memmap_init_zone_range(struct zone *zone, @@ -1112,7 +1116,8 @@ void __ref memmap_init_zone_device(struct zone *zone, compound_nr_pages(pfn, altmap, pgmap)); } - pageblock_migratetype_init_range(start_pfn, nr_pages, MIGRATE_MOVABLE, false); + pageblock_migratetype_init_range(start_pfn, nr_pages, MIGRATE_MOVABLE, + /* isolate */ false, /* atomic */ false); pr_debug("%s initialised %lu pages in %ums\n", __func__, nr_pages, jiffies_to_msecs(jiffies - start)); @@ -1921,7 +1926,8 @@ static void __init deferred_free_pages(unsigned long pfn, if (!nr_pages) return; - pageblock_migratetype_init_range(pfn, nr_pages, MIGRATE_MOVABLE, true); + pageblock_migratetype_init_range(pfn, nr_pages, MIGRATE_MOVABLE, + /* isolate */ false, /* atomic */ true); page = pfn_to_page(pfn); diff --git a/mm/sparse.h b/mm/sparse.h index 3b744667a7e624..da79c83adeae11 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -10,6 +10,39 @@ #include +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +static inline unsigned int section_compound_order(const struct mem_section *section) +{ + return section->compound_page_order; +} + +static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) +{ + return section_compound_order(__pfn_to_section(pfn)); +} +#else +static inline unsigned int section_compound_order(const struct mem_section *section) +{ + return 0; +} + +static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) +{ + return 0; +} +#endif + +static inline bool vmemmap_optimizable_pfn(unsigned long pfn) +{ + const unsigned int order = pfn_to_section_compound_order(pfn); + const unsigned long nr_pages = 1UL << order; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES; +} + /* * mm/sparse.c */ From c38c3b74e3d0fc0d463c06566f4e52cb476cd3cc Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:43 +0800 Subject: [PATCH 0236/1012] mm/sparse-vmemmap: initialize shared tail vmemmap pages on allocation The shared tail vmemmap page allocated in vmemmap_get_tail() used to be left uninitialized, because memmap_init_range() would later overwrite it. That forced users such as HugeTLB to defer the initialization to their own setup paths. Now that memmap_init_range() skips shared tail vmemmap pages, initialize them immediately in vmemmap_get_tail() with init_compound_tail() instead. This adds initialization at the point where the shared tail page is allocated. The existing HugeTLB initialization remains necessary until the compound page order is set and memmap_init_range() starts skipping shared tails. It is removed when HugeTLB switches to the section-based mechanism. Link: https://lore.kernel.org/20260910063256.64386-5-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/sparse-vmemmap.c | 12 ++---------- 1 file changed, 2 insertions(+), 10 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index aa6a4a2fae9886..107215cf8488ce 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -338,19 +338,11 @@ static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone * tail = zone->vmemmap_tails[idx]; if (tail) return tail; - - /* - * Only allocate the page, but do not initialize it. - * - * Any initialization done here will be overwritten by memmap_init(). - * - * hugetlb_bootmem_struct_page_init() will take care of initialization - * after memmap_init(). - */ - p = vmemmap_alloc_block_zero(PAGE_SIZE, node); if (!p) return NULL; + for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) + init_compound_tail(p + i, NULL, order, zone); tail = virt_to_page(p); zone->vmemmap_tails[idx] = tail; From 85323775a2be98a73d298562391bad1737077a09 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:44 +0800 Subject: [PATCH 0237/1012] mm/sparse-vmemmap: support section-based vmemmap accounting section_nr_vmemmap_pages() can account ordinary sections and DAX sections, but section-based vmemmap optimization stores the compound page order in struct mem_section and retains a different number of vmemmap pages. Teach section_nr_vmemmap_pages() to recognize section-based optimized sections and calculate their vmemmap page count from the stored compound page order and the HVO retained page count. Link: https://lore.kernel.org/20260910063256.64386-6-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mmzone.h | 6 ++++-- mm/sparse-vmemmap.c | 10 ++++++---- mm/sparse.h | 16 ++++++++++++++++ 3 files changed, 26 insertions(+), 6 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 738bf3d272563b..2c4f63e379da7f 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -107,8 +107,10 @@ is_power_of_2(sizeof(struct page)) ? \ MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) -/* The number of struct pages covered by the retained vmemmap pages with HVO enabled. */ -#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES (PAGE_SIZE / sizeof(struct page)) +/* The number of retained vmemmap pages with HVO enabled. */ +#define VMEMMAP_OPTIMIZATION_PAGES 1 +#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES \ + (VMEMMAP_OPTIMIZATION_PAGES * PAGE_SIZE / sizeof(struct page)) #define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) #define __VMEMMAP_OPTIMIZATION_NR_ORDERS \ diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 107215cf8488ce..dea7179fbc90b0 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -649,24 +649,26 @@ void offline_mem_sections(unsigned long start_pfn, unsigned long end_pfn) static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { - const unsigned int order = pgmap ? pgmap->vmemmap_shift : 0; + const struct mem_section *ms = __pfn_to_section(pfn); + const int order = pgmap ? pgmap->vmemmap_shift : section_compound_order(ms); + const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; const unsigned long pages_per_compound = 1UL << order; VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); VM_WARN_ON_ONCE(nr_pages > PAGES_PER_SECTION); - if (!vmemmap_can_optimize(altmap, pgmap)) + if (!vmemmap_can_optimize(altmap, pgmap) && !section_vmemmap_optimizable(ms)) return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); if (order < PFN_SECTION_SHIFT) { VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, pages_per_compound)); - return VMEMMAP_RESERVE_NR * nr_pages / pages_per_compound; + return vmemmap_pages * nr_pages / pages_per_compound; } VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)); if (IS_ALIGNED(pfn, pages_per_compound)) - return VMEMMAP_RESERVE_NR; + return vmemmap_pages; return 0; } diff --git a/mm/sparse.h b/mm/sparse.h index da79c83adeae11..563fc1f4d71778 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -43,6 +43,17 @@ static inline bool vmemmap_optimizable_pfn(unsigned long pfn) return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES; } +static inline bool vmemmap_optimizable_order(unsigned int order) +{ + if (!IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP)) + return false; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return order >= VMEMMAP_OPTIMIZATION_MIN_ORDER; +} + /* * mm/sparse.c */ @@ -86,6 +97,11 @@ static inline size_t mem_section_usage_size(void) return struct_size_t(struct mem_section_usage, pageblock_flags, BITS_TO_LONGS(SECTION_BLOCKFLAGS_BITS)); } + +static inline bool section_vmemmap_optimizable(const struct mem_section *ms) +{ + return vmemmap_optimizable_order(section_compound_order(ms)); +} #else static inline void sparse_init(void) {} #endif /* CONFIG_SPARSEMEM */ From 83256d0585348332f24cffec9d5ff443b47684d8 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:45 +0800 Subject: [PATCH 0238/1012] mm/mm_init: factor out pfn_to_zone() pfn_to_zone() in hugetlb_vmemmap.c duplicates the zone lookup logic in __init_deferred_page(). Move it to mm_init.c, declare it in mm/mm_init.h, and reuse it from __init_deferred_page() and HugeTLB early vmemmap initialization instead of open-coding the zone walk there. Link: https://lore.kernel.org/20260910063256.64386-7-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/hugetlb_vmemmap.c | 17 ++--------------- mm/mm_init.c | 28 ++++++++++++++++++---------- mm/mm_init.h | 1 + 3 files changed, 21 insertions(+), 25 deletions(-) diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index ae8fdaa4211891..c48fcea076a522 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -19,6 +19,7 @@ #include #include "hugetlb_vmemmap.h" #include "internal.h" +#include "mm_init.h" /** * struct vmemmap_remap_walk - walk vmemmap page table @@ -744,20 +745,6 @@ static bool vmemmap_should_optimize_bootmem_page(struct huge_bootmem_page *m) return true; } -static struct zone *pfn_to_zone(unsigned nid, unsigned long pfn) -{ - struct zone *zone; - enum zone_type zone_type; - - for (zone_type = 0; zone_type < MAX_NR_ZONES; zone_type++) { - zone = &NODE_DATA(nid)->node_zones[zone_type]; - if (zone_spans_pfn(zone, pfn)) - return zone; - } - - return NULL; -} - /* * Initialize memmap section for a gigantic page, HVO-style. */ @@ -787,7 +774,7 @@ void __init hugetlb_vmemmap_init_early(int nid) map = pfn_to_page(pfn); start = (unsigned long)map; end = start + hugetlb_vmemmap_size(m->hstate); - zone = pfn_to_zone(nid, pfn); + zone = pfn_to_zone(pfn, nid); if (vmemmap_populate_hvo(start, end, huge_page_order(m->hstate), zone, HUGETLB_VMEMMAP_RESERVE_SIZE)) diff --git a/mm/mm_init.c b/mm/mm_init.c index 6655fe696e0f01..ee3bd792176890 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -692,6 +692,20 @@ static __meminit void pageblock_migratetype_init_range(unsigned long pfn, } } +struct zone __meminit *pfn_to_zone(unsigned long pfn, int nid) +{ + pg_data_t *pgdat = NODE_DATA(nid); + + for (enum zone_type zone_type = 0; zone_type < MAX_NR_ZONES; zone_type++) { + struct zone *zone = &pgdat->node_zones[zone_type]; + + if (zone_spans_pfn(zone, pfn)) + return zone; + } + + return NULL; +} + #ifdef CONFIG_DEFERRED_STRUCT_PAGE_INIT static inline void pgdat_set_deferred_range(pg_data_t *pgdat) { @@ -750,20 +764,14 @@ defer_init(int nid, unsigned long pfn, unsigned long end_pfn) static void __meminit __init_deferred_page(unsigned long pfn, int nid) { - pg_data_t *pgdat = NODE_DATA(nid); - int zid; + struct zone *zone; if (early_page_initialised(pfn, nid)) return; - for (zid = 0; zid < MAX_NR_ZONES; zid++) { - struct zone *zone = &pgdat->node_zones[zid]; - - if (zone_spans_pfn(zone, pfn)) - break; - } - __init_single_page(pfn_to_page(pfn), pfn, zid, nid); - + zone = pfn_to_zone(pfn, nid); + __init_single_page(pfn_to_page(pfn), pfn, + zone ? zone_idx(zone) : MAX_NR_ZONES, nid); if (pageblock_aligned(pfn)) { enum migratetype mt = kho_scratch_migratetype(pfn, MIGRATE_MOVABLE); diff --git a/mm/mm_init.h b/mm/mm_init.h index 39f75df9be1c26..c9fc35e7e9f1f3 100644 --- a/mm/mm_init.h +++ b/mm/mm_init.h @@ -39,6 +39,7 @@ void memmap_init_range(unsigned long size, int nid, unsigned long zone, enum meminit_context context, struct vmem_altmap *altmap, int migratetype, bool isolate_pageblock); +struct zone *pfn_to_zone(unsigned long pfn, int nid); #if defined CONFIG_COMPACTION || defined CONFIG_CMA /* Free whole pageblock and set its migration type to MIGRATE_CMA. */ From d4c1b5dab9d1709e26f9edcf47c0a6bdd6a59368 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:46 +0800 Subject: [PATCH 0239/1012] mm/sparse-vmemmap: move helpers ahead of future callers Prepare for section-based vmemmap optimization by moving helpers that follow-up changes will use. vmemmap_get_tail() will be called from the PTE population path. section_nr_vmemmap_pages() will be made visible outside the memory hotplug code and called from sparse_init_nid(). Move vmemmap_alloc_block_zero() together with vmemmap_get_tail(), since the tail helper depends on it. Move section_nr_vmemmap_pages() earlier into its own CONFIG_MEMORY_HOTPLUG block. That lets the later patch change its visibility and callers without also moving the function body. No functional change is intended. Link: https://lore.kernel.org/20260910063256.64386-8-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/sparse-vmemmap.c | 134 +++++++++++++++++++++++--------------------- 1 file changed, 69 insertions(+), 65 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index dea7179fbc90b0..e2a2a5eab4dc43 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -148,6 +148,75 @@ void __meminit vmemmap_verify(pte_t *pte, int node, start, end - 1); } +#ifdef CONFIG_MEMORY_HOTPLUG +static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, + struct vmem_altmap *altmap, struct dev_pagemap *pgmap) +{ + const struct mem_section *ms = __pfn_to_section(pfn); + const int order = pgmap ? pgmap->vmemmap_shift : section_compound_order(ms); + const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; + const unsigned long pages_per_compound = 1UL << order; + + VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); + VM_WARN_ON_ONCE(nr_pages > PAGES_PER_SECTION); + + if (!vmemmap_can_optimize(altmap, pgmap) && !section_vmemmap_optimizable(ms)) + return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); + + if (order < PFN_SECTION_SHIFT) { + VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, pages_per_compound)); + return vmemmap_pages * nr_pages / pages_per_compound; + } + + VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)); + + if (IS_ALIGNED(pfn, pages_per_compound)) + return vmemmap_pages; + + return 0; +} +#endif + +static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) +{ + void *p = vmemmap_alloc_block(size, node); + + if (!p) + return NULL; + memset(p, 0, size); + + return p; +} + +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) +{ + struct page *p, *tail; + unsigned int idx; + int node = zone_to_nid(zone); + + if (WARN_ON_ONCE(order < VMEMMAP_OPTIMIZATION_MIN_ORDER)) + return NULL; + if (WARN_ON_ONCE(order > MAX_FOLIO_ORDER)) + return NULL; + + idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; + tail = zone->vmemmap_tails[idx]; + if (tail) + return tail; + p = vmemmap_alloc_block_zero(PAGE_SIZE, node); + if (!p) + return NULL; + for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) + init_compound_tail(p + i, NULL, order, zone); + + tail = virt_to_page(p); + zone->vmemmap_tails[idx] = tail; + + return tail; +} +#endif + static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, int node, struct vmem_altmap *altmap, unsigned long ptpfn, unsigned long flags) @@ -181,17 +250,6 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in return pte; } -static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) -{ - void *p = vmemmap_alloc_block(size, node); - - if (!p) - return NULL; - memset(p, 0, size); - - return p; -} - static pmd_t * __meminit vmemmap_pmd_populate(pud_t *pud, unsigned long addr, int node) { pmd_t *pmd = pmd_offset(pud, addr); @@ -323,33 +381,6 @@ void vmemmap_wrprotect_hvo(unsigned long addr, unsigned long end, } #ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP -static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) -{ - struct page *p, *tail; - unsigned int idx; - int node = zone_to_nid(zone); - - if (WARN_ON_ONCE(order < VMEMMAP_OPTIMIZATION_MIN_ORDER)) - return NULL; - if (WARN_ON_ONCE(order > MAX_FOLIO_ORDER)) - return NULL; - - idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; - tail = zone->vmemmap_tails[idx]; - if (tail) - return tail; - p = vmemmap_alloc_block_zero(PAGE_SIZE, node); - if (!p) - return NULL; - for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) - init_compound_tail(p + i, NULL, order, zone); - - tail = virt_to_page(p); - zone->vmemmap_tails[idx] = tail; - - return tail; -} - int __meminit vmemmap_populate_hvo(unsigned long addr, unsigned long end, unsigned int order, struct zone *zone, unsigned long headsize) @@ -646,33 +677,6 @@ void offline_mem_sections(unsigned long start_pfn, unsigned long end_pfn) } } -static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, - struct vmem_altmap *altmap, struct dev_pagemap *pgmap) -{ - const struct mem_section *ms = __pfn_to_section(pfn); - const int order = pgmap ? pgmap->vmemmap_shift : section_compound_order(ms); - const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; - const unsigned long pages_per_compound = 1UL << order; - - VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); - VM_WARN_ON_ONCE(nr_pages > PAGES_PER_SECTION); - - if (!vmemmap_can_optimize(altmap, pgmap) && !section_vmemmap_optimizable(ms)) - return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); - - if (order < PFN_SECTION_SHIFT) { - VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, pages_per_compound)); - return vmemmap_pages * nr_pages / pages_per_compound; - } - - VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)); - - if (IS_ALIGNED(pfn, pages_per_compound)) - return vmemmap_pages; - - return 0; -} - static struct page * __meminit populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) From b423954b519ccc9aa201b5a60891d9cae3737bd2 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:47 +0800 Subject: [PATCH 0240/1012] mm/sparse-vmemmap: support section-based vmemmap optimization Teach sparse-vmemmap population code to use the compound page order when deciding whether a vmemmap page can be optimized. With this information, the common sparse-vmemmap population path can allocate or reuse shared tail vmemmap pages directly instead of relying on HugeTLB-specific handling. This centralizes vmemmap optimization logic in the sparse-vmemmap code, based on section metadata, and prepares for sharing the same mechanism across different users of vmemmap optimization, including HugeTLB and DAX. Link: https://lore.kernel.org/20260910063256.64386-9-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/sparse-vmemmap.c | 46 +++++++++++++++++++++++++++++++++++++-------- mm/sparse.c | 4 ++-- mm/sparse.h | 7 +++++++ 3 files changed, 47 insertions(+), 10 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index e2a2a5eab4dc43..faebf344cdac9b 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -148,8 +148,7 @@ void __meminit vmemmap_verify(pte_t *pte, int node, start, end - 1); } -#ifdef CONFIG_MEMORY_HOTPLUG -static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, +int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { const struct mem_section *ms = __pfn_to_section(pfn); @@ -175,7 +174,6 @@ static int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long n return 0; } -#endif static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) { @@ -215,19 +213,44 @@ static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone * return tail; } +#else +static inline struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) +{ + return NULL; +} #endif +static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, + struct vmem_altmap *altmap) +{ + struct zone *zone; + struct page *page; + const unsigned int order = pfn_to_section_compound_order(pfn); + + if (!vmemmap_optimizable_pfn(pfn)) + return vmemmap_alloc_block_buf(PAGE_SIZE, node, altmap); + + zone = pfn_to_zone(pfn, node); + page = vmemmap_get_tail(order, zone); + if (!page) + return NULL; + + return page_address(page); +} + static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, int node, struct vmem_altmap *altmap, unsigned long ptpfn, unsigned long flags) { pte_t *pte = pte_offset_kernel(pmd, addr); + unsigned long pfn = page_to_pfn((struct page *)addr); + if (pte_none(ptep_get(pte))) { pte_t entry; - void *p; if (ptpfn == (unsigned long)-1) { - p = vmemmap_alloc_block_buf(PAGE_SIZE, node, altmap); + void *p = vmemmap_alloc_pte(pfn, node, altmap); + if (!p) return NULL; ptpfn = PHYS_PFN(__pa(p)); @@ -246,7 +269,8 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in } entry = pfn_pte(ptpfn, PAGE_KERNEL); set_pte_at(&init_mm, addr, pte, entry); - } + } else if (WARN_ON_ONCE(vmemmap_optimizable_pfn(pfn))) + return NULL; return pte; } @@ -435,6 +459,9 @@ int __meminit vmemmap_populate_hugepages(unsigned long start, unsigned long end, pmd_t *pmd; for (addr = start; addr < end; addr = next) { + unsigned long pfn = page_to_pfn((struct page *)addr); + const struct mem_section *ms = __pfn_to_section(pfn); + next = pmd_addr_end(addr, end); pgd = vmemmap_pgd_populate(addr, node); @@ -450,7 +477,7 @@ int __meminit vmemmap_populate_hugepages(unsigned long start, unsigned long end, return -ENOMEM; pmd = pmd_offset(pud, addr); - if (pmd_none(pmdp_get(pmd))) { + if (pmd_none(pmdp_get(pmd)) && !section_vmemmap_optimizable(ms)) { void *p; p = vmemmap_alloc_block_buf(PMD_SIZE, node, altmap); @@ -468,8 +495,11 @@ int __meminit vmemmap_populate_hugepages(unsigned long start, unsigned long end, */ return -ENOMEM; } - } else if (vmemmap_check_pmd(pmd, node, addr, next)) + } else if (vmemmap_check_pmd(pmd, node, addr, next)) { + if (WARN_ON_ONCE(section_vmemmap_optimizable(ms))) + return -EOPNOTSUPP; continue; + } if (vmemmap_populate_basepages(addr, next, node, altmap)) return -ENOMEM; } diff --git a/mm/sparse.c b/mm/sparse.c index c84b4c7b8c7069..e6cb67ca9c8d10 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -305,8 +305,8 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, nid, NULL, NULL); if (!map) panic("Failed to allocate memmap for section %lu\n", pnum); - memmap_boot_pages_add(DIV_ROUND_UP(PAGES_PER_SECTION * sizeof(struct page), - PAGE_SIZE)); + memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, + NULL, NULL)); sparse_init_early_section(nid, map, pnum, 0); } } diff --git a/mm/sparse.h b/mm/sparse.h index 563fc1f4d71778..74dd79c51e7598 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -111,8 +111,15 @@ static inline void sparse_init(void) {} */ #ifdef CONFIG_SPARSEMEM_VMEMMAP void sparse_init_subsection_map(void); +int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, + struct vmem_altmap *altmap, struct dev_pagemap *pgmap); #else static inline void sparse_init_subsection_map(void) {} +static inline int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, + struct vmem_altmap *altmap, struct dev_pagemap *pgmap) +{ + return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); +} #endif /* CONFIG_SPARSEMEM_VMEMMAP */ #endif /* __MM_SPARSE_H */ From 41e4099ba98baa2d3607aecc69d18c4ddbd5d2f3 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:48 +0800 Subject: [PATCH 0241/1012] mm/sparse: initialize memory sections earlier Upcoming HugeTLB bootmem changes need sparsemem section metadata before the HugeTLB bootmem allocation path runs. The memory sections are initialized from sparse_init(), which is called too late for that setup. Move the code that initializes sparsemem section metadata for memblock ranges into mm_core_init_early(), before free_area_init() and the HugeTLB bootmem setup. Rename the helper to sparse_sections_init() so the new caller describes the sparsemem-specific initialization step. This is a preparatory change. Link: https://lore.kernel.org/20260910063256.64386-10-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/mm_init.c | 1 + mm/sparse.c | 9 +-------- mm/sparse.h | 2 ++ 3 files changed, 4 insertions(+), 8 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index ee3bd792176890..304c88da5cce21 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -2642,6 +2642,7 @@ void __init mm_core_init_early(void) { kho_memory_init_early(); + sparse_sections_init(); free_area_init(); hugetlb_cma_reserve(); diff --git a/mm/sparse.c b/mm/sparse.c index e6cb67ca9c8d10..428d81838ece4a 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -191,12 +191,7 @@ static void __init memory_present(int nid, unsigned long start, unsigned long en } } -/* - * Mark all memblocks as present using memory_present(). - * This is a convenience function that is useful to mark all of the systems - * memory as present during initialization. - */ -static void __init memblocks_present(void) +void __init sparse_sections_init(void) { unsigned long start, end; int i, nid; @@ -322,8 +317,6 @@ void __init sparse_init(void) unsigned long pnum_end, pnum_begin, map_count = 1; int nid_begin; - memblocks_present(); - if (compound_info_has_mask()) { VM_WARN_ON_ONCE(!IS_ALIGNED((unsigned long) pfn_to_page(0), MAX_FOLIO_VMEMMAP_ALIGN)); diff --git a/mm/sparse.h b/mm/sparse.h index 74dd79c51e7598..e998347867f4cf 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -59,6 +59,7 @@ static inline bool vmemmap_optimizable_order(unsigned int order) */ #ifdef CONFIG_SPARSEMEM void sparse_init(void); +void sparse_sections_init(void); int sparse_index_init(unsigned long section_nr, int nid); static inline void sparse_init_one_section(struct mem_section *ms, @@ -104,6 +105,7 @@ static inline bool section_vmemmap_optimizable(const struct mem_section *ms) } #else static inline void sparse_init(void) {} +static inline void sparse_sections_init(void) {} #endif /* CONFIG_SPARSEMEM */ /* From 2257d64ea1bf9c1fc9076be004fde041863731d1 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:49 +0800 Subject: [PATCH 0242/1012] mm/hugetlb: switch HugeTLB to section-based vmemmap optimization HugeTLB bootmem vmemmap optimization still carries its own early setup path, including pre-populating optimized mappings before the generic sparse-vmemmap code runs. Now that the section-based vmemmap optimization can derive HugeTLB vmemmap deduplication from section metadata, HugeTLB only needs to mark the bootmem huge page range with the appropriate order. The generic sparse-vmemmap population path can then allocate and map the shared tail vmemmap pages without any HugeTLB-specific early population code. Do that by recording the compound page order when a bootmem huge page is allocated and dropping the dedicated pre-HVO helpers and related special-casing. This removes duplicate early setup logic and switches HugeTLB to the section-based vmemmap optimization path. Link: https://lore.kernel.org/20260910063256.64386-11-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/hugetlb.h | 1 - include/linux/mm.h | 3 -- mm/hugetlb.c | 31 +++----------- mm/hugetlb_vmemmap.c | 91 ++++------------------------------------- mm/hugetlb_vmemmap.h | 14 +++---- mm/sparse-vmemmap.c | 31 -------------- mm/sparse.h | 30 ++++++++++++++ 7 files changed, 47 insertions(+), 154 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 16c4c4caa126ce..fe28f98e1b220e 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -171,7 +171,6 @@ struct address_space *hugetlb_folio_mapping_lock_write(struct folio *folio); extern int movable_gigantic_pages __read_mostly; extern int sysctl_hugetlb_shm_group __read_mostly; -extern struct list_head huge_boot_pages[MAX_NUMNODES]; void hugetlb_bootmem_struct_page_init(void); void hugetlb_bootmem_alloc(void); diff --git a/include/linux/mm.h b/include/linux/mm.h index dd09c438fa23ec..441bd39eab7343 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -5159,9 +5159,6 @@ int vmemmap_populate_hugepages(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap); int vmemmap_populate(unsigned long start, unsigned long end, int node, struct vmem_altmap *altmap); -int vmemmap_populate_hvo(unsigned long start, unsigned long end, - unsigned int order, struct zone *zone, - unsigned long headsize); void vmemmap_wrprotect_hvo(unsigned long start, unsigned long end, int node, unsigned long headsize); void vmemmap_populate_print_last(void); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index f1c92e80c34672..89d7d684b8d968 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -52,6 +52,7 @@ #include "hugetlb_cma.h" #include "hugetlb_internal.h" #include "mm_init.h" +#include "sparse.h" #include int hugetlb_max_hstate __read_mostly; @@ -59,7 +60,7 @@ unsigned int default_hstate_idx; struct hstate hstates[HUGE_MAX_HSTATE]; __initdata nodemask_t hugetlb_bootmem_nodes; -__initdata struct list_head huge_boot_pages[MAX_NUMNODES]; +static struct list_head huge_boot_pages[MAX_NUMNODES] __initdata; /* * Due to ordering constraints across the init code for various @@ -3161,6 +3162,7 @@ static bool __init alloc_bootmem_huge_page(struct hstate *h, int nid) } else { list_add_tail(&m->list, &huge_boot_pages[nid]); m->flags |= HUGE_BOOTMEM_ZONES_VALID; + hugetlb_vmemmap_optimize_bootmem_page(m); /* * Only initialize the head struct page in memmap_init_reserved_pages, * rest of the struct pages will be initialized by the HugeTLB @@ -3321,6 +3323,8 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) * this folio. */ folio_set_hugetlb_vmemmap_optimized(folio); + section_set_compound_order_range(folio_pfn(folio), + folio_nr_pages(folio), 0); if (hugetlb_bootmem_page_earlycma(m)) folio_set_hugetlb_cma(folio); @@ -3364,31 +3368,6 @@ void __init hugetlb_bootmem_struct_page_init(void) .max_threads = num_node_state(N_MEMORY), .numa_aware = true, }; -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP - struct zone *zone; - - for_each_zone(zone) { - for (int i = 0; i < VMEMMAP_OPTIMIZATION_NR_ORDERS; i++) { - struct page *tail, *p; - unsigned int order; - - tail = zone->vmemmap_tails[i]; - if (!tail) - continue; - - order = i + VMEMMAP_OPTIMIZATION_MIN_ORDER; - p = page_to_virt(tail); - /* - * prep_and_add_bootmem_folios() can access pageblock - * flags on bootmem HugeTLB pages, so initialize the - * shared tail struct pages here before bootmem folios - * start using them. - */ - for (int j = 0; j < PAGE_SIZE / sizeof(struct page); j++) - init_compound_tail(p + j, NULL, order, zone); - } - } -#endif padata_do_multithreaded(&job); } diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index c48fcea076a522..d1b031dcd17740 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -18,8 +18,7 @@ #include #include "hugetlb_vmemmap.h" -#include "internal.h" -#include "mm_init.h" +#include "sparse.h" /** * struct vmemmap_remap_walk - walk vmemmap page table @@ -706,95 +705,19 @@ void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head __hugetlb_vmemmap_optimize_folios(h, folio_list, true); } -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT - -/* Return true of a bootmem allocated HugeTLB page should be pre-HVO-ed */ -static bool vmemmap_should_optimize_bootmem_page(struct huge_bootmem_page *m) +void __init hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) { - unsigned long section_size, psize, pmd_vmemmap_size; - phys_addr_t paddr; - - if (!READ_ONCE(vmemmap_optimize_enabled)) - return false; - - if (!hugetlb_vmemmap_optimizable(m->hstate)) - return false; - - psize = huge_page_size(m->hstate); - paddr = virt_to_phys(m); - - /* - * Pre-HVO only works if the bootmem huge page - * is aligned to the section size. - */ - section_size = (1UL << PA_SECTION_SHIFT); - if (!IS_ALIGNED(paddr, section_size) || - !IS_ALIGNED(psize, section_size)) - return false; - - /* - * The pre-HVO code does not deal with splitting PMDS, - * so the bootmem page must be aligned to the number - * of base pages that can be mapped with one vmemmap PMD. - */ - pmd_vmemmap_size = (PMD_SIZE / (sizeof(struct page))) << PAGE_SHIFT; - if (!IS_ALIGNED(paddr, pmd_vmemmap_size) || - !IS_ALIGNED(psize, pmd_vmemmap_size)) - return false; - - return true; -} - -/* - * Initialize memmap section for a gigantic page, HVO-style. - */ -void __init hugetlb_vmemmap_init_early(int nid) -{ - unsigned long psize, paddr, section_size; - unsigned long ns, i, pnum, pfn, nr_pages; - unsigned long start, end; - struct huge_bootmem_page *m = NULL; - void *map; + struct hstate *h = m->hstate; + unsigned long pfn = PHYS_PFN(__pa(m)); if (!READ_ONCE(vmemmap_optimize_enabled)) return; - section_size = (1UL << PA_SECTION_SHIFT); - - list_for_each_entry(m, &huge_boot_pages[nid], list) { - struct zone *zone; - - if (!vmemmap_should_optimize_bootmem_page(m)) - continue; - - nr_pages = pages_per_huge_page(m->hstate); - psize = nr_pages << PAGE_SHIFT; - paddr = virt_to_phys(m); - pfn = PHYS_PFN(paddr); - map = pfn_to_page(pfn); - start = (unsigned long)map; - end = start + hugetlb_vmemmap_size(m->hstate); - zone = pfn_to_zone(pfn, nid); - - if (vmemmap_populate_hvo(start, end, huge_page_order(m->hstate), - zone, HUGETLB_VMEMMAP_RESERVE_SIZE)) - panic("Failed to allocate memmap for HugeTLB page\n"); - memmap_boot_pages_add(DIV_ROUND_UP(HUGETLB_VMEMMAP_RESERVE_SIZE, PAGE_SIZE)); - - pnum = pfn_to_section_nr(pfn); - ns = psize / section_size; - - for (i = 0; i < ns; i++) { - sparse_init_early_section(nid, map, pnum, - SECTION_IS_VMEMMAP_PREINIT); - map += section_map_size(); - pnum++; - } - + section_set_compound_order_range(pfn, pages_per_huge_page(h), + huge_page_order(h)); + if (vmemmap_optimizable_order(pfn_to_section_compound_order(pfn))) m->flags |= HUGE_BOOTMEM_HVO; - } } -#endif static const struct ctl_table hugetlb_vmemmap_sysctls[] = { { diff --git a/mm/hugetlb_vmemmap.h b/mm/hugetlb_vmemmap.h index 7ac49c52457ddf..20eb03df542a43 100644 --- a/mm/hugetlb_vmemmap.h +++ b/mm/hugetlb_vmemmap.h @@ -9,8 +9,7 @@ #ifndef _LINUX_HUGETLB_VMEMMAP_H #define _LINUX_HUGETLB_VMEMMAP_H #include -#include -#include +#include "internal.h" /* * Reserve one vmemmap page, all vmemmap addresses are mapped to it. See @@ -27,10 +26,7 @@ long hugetlb_vmemmap_restore_folios(const struct hstate *h, void hugetlb_vmemmap_optimize_folio(const struct hstate *h, struct folio *folio); void hugetlb_vmemmap_optimize_folios(struct hstate *h, struct list_head *folio_list); void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head *folio_list); -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -void hugetlb_vmemmap_init_early(int nid); -#endif - +void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m); static inline unsigned int hugetlb_vmemmap_size(const struct hstate *h) { @@ -76,13 +72,13 @@ static inline void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, { } -static inline void hugetlb_vmemmap_init_early(int nid) +static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate *h) { + return 0; } -static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate *h) +static inline void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) { - return 0; } #endif /* CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP */ diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index faebf344cdac9b..8c2b9bed236d15 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -32,8 +32,6 @@ #include #include -#include "hugetlb_vmemmap.h" - /* * Flags for vmemmap_populate_range and friends. */ @@ -404,34 +402,6 @@ void vmemmap_wrprotect_hvo(unsigned long addr, unsigned long end, } } -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP -int __meminit vmemmap_populate_hvo(unsigned long addr, unsigned long end, - unsigned int order, struct zone *zone, - unsigned long headsize) -{ - unsigned long maddr; - struct page *tail; - pte_t *pte; - int node = zone_to_nid(zone); - - tail = vmemmap_get_tail(order, zone); - if (!tail) - return -ENOMEM; - - for (maddr = addr; maddr < addr + headsize; maddr += PAGE_SIZE) { - pte = vmemmap_populate_address(maddr, node, NULL, -1, 0); - if (!pte) - return -ENOMEM; - } - - /* - * Reuse the last page struct page mapped above for the rest. - */ - return vmemmap_populate_range(maddr, end, node, NULL, - page_to_pfn(tail), 0); -} -#endif - void __weak __meminit vmemmap_set_pmd(pmd_t *pmd, void *p, int node, unsigned long addr, unsigned long next) { @@ -634,7 +604,6 @@ struct page * __meminit __populate_section_memmap(unsigned long pfn, */ void __init sparse_vmemmap_init_nid_early(int nid) { - hugetlb_vmemmap_init_early(nid); } #endif diff --git a/mm/sparse.h b/mm/sparse.h index e998347867f4cf..d3a71ef4fad0fe 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -16,6 +16,26 @@ static inline unsigned int section_compound_order(const struct mem_section *sect return section->compound_page_order; } +static inline void section_set_compound_order(struct mem_section *section, + unsigned int order) +{ + VM_WARN_ON(section_compound_order(section) && order && + section_compound_order(section) != order); + section->compound_page_order = order; +} + +static inline void section_set_compound_order_range(unsigned long pfn, + unsigned long nr_pages, unsigned int order) +{ + unsigned long section_nr = pfn_to_section_nr(pfn); + + if (!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)) + return; + + for (unsigned long i = 0; i < nr_pages / PAGES_PER_SECTION; i++) + section_set_compound_order(__nr_to_section(section_nr + i), order); +} + static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) { return section_compound_order(__pfn_to_section(pfn)); @@ -26,6 +46,16 @@ static inline unsigned int section_compound_order(const struct mem_section *sect return 0; } +static inline void section_set_compound_order(struct mem_section *section, + unsigned int order) +{ +} + +static inline void section_set_compound_order_range(unsigned long pfn, + unsigned long nr_pages, unsigned int order) +{ +} + static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) { return 0; From 9367afc8101a128f47a769f733b3ad6100eeebb5 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:50 +0800 Subject: [PATCH 0243/1012] mm/sparse-vmemmap: remove SPARSEMEM_VMEMMAP_PREINIT support SPARSEMEM_VMEMMAP_PREINIT existed only to support HugeTLB's early vmemmap optimization setup. Now that HugeTLB bootmem vmemmap optimization uses the common section-based sparse-vmemmap path, the sparse initialization code no longer needs a separate pre-initialization mechanism for vmemmap population. Remove the related Kconfig symbols, section flag, and empty early hook, so present sections always go through the normal sparse setup path. Link: https://lore.kernel.org/20260910063256.64386-12-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Acked-by: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- arch/x86/Kconfig | 1 - fs/Kconfig | 1 - include/linux/mmzone.h | 25 ------------------------- mm/Kconfig | 5 ----- mm/sparse-vmemmap.c | 13 ------------- mm/sparse.c | 23 ++++++++--------------- 6 files changed, 8 insertions(+), 60 deletions(-) diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig index 15fd9ec5ecacb7..7aa74bcc72f9db 100644 --- a/arch/x86/Kconfig +++ b/arch/x86/Kconfig @@ -150,7 +150,6 @@ config X86 select ARCH_WANT_LD_ORPHAN_WARN select ARCH_WANT_OPTIMIZE_DAX_VMEMMAP if X86_64 select ARCH_WANT_OPTIMIZE_HUGETLB_VMEMMAP if X86_64 - select ARCH_WANT_HUGETLB_VMEMMAP_PREINIT if X86_64 select ARCH_WANTS_THP_SWAP if X86_64 select ARCH_HAS_PARANOID_L1D_FLUSH select ARCH_WANT_IRQS_OFF_ACTIVATE_MM diff --git a/fs/Kconfig b/fs/Kconfig index e05917adcd608e..d1c210c6508f0a 100644 --- a/fs/Kconfig +++ b/fs/Kconfig @@ -278,7 +278,6 @@ config HUGETLB_PAGE_OPTIMIZE_VMEMMAP def_bool HUGETLB_PAGE depends on ARCH_WANT_OPTIMIZE_HUGETLB_VMEMMAP depends on SPARSEMEM_VMEMMAP - select SPARSEMEM_VMEMMAP_PREINIT if ARCH_WANT_HUGETLB_VMEMMAP_PREINIT config HUGETLB_PMD_PAGE_TABLE_SHARING def_bool HUGETLB_PAGE diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 2c4f63e379da7f..f2d39a888eac51 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2095,9 +2095,6 @@ enum { SECTION_IS_EARLY_BIT, #ifdef CONFIG_ZONE_DEVICE SECTION_TAINT_ZONE_DEVICE_BIT, -#endif -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT - SECTION_IS_VMEMMAP_PREINIT_BIT, #endif SECTION_MAP_LAST_BIT, }; @@ -2109,9 +2106,6 @@ enum { #ifdef CONFIG_ZONE_DEVICE #define SECTION_TAINT_ZONE_DEVICE BIT(SECTION_TAINT_ZONE_DEVICE_BIT) #endif -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -#define SECTION_IS_VMEMMAP_PREINIT BIT(SECTION_IS_VMEMMAP_PREINIT_BIT) -#endif #define SECTION_MAP_MASK (~(BIT(SECTION_MAP_LAST_BIT) - 1)) #define SECTION_NID_SHIFT SECTION_MAP_LAST_BIT @@ -2166,24 +2160,6 @@ static inline int online_device_section(const struct mem_section *section) } #endif -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -static inline int preinited_vmemmap_section(const struct mem_section *section) -{ - return (section && - (section->section_mem_map & SECTION_IS_VMEMMAP_PREINIT)); -} - -void sparse_vmemmap_init_nid_early(int nid); -#else -static inline int preinited_vmemmap_section(const struct mem_section *section) -{ - return 0; -} -static inline void sparse_vmemmap_init_nid_early(int nid) -{ -} -#endif - static inline int online_section_nr(unsigned long nr) { return online_section(__nr_to_section(nr)); @@ -2385,7 +2361,6 @@ static inline unsigned long next_present_section_nr(unsigned long section_nr) #endif #else -#define sparse_vmemmap_init_nid_early(_nid) do {} while (0) #define pfn_in_present_section pfn_valid #endif /* CONFIG_SPARSEMEM */ diff --git a/mm/Kconfig b/mm/Kconfig index 604c58199acbf8..2c385f8b29445e 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -461,8 +461,6 @@ config SPARSEMEM_VMEMMAP pfn_to_page and page_to_pfn operations. This is the most efficient option when sufficient kernel resources are available. -config SPARSEMEM_VMEMMAP_PREINIT - bool # # Select this config option from the architecture Kconfig, if it is preferred # to enable the feature of HugeTLB/dev_dax vmemmap optimization. @@ -473,9 +471,6 @@ config ARCH_WANT_OPTIMIZE_DAX_VMEMMAP config ARCH_WANT_OPTIMIZE_HUGETLB_VMEMMAP bool -config ARCH_WANT_HUGETLB_VMEMMAP_PREINIT - bool - config HAVE_MEMBLOCK_PHYS_MAP bool diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 8c2b9bed236d15..f22d815d7af04e 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -594,19 +594,6 @@ struct page * __meminit __populate_section_memmap(unsigned long pfn, return pfn_to_page(pfn); } -#ifdef CONFIG_SPARSEMEM_VMEMMAP_PREINIT -/* - * This is called just before initializing sections for a NUMA node. - * Any special initialization that needs to be done before the - * generic initialization can be done from here. Sections that - * are initialized in hooks called from here will be skipped by - * the generic initialization. - */ -void __init sparse_vmemmap_init_nid_early(int nid) -{ -} -#endif - static void subsection_mask_set(unsigned long *map, unsigned long pfn, unsigned long nr_pages) { diff --git a/mm/sparse.c b/mm/sparse.c index 428d81838ece4a..f4f393a033a424 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -283,27 +283,20 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, if (sparse_usage_init(nid, map_count)) panic("Failed to allocate usemap for node %d\n", nid); - sparse_vmemmap_init_nid_early(nid); - for_each_present_section_nr(pnum_begin, pnum) { - struct mem_section *ms; unsigned long pfn = section_nr_to_pfn(pnum); + struct page *map; if (pnum >= pnum_end) break; - ms = __nr_to_section(pnum); - if (!preinited_vmemmap_section(ms)) { - struct page *map; - - map = __populate_section_memmap(pfn, PAGES_PER_SECTION, - nid, NULL, NULL); - if (!map) - panic("Failed to allocate memmap for section %lu\n", pnum); - memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, - NULL, NULL)); - sparse_init_early_section(nid, map, pnum, 0); - } + map = __populate_section_memmap(pfn, PAGES_PER_SECTION, + nid, NULL, NULL); + if (!map) + panic("Failed to allocate memmap for section %lu\n", pnum); + memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, + NULL, NULL)); + sparse_init_early_section(nid, map, pnum, 0); } sparse_usage_fini(); } From a67442d84b08187807ff2b66d08f60e64630cf31 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:51 +0800 Subject: [PATCH 0244/1012] mm/sparse: inline usemap allocation into sparse_init_nid() After removing SPARSEMEM_VMEMMAP_PREINIT, sparse_init_nid() no longer needs the transient sparse_usagebuf state and its helper wrappers. Allocate the usemap buffer directly in sparse_init_nid(), pass it to sparse_init_one_section(), and drop sparse_usage_init(), sparse_usage_fini(), and sparse_init_early_section(). Link: https://lore.kernel.org/20260910063256.64386-13-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Qi Zheng Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mmzone.h | 3 --- mm/sparse.c | 46 +++++++----------------------------------- 2 files changed, 7 insertions(+), 42 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index f2d39a888eac51..6acc14b169bbe6 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2223,9 +2223,6 @@ static inline bool pfn_section_first_valid(struct mem_section *ms, unsigned long } #endif -void sparse_init_early_section(int nid, struct page *map, unsigned long pnum, - unsigned long flags); - #ifndef CONFIG_HAVE_ARCH_PFN_VALID /** * pfn_valid - check if there is a valid memory map entry for a PFN diff --git a/mm/sparse.c b/mm/sparse.c index f4f393a033a424..d19c173b40feb8 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -234,42 +234,6 @@ void __weak __meminit vmemmap_populate_print_last(void) { } -static void *sparse_usagebuf __initdata; -static void *sparse_usagebuf_end __initdata; - -/* - * Helper function that is used for generic section initialization, and - * can also be used by any hooks added above. - */ -void __init sparse_init_early_section(int nid, struct page *map, - unsigned long pnum, unsigned long flags) -{ - BUG_ON(!sparse_usagebuf || sparse_usagebuf >= sparse_usagebuf_end); - sparse_init_one_section(__nr_to_section(pnum), pnum, map, - sparse_usagebuf, SECTION_IS_EARLY | flags); - sparse_usagebuf = (void *)sparse_usagebuf + mem_section_usage_size(); -} - -static int __init sparse_usage_init(int nid, unsigned long map_count) -{ - unsigned long size; - - size = mem_section_usage_size() * map_count; - sparse_usagebuf = memblock_alloc_node(size, SMP_CACHE_BYTES, nid); - if (!sparse_usagebuf) { - sparse_usagebuf_end = NULL; - return -ENOMEM; - } - - sparse_usagebuf_end = sparse_usagebuf + size; - return 0; -} - -static void __init sparse_usage_fini(void) -{ - sparse_usagebuf = sparse_usagebuf_end = NULL; -} - /* * Initialize sparse on a specific node. The node spans [pnum_begin, pnum_end) * And number of present sections in this node is map_count. @@ -279,8 +243,11 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, unsigned long map_count) { unsigned long pnum; + struct mem_section_usage *usage; - if (sparse_usage_init(nid, map_count)) + usage = memblock_alloc_node(map_count * mem_section_usage_size(), + SMP_CACHE_BYTES, nid); + if (!usage) panic("Failed to allocate usemap for node %d\n", nid); for_each_present_section_nr(pnum_begin, pnum) { @@ -296,9 +263,10 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, panic("Failed to allocate memmap for section %lu\n", pnum); memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, NULL, NULL)); - sparse_init_early_section(nid, map, pnum, 0); + sparse_init_one_section(__nr_to_section(pnum), pnum, map, usage, + SECTION_IS_EARLY); + usage = (void *)usage + mem_section_usage_size(); } - sparse_usage_fini(); } /* From 7cb477243b83c697583619d8b0d8b1c903987780 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:52 +0800 Subject: [PATCH 0245/1012] mm/sparse: remove section_map_size() section_map_size() no longer provides any shared logic. After the sparse-vmemmap changes, its only remaining user is the !CONFIG_SPARSEMEM_VMEMMAP path in __populate_section_memmap(), which can compute the size inline with PAGE_ALIGN(sizeof(struct page) * PAGES_PER_SECTION). Remove section_map_size() and inline the remaining calculation. Link: https://lore.kernel.org/20260910063256.64386-14-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Qi Zheng Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mm.h | 1 - mm/sparse.c | 15 ++------------- 2 files changed, 2 insertions(+), 14 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 441bd39eab7343..b19711b6dbc69a 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -5140,7 +5140,6 @@ static inline void print_vma_addr(char *prefix, unsigned long rip) } #endif -unsigned long section_map_size(void); struct page * __populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, struct dev_pagemap *pgmap); diff --git a/mm/sparse.c b/mm/sparse.c index d19c173b40feb8..cc28bb41fdb1f1 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -208,23 +208,12 @@ void __init sparse_sections_init(void) memory_present(nid, start, end); } -#ifdef CONFIG_SPARSEMEM_VMEMMAP -unsigned long __init section_map_size(void) -{ - return ALIGN(sizeof(struct page) * PAGES_PER_SECTION, PMD_SIZE); -} - -#else -unsigned long __init section_map_size(void) -{ - return PAGE_ALIGN(sizeof(struct page) * PAGES_PER_SECTION); -} - +#ifndef CONFIG_SPARSEMEM_VMEMMAP struct page __init *__populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { - unsigned long size = section_map_size(); + const unsigned long size = PAGE_ALIGN(sizeof(struct page) * PAGES_PER_SECTION); return memmap_alloc(size, size, __pa(MAX_DMA_ADDRESS), nid, false); } From 34f06c274fd2b64f325dfe46c9f08d1e0d222228 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:53 +0800 Subject: [PATCH 0246/1012] mm/hugetlb: remove HUGE_BOOTMEM_HVO The HUGE_BOOTMEM_HVO flag tracked whether a bootmem huge page had already gone through the old early vmemmap optimization path. Now that HugeTLB uses section-based vmemmap optimization, that state is already reflected in the compound page order stored in section metadata. Remove HUGE_BOOTMEM_HVO and its helper, and use the section state directly when deciding whether to mark a folio as vmemmap-optimized. Link: https://lore.kernel.org/20260910063256.64386-15-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/hugetlb.h | 5 ++--- mm/hugetlb.c | 16 +++------------- mm/hugetlb_vmemmap.c | 2 -- 3 files changed, 5 insertions(+), 18 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index fe28f98e1b220e..3559041a5a5780 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -675,9 +675,8 @@ struct hstate { char name[HSTATE_NAME_LEN]; }; -#define HUGE_BOOTMEM_HVO 0x0001 -#define HUGE_BOOTMEM_ZONES_VALID 0x0002 -#define HUGE_BOOTMEM_CMA 0x0004 +#define HUGE_BOOTMEM_ZONES_VALID BIT(0) +#define HUGE_BOOTMEM_CMA BIT(1) int isolate_or_dissolve_huge_folio(struct folio *folio, struct list_head *list); int replace_free_hugepage_folios(unsigned long start_pfn, unsigned long end_pfn); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 89d7d684b8d968..73999023cfdd39 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3220,11 +3220,6 @@ static void __init hugetlb_folio_init_vmemmap(struct folio *folio, prep_compound_head(&folio->page, huge_page_order(h)); } -static bool __init hugetlb_bootmem_page_prehvo(struct huge_bootmem_page *m) -{ - return m->flags & HUGE_BOOTMEM_HVO; -} - static bool __init hugetlb_bootmem_page_earlycma(struct huge_bootmem_page *m) { return m->flags & HUGE_BOOTMEM_CMA; @@ -3299,6 +3294,7 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) list_for_each_entry_safe(m, tm, &huge_boot_pages[nid], list) { struct page *page = virt_to_page(m); struct folio *folio = (void *)page; + const unsigned long pfn = folio_pfn(folio); h = m->hstate; /* @@ -3316,15 +3312,9 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) HUGETLB_VMEMMAP_RESERVE_PAGES); init_new_hugetlb_folio(folio); - if (hugetlb_bootmem_page_prehvo(m)) - /* - * If pre-HVO was done, just set the - * flag, the HVO code will then skip - * this folio. - */ + if (vmemmap_optimizable_order(pfn_to_section_compound_order(pfn))) folio_set_hugetlb_vmemmap_optimized(folio); - section_set_compound_order_range(folio_pfn(folio), - folio_nr_pages(folio), 0); + section_set_compound_order_range(pfn, folio_nr_pages(folio), 0); if (hugetlb_bootmem_page_earlycma(m)) folio_set_hugetlb_cma(folio); diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index d1b031dcd17740..fba0c5a7d46fc4 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -715,8 +715,6 @@ void __init hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) section_set_compound_order_range(pfn, pages_per_huge_page(h), huge_page_order(h)); - if (vmemmap_optimizable_order(pfn_to_section_compound_order(pfn))) - m->flags |= HUGE_BOOTMEM_HVO; } static const struct ctl_table hugetlb_vmemmap_sysctls[] = { From b71b7400d0a104fd778eca73ce2cddf287fc310f Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:54 +0800 Subject: [PATCH 0247/1012] mm/hugetlb: remove HUGE_BOOTMEM_CMA Track early CMA hugetlb pages from the hstate instead of storing a redundant bootmem flag. This removes the unused helper and keeps the bootmem metadata minimal. Link: https://lore.kernel.org/20260910063256.64386-16-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/hugetlb.h | 1 - mm/hugetlb.c | 14 ++++---------- 2 files changed, 4 insertions(+), 11 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 3559041a5a5780..255a258f11d13c 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -676,7 +676,6 @@ struct hstate { }; #define HUGE_BOOTMEM_ZONES_VALID BIT(0) -#define HUGE_BOOTMEM_CMA BIT(1) int isolate_or_dissolve_huge_folio(struct folio *folio, struct list_head *list); int replace_free_hugepage_folios(unsigned long start_pfn, unsigned long end_pfn); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 73999023cfdd39..4ecf5db73de451 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -3144,7 +3144,7 @@ static bool __init alloc_bootmem_huge_page(struct hstate *h, int nid) */ INIT_LIST_HEAD(&m->list); m->hstate = h; - m->flags = hugetlb_early_cma(h) ? HUGE_BOOTMEM_CMA : 0; + m->flags = 0; /* CMA pages: zone-crossing is validated in hugetlb_cma_reserve(). */ if (!hugetlb_early_cma(h) && @@ -3220,11 +3220,6 @@ static void __init hugetlb_folio_init_vmemmap(struct folio *folio, prep_compound_head(&folio->page, huge_page_order(h)); } -static bool __init hugetlb_bootmem_page_earlycma(struct huge_bootmem_page *m) -{ - return m->flags & HUGE_BOOTMEM_CMA; -} - /* * memblock-allocated pageblocks might not have the migrate type set * if marked with the 'noinit' flag. Set it to the default (MIGRATE_MOVABLE) @@ -3316,9 +3311,6 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) folio_set_hugetlb_vmemmap_optimized(folio); section_set_compound_order_range(pfn, folio_nr_pages(folio), 0); - if (hugetlb_bootmem_page_earlycma(m)) - folio_set_hugetlb_cma(folio); - list_add(&folio->lru, &folio_list); /* @@ -3329,7 +3321,9 @@ static void __init gather_bootmem_prealloc_node(unsigned long nid) * For CMA pages, this is done in init_cma_pageblock * (via hugetlb_bootmem_init_migratetype), so skip it here. */ - if (!folio_test_hugetlb_cma(folio)) + if (hugetlb_early_cma(h)) + folio_set_hugetlb_cma(folio); + else adjust_managed_page_count(page, pages_per_huge_page(h)); cond_resched(); } From b53e52cfba4601022d67ee201d32391f8533d503 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:55 +0800 Subject: [PATCH 0248/1012] mm/hugetlb: localize struct huge_bootmem_page struct huge_bootmem_page is only used by hugetlb boot-time allocation code, but its definition currently lives in mm/internal.h because hugetlb_vmemmap_optimize_bootmem_page() takes it as an argument. This exposes a hugetlb-specific internal type more broadly than needed. Change hugetlb_vmemmap_optimize_bootmem_page() to take the information it actually needs. With that interface, mm/hugetlb_vmemmap.h no longer needs to include mm/internal.h, and struct huge_bootmem_page can move into mm/hugetlb.c. Link: https://lore.kernel.org/20260910063256.64386-17-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/hugetlb.c | 8 +++++++- mm/hugetlb_vmemmap.c | 9 +++------ mm/hugetlb_vmemmap.h | 5 ++--- mm/internal.h | 7 ------- 4 files changed, 12 insertions(+), 17 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 4ecf5db73de451..cdd061953ab130 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -55,6 +55,12 @@ #include "sparse.h" #include +struct huge_bootmem_page { + struct list_head list; + struct hstate *hstate; + unsigned long flags; +}; + int hugetlb_max_hstate __read_mostly; unsigned int default_hstate_idx; struct hstate hstates[HUGE_MAX_HSTATE]; @@ -3162,7 +3168,7 @@ static bool __init alloc_bootmem_huge_page(struct hstate *h, int nid) } else { list_add_tail(&m->list, &huge_boot_pages[nid]); m->flags |= HUGE_BOOTMEM_ZONES_VALID; - hugetlb_vmemmap_optimize_bootmem_page(m); + hugetlb_vmemmap_optimize_bootmem_page(pfn, huge_page_order(h)); /* * Only initialize the head struct page in memmap_init_reserved_pages, * rest of the struct pages will be initialized by the HugeTLB diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index fba0c5a7d46fc4..f977d0a7e00274 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -19,6 +19,7 @@ #include #include "hugetlb_vmemmap.h" #include "sparse.h" +#include "internal.h" /** * struct vmemmap_remap_walk - walk vmemmap page table @@ -705,16 +706,12 @@ void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head __hugetlb_vmemmap_optimize_folios(h, folio_list, true); } -void __init hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) +void __init hugetlb_vmemmap_optimize_bootmem_page(unsigned long pfn, unsigned int order) { - struct hstate *h = m->hstate; - unsigned long pfn = PHYS_PFN(__pa(m)); - if (!READ_ONCE(vmemmap_optimize_enabled)) return; - section_set_compound_order_range(pfn, pages_per_huge_page(h), - huge_page_order(h)); + section_set_compound_order_range(pfn, 1UL << order, order); } static const struct ctl_table hugetlb_vmemmap_sysctls[] = { diff --git a/mm/hugetlb_vmemmap.h b/mm/hugetlb_vmemmap.h index 20eb03df542a43..464192e32decf4 100644 --- a/mm/hugetlb_vmemmap.h +++ b/mm/hugetlb_vmemmap.h @@ -9,7 +9,6 @@ #ifndef _LINUX_HUGETLB_VMEMMAP_H #define _LINUX_HUGETLB_VMEMMAP_H #include -#include "internal.h" /* * Reserve one vmemmap page, all vmemmap addresses are mapped to it. See @@ -26,7 +25,7 @@ long hugetlb_vmemmap_restore_folios(const struct hstate *h, void hugetlb_vmemmap_optimize_folio(const struct hstate *h, struct folio *folio); void hugetlb_vmemmap_optimize_folios(struct hstate *h, struct list_head *folio_list); void hugetlb_vmemmap_optimize_bootmem_folios(struct hstate *h, struct list_head *folio_list); -void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m); +void hugetlb_vmemmap_optimize_bootmem_page(unsigned long pfn, unsigned int order); static inline unsigned int hugetlb_vmemmap_size(const struct hstate *h) { @@ -77,7 +76,7 @@ static inline unsigned int hugetlb_vmemmap_optimizable_size(const struct hstate return 0; } -static inline void hugetlb_vmemmap_optimize_bootmem_page(struct huge_bootmem_page *m) +static inline void hugetlb_vmemmap_optimize_bootmem_page(unsigned long pfn, unsigned int order) { } #endif /* CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP */ diff --git a/mm/internal.h b/mm/internal.h index da833cafcd599d..5cc220db907668 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -23,13 +23,6 @@ #include "vma.h" struct folio_batch; -struct hstate; - -struct huge_bootmem_page { - struct list_head list; - struct hstate *hstate; - unsigned long flags; -}; /* mm/workingset.c */ bool workingset_test_recent(void *shadow, bool file, bool *workingset, From 728b1c355643e358f5d34754665627ebf6172d90 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Thu, 10 Sep 2026 14:32:56 +0800 Subject: [PATCH 0249/1012] mm/hugetlb: localize HUGE_BOOTMEM_ZONES_VALID HUGE_BOOTMEM_ZONES_VALID is only used by the huge_bootmem_page flag handling in mm/hugetlb.c. Keep the definition next to that private data structure instead of exposing it through the public hugetlb header. No functional change is intended. Link: https://lore.kernel.org/20260910063256.64386-18-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Qi Zheng Cc: David Hildenbrand (Arm) Cc: David Laight Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Oscar Salvador Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/hugetlb.h | 2 -- mm/hugetlb.c | 2 ++ 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 255a258f11d13c..900c95e346b2e3 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -675,8 +675,6 @@ struct hstate { char name[HSTATE_NAME_LEN]; }; -#define HUGE_BOOTMEM_ZONES_VALID BIT(0) - int isolate_or_dissolve_huge_folio(struct folio *folio, struct list_head *list); int replace_free_hugepage_folios(unsigned long start_pfn, unsigned long end_pfn); void wait_for_freed_hugetlb_folios(void); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index cdd061953ab130..74de391c39730d 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -55,6 +55,8 @@ #include "sparse.h" #include +#define HUGE_BOOTMEM_ZONES_VALID BIT(0) + struct huge_bootmem_page { struct list_head list; struct hstate *hstate; From bf371ec630c6b03113d8b02a36a2994b58fab5bf Mon Sep 17 00:00:00 2001 From: Song Hu Date: Tue, 25 Aug 2026 16:57:54 +0800 Subject: [PATCH 0250/1012] selftests/mm: emit TAP header in uffd-wp-mremap Patch series "selftests/mm: TAP output and global-state fixes", v4. uffd-wp-mremap and mremap_test never print the TAP header (and mremap_test skips with a bare exit(KSFT_SKIP) rather than a KTAP skip), so their output is not valid KTAP; hugetlb-soft-offline toggles enable_soft_offline during the run and leaves it disabled afterwards. This patch (of 3): uffd-wp-mremap calls ksft_set_plan() without ksft_print_header(), so its output is not valid KTAP. Add the header, like the sibling uffd tests (uffd-stress, uffd-unit-tests). Link: https://lore.kernel.org/20260825085756.63030-1-husong@kylinos.cn Link: https://lore.kernel.org/20260825085756.63030-2-husong@kylinos.cn Signed-off-by: Song Hu Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Reviewed-by: Sarthak Sharma Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Muhammad Usama Anjum Tested-by: Muhammad Usama Anjum Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Peter Xu Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/uffd-wp-mremap.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tools/testing/selftests/mm/uffd-wp-mremap.c b/tools/testing/selftests/mm/uffd-wp-mremap.c index c973d6722720c3..572c2516e874d7 100644 --- a/tools/testing/selftests/mm/uffd-wp-mremap.c +++ b/tools/testing/selftests/mm/uffd-wp-mremap.c @@ -347,6 +347,8 @@ int main(int argc, char **argv) struct thp_settings settings; int i, j, plan = 0; + ksft_print_header(); + hugepage_save_settings(true, true); check_uffd_wp_feature_supported(); From 6b2275412a1ab24b98ff0a8bf43f4a06e535a8a2 Mon Sep 17 00:00:00 2001 From: Song Hu Date: Tue, 25 Aug 2026 16:57:55 +0800 Subject: [PATCH 0251/1012] selftests/mm: emit TAP header and use TAP skip in mremap_test mremap_test calls ksft_set_plan() without ksft_print_header(), and its get_mmap_min_addr() skip path uses a bare exit(KSFT_SKIP) that prints no TAP line, so its output is not valid KTAP. Add the header and switch the skip to ksft_exit_skip(). Also fix two more KTAP compliance issues spotted in review: - get_mmap_min_addr() calls strerror(errno) after fclose(), which may clobber errno; save errno before fclose() instead. - Some ksft_*() messages embed "\n\t", so the text after each embedded newline is printed without the "# " prefix. Split those into separate messages. And cache mmap_min_addr in main() before ksft_set_plan(), so that the skip paths in get_mmap_min_addr() are taken before the plan is set; a skip after the plan leaves the run with fewer tests than planned. Link: https://lore.kernel.org/20260825085756.63030-3-husong@kylinos.cn Signed-off-by: Song Hu Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Reviewed-by: Sarthak Sharma Reviewed-by: Muhammad Usama Anjum Tested-by: Muhammad Usama Anjum Acked-by: Lorenzo Stoakes (ARM) Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Michal Hocko Cc: Peter Xu Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/mremap_test.c | 43 ++++++++++++++---------- 1 file changed, 25 insertions(+), 18 deletions(-) diff --git a/tools/testing/selftests/mm/mremap_test.c b/tools/testing/selftests/mm/mremap_test.c index 779ef2d5f1b9f7..97abf4713cc595 100644 --- a/tools/testing/selftests/mm/mremap_test.c +++ b/tools/testing/selftests/mm/mremap_test.c @@ -111,18 +111,17 @@ static unsigned long long get_mmap_min_addr(void) return addr; fp = fopen("/proc/sys/vm/mmap_min_addr", "r"); - if (fp == NULL) { - ksft_print_msg("Failed to open /proc/sys/vm/mmap_min_addr: %s\n", - strerror(errno)); - exit(KSFT_SKIP); - } + if (!fp) + ksft_exit_skip("Failed to open /proc/sys/vm/mmap_min_addr: %s\n", + strerror(errno)); n_matched = fscanf(fp, "%llu", &addr); if (n_matched != 1) { - ksft_print_msg("Failed to read /proc/sys/vm/mmap_min_addr: %s\n", - strerror(errno)); + int err = errno; + fclose(fp); - exit(KSFT_SKIP); + ksft_exit_skip("Failed to read /proc/sys/vm/mmap_min_addr: %s\n", + strerror(err)); } fclose(fp); @@ -1165,10 +1164,11 @@ static void run_mremap_test_case(struct test test_case, int *failures, rand_addr); if (remap_time < 0) { - if (test_case.expect_failure) - ksft_test_result_xfail("%s\n\tExpected mremap failure\n", - test_case.name); - else { + if (test_case.expect_failure) { + ksft_print_msg("%s: expected mremap failure\n", + test_case.name); + ksft_test_result_xfail("%s\n", test_case.name); + } else { ksft_test_result_fail("%s\n", test_case.name); *failures += 1; } @@ -1178,11 +1178,13 @@ static void run_mremap_test_case(struct test test_case, int *failures, * was faulted in. */ if (threshold_mb == VALIDATION_NO_THRESHOLD || - test_case.config.region_size <= threshold_mb * _1MB) - ksft_test_result_pass("%s\n\tmremap time: %12lldns\n", - test_case.name, remap_time); - else + test_case.config.region_size <= threshold_mb * _1MB) { + ksft_print_msg("%s: mremap time: %12lldns\n", + test_case.name, remap_time); ksft_test_result_pass("%s\n", test_case.name); + } else { + ksft_test_result_pass("%s\n", test_case.name); + } } } @@ -1251,13 +1253,18 @@ int main(int argc, char **argv) time_t t; FILE *maps_fp; + ksft_print_header(); + + get_mmap_min_addr(); + pattern_seed = (unsigned int) time(&t); if (parse_args(argc, argv, &threshold_mb, &pattern_seed) < 0) exit(EXIT_FAILURE); - ksft_print_msg("Test configs:\n\tthreshold_mb=%u\n\tpattern_seed=%u\n\n", - threshold_mb, pattern_seed); + ksft_print_msg("Test configs:\n"); + ksft_print_msg("threshold_mb=%u\n", threshold_mb); + ksft_print_msg("pattern_seed=%u\n", pattern_seed); /* * set preallocated random array according to test configs; see the From 4a126c9780653cf0863363ccf063c92a3e355c78 Mon Sep 17 00:00:00 2001 From: Song Hu Date: Tue, 25 Aug 2026 16:57:56 +0800 Subject: [PATCH 0252/1012] selftests/mm: restore enable_soft_offline in hugetlb-soft-offline hugetlb-soft-offline toggles /proc/sys/vm/enable_soft_offline between 1 and 0 (test_soft_offline_common(1) then (0)) and leaves it at 0 when it finishes, silently disabling soft offlining for the whole system after the run. Save the original value before the test and restore it from an atexit() handler, as hugepage_restore_settings_atexit() in hugepage_settings.c already does. Use read_num()/write_num() from vm_util instead of hand-rolled popen()/fopen() helpers. The restore handler must not call write_num(): on failure it re-enters exit() through ksft_exit_fail_msg(), which is undefined behavior from inside an atexit handler. A non-root run hits it directly - the restore write fails the same way the write that triggered the exit did. Restore with plain open()/write(), best effort. Link: https://lore.kernel.org/20260825085756.63030-4-husong@kylinos.cn Signed-off-by: Song Hu Signed-off-by: Andrew Morton Reviewed-by: Muhammad Usama Anjum Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Peter Xu Cc: Sarthak Sharma Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- .../selftests/mm/hugetlb-soft-offline.c | 49 +++++++++++-------- 1 file changed, 28 insertions(+), 21 deletions(-) diff --git a/tools/testing/selftests/mm/hugetlb-soft-offline.c b/tools/testing/selftests/mm/hugetlb-soft-offline.c index bc202e4ed2bda7..4af9d3db7b5b6f 100644 --- a/tools/testing/selftests/mm/hugetlb-soft-offline.c +++ b/tools/testing/selftests/mm/hugetlb-soft-offline.c @@ -11,6 +11,7 @@ #define _GNU_SOURCE #include +#include #include #include #include @@ -23,6 +24,7 @@ #include #include "kselftest.h" +#include "vm_util.h" #include "hugepage_settings.h" #ifndef MADV_SOFT_OFFLINE @@ -31,6 +33,8 @@ #define EPREFIX " !!! " +#define ENABLE_SOFT_OFFLINE_PATH "/proc/sys/vm/enable_soft_offline" + static int do_soft_offline(int fd, size_t len, int expect_errno) { char *filemap = NULL; @@ -77,26 +81,29 @@ static int do_soft_offline(int fd, size_t len, int expect_errno) return ret; } -static int set_enable_soft_offline(int value) -{ - char cmd[256] = {0}; - FILE *cmdfile = NULL; - - if (value != 0 && value != 1) - return -EINVAL; +static unsigned long orig_enable_soft_offline = -1UL; - sprintf(cmd, "echo %d > /proc/sys/vm/enable_soft_offline", value); - cmdfile = popen(cmd, "r"); +/* + * Runs from an atexit handler, so it must not call anything that + * exits on failure: write_num() would re-enter exit() through + * ksft_exit_fail_msg(). + */ +static void restore_enable_soft_offline(void) +{ + char buf[24]; + int fd, len; - if (cmdfile) - ksft_print_msg("enable_soft_offline => %d\n", value); - else { - ksft_perror(EPREFIX "failed to set enable_soft_offline"); - return errno; - } + if (orig_enable_soft_offline == -1UL) + return; - pclose(cmdfile); - return 0; + len = snprintf(buf, sizeof(buf), "%lu", orig_enable_soft_offline); + fd = open(ENABLE_SOFT_OFFLINE_PATH, O_WRONLY); + if (fd < 0) + return; + if (write(fd, buf, len) != len) + ksft_print_msg("failed to restore enable_soft_offline: %s\n", + strerror(errno)); + close(fd); } static int create_hugetlbfs_file(struct statfs *file_stat) @@ -145,10 +152,7 @@ static void test_soft_offline_common(int enable_soft_offline) hugepagesize_kb = file_stat.f_bsize / 1024; ksft_print_msg("Hugepagesize is %ldkB\n", hugepagesize_kb); - if (set_enable_soft_offline(enable_soft_offline) != 0) { - close(fd); - ksft_exit_fail_msg("Failed to set enable_soft_offline\n"); - } + write_num(ENABLE_SOFT_OFFLINE_PATH, enable_soft_offline); nr_hugepages_before = hugetlb_nr_default_pages(); @@ -192,6 +196,9 @@ int main(int argc, char **argv) ksft_set_plan(2); + orig_enable_soft_offline = read_num(ENABLE_SOFT_OFFLINE_PATH); + atexit(restore_enable_soft_offline); + test_soft_offline_common(1); test_soft_offline_common(0); From a31b4968a76343ba2f361c51ad8e9271ceddf50f Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Sun, 23 Aug 2026 12:40:52 +0800 Subject: [PATCH 0253/1012] mm/hugetlb: warn instead of silently bailing gigantic pages without runtime support remove_hugetlb_folio() and __update_and_free_hugetlb_folio() silently return for gigantic hstates that lack runtime freeing support. All callers should already filter such hstates upstream, so turn the silent bail into a VM_WARN_ON_ONCE to catch caller regressions instead of masking them. Link: https://lore.kernel.org/20260823044118.1097121-3-xialonglong2025@163.com Signed-off-by: Longlong Xia Signed-off-by: Andrew Morton Acked-by: Muchun Song Assisted-by: Codex:gpt-5.6-sol Cc: David Hildenbrand Cc: Miaohe Lin Cc: Michal Hocko Cc: Oscar Salvador --- mm/hugetlb.c | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 74de391c39730d..03182cc28a7dbc 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -1396,8 +1396,11 @@ void remove_hugetlb_folio(struct hstate *h, struct folio *folio, VM_BUG_ON_FOLIO(hugetlb_cgroup_from_folio_rsvd(folio), folio); lockdep_assert_held(&hugetlb_lock); - if (hstate_is_gigantic_no_runtime(h)) + if (hstate_is_gigantic_no_runtime(h)) { + /* Callers must filter gigantic_no_runtime upstream. */ + VM_WARN_ON_ONCE(1); return; + } list_del(&folio->lru); @@ -1458,8 +1461,11 @@ static void __update_and_free_hugetlb_folio(struct hstate *h, { bool clear_flag = folio_test_hugetlb_vmemmap_optimized(folio); - if (hstate_is_gigantic_no_runtime(h)) + if (hstate_is_gigantic_no_runtime(h)) { + /* Callers must filter gigantic_no_runtime upstream. */ + VM_WARN_ON_ONCE(1); return; + } /* * If we don't know which subpages are hwpoisoned, we can't free From 0739bcc8dcf5bb6d984f344c23c6aea005eb521a Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 12:28:27 +0300 Subject: [PATCH 0254/1012] set_memory: add number of pages parameter to set_direct_map APIs Patch series "arch, mm/execmem: resolve confusion about set_direct_map_valid_noflush()", v3. Recent discussion about implementation of execmem's ROX caches on arm64 revealed a confusion about how set_direct_map_valid_noflush() implemented on different architectures. On arm64 it sets or clears the PTE_VALID bit marking a PTE as present or not present. On other architectures it's a range version of set_direct_map_invalid_noflush() and set_direct_map_default_noflush() Unlike arm64::set_direct_map_valid_noflush(), set_direct_map_default_noflush() not only marks PTE as present, but also sets its default protection mode. Other than that, initial design of execmem ROX caches didn't rely on restoration of large mappings that's now available on x86, but completely removed the memory allocated for the ROX cache from the direct map to ensure that large mappings are not split. This precluded usage of VM_FLUSH_RESET_PERMS for the ROX cache allocations and required execmem to implement manipulation of the direct map alias. Current implementation of ROX caches does not remove the direct map alias but simply calls set_memory_rox() that updates the permissions in both vmalloc address space and the direct map and relies on collapse_large_pages() in x86 CPA to keep large mappings. This allow using VM_FLUSH_RESET_PERMS for execmem ROX cache allocations with small adjustments to set_direct_map APIs and vmalloc::reset_perms() behaviour: adding number of pages parameter to set_direct_map APIs and making resetting of the direct map permissions in vmalloc VMAP_HUGE friendly. Implement these adjustments, make execmem always use VM_FLUSH_RESET_PERMS and revert set_direct_map_valid_noflush() changes. This patch (of 6): When set_direct_map APIs were introduced by the commit d253ca0c3865 ("x86/mm/cpa: Add set_direct_map_*() functions") the single page parameter made sense because the initial callers (vmalloc and hibernation) had sets of unsorted struct pages that required changes of their mappings in the direct map. Since there is an increasing demand for direct map manipulation and it is also desirable to be able to update larger physically contiguous mappings, for example an entire large folio, extend set_direct_map APIs to receive number of pages parameter. As there is still only a handful of callers, change the existing functions directly and update all the call sites rather than adding wrappers for single page case. Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-0-949b64a9f755@kernel.org Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-1-949b64a9f755@kernel.org Link: https://lore.kernel.org/all/20260611130144.1385343-4-abarnas@google.com [1] Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Kevin Brodsky Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: "Uladzislau Rezki (Sony)" Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Cc: Dev Jain --- arch/arm64/include/asm/set_memory.h | 4 ++-- arch/arm64/mm/pageattr.c | 8 ++++---- arch/loongarch/include/asm/set_memory.h | 4 ++-- arch/loongarch/mm/pageattr.c | 8 ++++---- arch/riscv/include/asm/set_memory.h | 4 ++-- arch/riscv/mm/pageattr.c | 8 ++++---- arch/s390/include/asm/set_memory.h | 4 ++-- arch/s390/mm/pageattr.c | 8 ++++---- arch/x86/include/asm/set_memory.h | 4 ++-- arch/x86/mm/pat/set_memory.c | 8 ++++---- include/linux/set_memory.h | 6 ++++-- kernel/power/snapshot.c | 4 ++-- mm/secretmem.c | 6 +++--- mm/vmalloc.c | 5 +++-- 14 files changed, 42 insertions(+), 39 deletions(-) diff --git a/arch/arm64/include/asm/set_memory.h b/arch/arm64/include/asm/set_memory.h index 90f61b17275e1b..b07fd4e026eac0 100644 --- a/arch/arm64/include/asm/set_memory.h +++ b/arch/arm64/include/asm/set_memory.h @@ -11,8 +11,8 @@ bool can_set_direct_map(void); int set_memory_valid(unsigned long addr, int numpages, int enable); -int set_direct_map_invalid_noflush(struct page *page); -int set_direct_map_default_noflush(struct page *page); +int set_direct_map_invalid_noflush(struct page *page, unsigned int numpages); +int set_direct_map_default_noflush(struct page *page, unsigned int numpages); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); diff --git a/arch/arm64/mm/pageattr.c b/arch/arm64/mm/pageattr.c index bbe98ac9ad8c67..db8d60a84d1441 100644 --- a/arch/arm64/mm/pageattr.c +++ b/arch/arm64/mm/pageattr.c @@ -251,7 +251,7 @@ int set_memory_valid(unsigned long addr, int numpages, int enable) __pgprot(PTE_PRESENT_VALID_KERNEL)); } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int numpages) { pgprot_t clear_mask = __pgprot(PTE_PRESENT_VALID_KERNEL); pgprot_t set_mask = __pgprot(PTE_PRESENT_INVALID); @@ -260,10 +260,10 @@ int set_direct_map_invalid_noflush(struct page *page) return 0; return update_range_prot((unsigned long)page_address(page), - PAGE_SIZE, set_mask, clear_mask); + PAGE_SIZE * numpages, set_mask, clear_mask); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int numpages) { pgprot_t set_mask = __pgprot(PTE_PRESENT_VALID_KERNEL | PTE_WRITE); pgprot_t clear_mask = __pgprot(PTE_PRESENT_INVALID | PTE_RDONLY); @@ -272,7 +272,7 @@ int set_direct_map_default_noflush(struct page *page) return 0; return update_range_prot((unsigned long)page_address(page), - PAGE_SIZE, set_mask, clear_mask); + PAGE_SIZE * numpages, set_mask, clear_mask); } static int __set_memory_enc_dec(unsigned long addr, diff --git a/arch/loongarch/include/asm/set_memory.h b/arch/loongarch/include/asm/set_memory.h index 55dfaefd02c8a6..563aab92896e9b 100644 --- a/arch/loongarch/include/asm/set_memory.h +++ b/arch/loongarch/include/asm/set_memory.h @@ -15,8 +15,8 @@ int set_memory_ro(unsigned long addr, int numpages); int set_memory_rw(unsigned long addr, int numpages); bool kernel_page_present(struct page *page); -int set_direct_map_default_noflush(struct page *page); -int set_direct_map_invalid_noflush(struct page *page); +int set_direct_map_default_noflush(struct page *page, unsigned int nr); +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); #endif /* _ASM_LOONGARCH_SET_MEMORY_H */ diff --git a/arch/loongarch/mm/pageattr.c b/arch/loongarch/mm/pageattr.c index 614ccc7afccbea..43ad2a104f19df 100644 --- a/arch/loongarch/mm/pageattr.c +++ b/arch/loongarch/mm/pageattr.c @@ -198,24 +198,24 @@ bool kernel_page_present(struct page *page) return pte_present(ptep_get(pte)); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int nr) { unsigned long addr = (unsigned long)page_address(page); if (addr < vm_map_base) return 0; - return __set_memory(addr, 1, PAGE_KERNEL, __pgprot(0)); + return __set_memory(addr, nr, PAGE_KERNEL, __pgprot(0)); } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { unsigned long addr = (unsigned long)page_address(page); if (addr < vm_map_base) return 0; - return __set_memory(addr, 1, __pgprot(0), __pgprot(_PAGE_PRESENT | _PAGE_VALID)); + return __set_memory(addr, nr, __pgprot(0), __pgprot(_PAGE_PRESENT | _PAGE_VALID)); } int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) diff --git a/arch/riscv/include/asm/set_memory.h b/arch/riscv/include/asm/set_memory.h index ef59e1716a2cfd..db1d0ed82b6962 100644 --- a/arch/riscv/include/asm/set_memory.h +++ b/arch/riscv/include/asm/set_memory.h @@ -40,8 +40,8 @@ static inline int set_kernel_memory(char *startp, char *endp, } #endif -int set_direct_map_invalid_noflush(struct page *page); -int set_direct_map_default_noflush(struct page *page); +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); +int set_direct_map_default_noflush(struct page *page, unsigned int nr); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); diff --git a/arch/riscv/mm/pageattr.c b/arch/riscv/mm/pageattr.c index 3f76db3d276992..20ef95b1d0c36e 100644 --- a/arch/riscv/mm/pageattr.c +++ b/arch/riscv/mm/pageattr.c @@ -374,15 +374,15 @@ int set_memory_nx(unsigned long addr, int numpages) return __set_memory(addr, numpages, __pgprot(0), __pgprot(_PAGE_EXEC)); } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { - return __set_memory((unsigned long)page_address(page), 1, + return __set_memory((unsigned long)page_address(page), nr, __pgprot(0), __pgprot(_PAGE_PRESENT)); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int nr) { - return __set_memory((unsigned long)page_address(page), 1, + return __set_memory((unsigned long)page_address(page), nr, PAGE_KERNEL, __pgprot(_PAGE_EXEC)); } diff --git a/arch/s390/include/asm/set_memory.h b/arch/s390/include/asm/set_memory.h index 94092f4ae76499..6b0aa9147ed8e6 100644 --- a/arch/s390/include/asm/set_memory.h +++ b/arch/s390/include/asm/set_memory.h @@ -60,8 +60,8 @@ __SET_MEMORY_FUNC(set_memory_rox, SET_MEMORY_RO | SET_MEMORY_X) __SET_MEMORY_FUNC(set_memory_rwnx, SET_MEMORY_RW | SET_MEMORY_NX) __SET_MEMORY_FUNC(set_memory_4k, SET_MEMORY_4K) -int set_direct_map_invalid_noflush(struct page *page); -int set_direct_map_default_noflush(struct page *page); +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); +int set_direct_map_default_noflush(struct page *page, unsigned int nr); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); diff --git a/arch/s390/mm/pageattr.c b/arch/s390/mm/pageattr.c index 1e202e3d08e75f..7549543d624125 100644 --- a/arch/s390/mm/pageattr.c +++ b/arch/s390/mm/pageattr.c @@ -382,14 +382,14 @@ int __set_memory(unsigned long addr, unsigned long numpages, unsigned long flags return rc; } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { - return __set_memory((unsigned long)page_to_virt(page), 1, SET_MEMORY_INV); + return __set_memory((unsigned long)page_to_virt(page), nr, SET_MEMORY_INV); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int nr) { - return __set_memory((unsigned long)page_to_virt(page), 1, SET_MEMORY_DEF); + return __set_memory((unsigned long)page_to_virt(page), nr, SET_MEMORY_DEF); } int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) diff --git a/arch/x86/include/asm/set_memory.h b/arch/x86/include/asm/set_memory.h index 4362c26aa992db..0c4235d159f483 100644 --- a/arch/x86/include/asm/set_memory.h +++ b/arch/x86/include/asm/set_memory.h @@ -86,8 +86,8 @@ int set_pages_wb(struct page *page, int numpages); int set_pages_ro(struct page *page, int numpages); int set_pages_rw(struct page *page, int numpages); -int set_direct_map_invalid_noflush(struct page *page); -int set_direct_map_default_noflush(struct page *page); +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); +int set_direct_map_default_noflush(struct page *page, unsigned int nr); int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); diff --git a/arch/x86/mm/pat/set_memory.c b/arch/x86/mm/pat/set_memory.c index 4652487b5572b0..d36a58ae94db64 100644 --- a/arch/x86/mm/pat/set_memory.c +++ b/arch/x86/mm/pat/set_memory.c @@ -2685,14 +2685,14 @@ static int __set_pages_np(struct page *page, int numpages, unsigned int cpa_flag return __change_page_attr_set_clr(&cpa, 1); } -int set_direct_map_invalid_noflush(struct page *page) +int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { - return __set_pages_np(page, 1, 0); + return __set_pages_np(page, nr, 0); } -int set_direct_map_default_noflush(struct page *page) +int set_direct_map_default_noflush(struct page *page, unsigned int nr) { - return __set_pages_p(page, 1, 0); + return __set_pages_p(page, nr, 0); } int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) diff --git a/include/linux/set_memory.h b/include/linux/set_memory.h index 3030d9245f5ac8..0b77f1d7d8b9ce 100644 --- a/include/linux/set_memory.h +++ b/include/linux/set_memory.h @@ -25,11 +25,13 @@ static inline int set_memory_rox(unsigned long addr, int numpages) #endif #ifndef CONFIG_ARCH_HAS_SET_DIRECT_MAP -static inline int set_direct_map_invalid_noflush(struct page *page) +static inline int set_direct_map_invalid_noflush(struct page *page, + unsigned int nr) { return 0; } -static inline int set_direct_map_default_noflush(struct page *page) +static inline int set_direct_map_default_noflush(struct page *page, + unsigned int nr) { return 0; } diff --git a/kernel/power/snapshot.c b/kernel/power/snapshot.c index b209712cb2c3ac..d5dba0e50b2eb6 100644 --- a/kernel/power/snapshot.c +++ b/kernel/power/snapshot.c @@ -88,7 +88,7 @@ static inline int hibernate_restore_unprotect_page(void *page_address) {return 0 static inline void hibernate_map_page(struct page *page) { if (IS_ENABLED(CONFIG_ARCH_HAS_SET_DIRECT_MAP)) { - int ret = set_direct_map_default_noflush(page); + int ret = set_direct_map_default_noflush(page, 1); if (ret) pr_warn_once("Failed to remap page\n"); @@ -101,7 +101,7 @@ static inline void hibernate_unmap_page(struct page *page) { if (IS_ENABLED(CONFIG_ARCH_HAS_SET_DIRECT_MAP)) { unsigned long addr = (unsigned long)page_address(page); - int ret = set_direct_map_invalid_noflush(page); + int ret = set_direct_map_invalid_noflush(page, 1); if (ret) pr_warn_once("Failed to remap page\n"); diff --git a/mm/secretmem.c b/mm/secretmem.c index 384f5cfc457f9e..6cbb8efc994a4d 100644 --- a/mm/secretmem.c +++ b/mm/secretmem.c @@ -139,7 +139,7 @@ static vm_fault_t secretmem_fault(struct vm_fault *vmf) goto out; } - err = set_direct_map_invalid_noflush(folio_page(folio, 0)); + err = set_direct_map_invalid_noflush(folio_page(folio, 0), 1); if (err) { secretmem_unaccount_folio(state, folio); folio_put(folio); @@ -156,7 +156,7 @@ static vm_fault_t secretmem_fault(struct vm_fault *vmf) * already happened when we marked the page invalid * which guarantees that this call won't fail */ - set_direct_map_default_noflush(folio_page(folio, 0)); + set_direct_map_default_noflush(folio_page(folio, 0), 1); folio_put(folio); if (err == -EEXIST) goto retry; @@ -228,7 +228,7 @@ static int secretmem_migrate_folio(struct address_space *mapping, static void secretmem_free_folio(struct folio *folio) { - set_direct_map_default_noflush(folio_page(folio, 0)); + set_direct_map_default_noflush(folio_page(folio, 0), 1); folio_zero_segment(folio, 0, folio_size(folio)); } diff --git a/mm/vmalloc.c b/mm/vmalloc.c index b879260d31a577..33ffc978ccbe8a 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3371,14 +3371,15 @@ struct vm_struct *remove_vm_area(const void *addr) } static inline void set_area_direct_map(const struct vm_struct *area, - int (*set_direct_map)(struct page *page)) + int (*set_direct_map)(struct page *page, + unsigned int nr)) { unsigned long i; /* HUGE_VMALLOC passes small pages to set_direct_map */ for (i = 0; i < area->nr_pages; i++) if (page_address(area->pages[i])) - set_direct_map(area->pages[i]); + set_direct_map(area->pages[i], 1); } /* From 52ea18910c4a4db427484b863cec3571d6918de7 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 12:28:28 +0300 Subject: [PATCH 0255/1012] mm/vmalloc: set area's page_order after allocation succeeds __vmalloc_area_node() calls set_vm_area_page_order() to set area's page_order before actually allocating pages to populate the area. If allocation of large pages in HUGE_VMAP case fails midway, this leaves the area with elevated page_order throughout the cleanup path. There is no actual issue with this because the only place that currently relies on area->page_order on the cleanup path is the loop calculating the direct map alias range in vm_reset_perms() and it anyway skips unpopulated pages. But having set_vm_area_page_order() in the middle of __vmalloc_area_node() makes things very obscure, hard to reason about and error prone against future changes of the cleanup path. Move the call to set_vm_area_page_order() just before the successful return from __vmalloc_area_node() where page order is guaranteed. While on it, initialize local page_order variable with its declaration. Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-2-949b64a9f755@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Reviewed-by: Dev Jain Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Cc: Kevin Brodsky --- mm/vmalloc.c | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 33ffc978ccbe8a..67c0fba2470f1b 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3886,7 +3886,7 @@ static void *__vmalloc_area_node(struct vm_struct *area, gfp_t gfp_mask, unsigned long size = get_vm_area_size(area); unsigned long array_size; unsigned long nr_small_pages = size >> PAGE_SHIFT; - unsigned int page_order; + unsigned int page_order = page_shift - PAGE_SHIFT; unsigned int flags; int ret; @@ -3914,9 +3914,6 @@ static void *__vmalloc_area_node(struct vm_struct *area, gfp_t gfp_mask, goto fail; } - set_vm_area_page_order(area, page_shift - PAGE_SHIFT); - page_order = vm_area_page_order(area); - /* * High-order nofail allocations are really expensive and * potentially dangerous (pre-mature OOM, disruptive reclaim @@ -3971,6 +3968,7 @@ static void *__vmalloc_area_node(struct vm_struct *area, gfp_t gfp_mask, goto fail; } + set_vm_area_page_order(area, page_order); return area->addr; fail: From 7c303f9127c3072e8003dd226eb7b0c4583eea06 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 12:28:29 +0300 Subject: [PATCH 0256/1012] mm/vmalloc: constify vm parameter of get_vm_area_page_order() get_vm_area_page_order() and vm_area_page_order() do not need to modify struct vm_struct passed to them. Constify the parameter. Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-3-949b64a9f755@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Reviewed-by: Dev Jain Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Kevin Brodsky Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon --- mm/vmalloc.c | 4 ++-- mm/vmalloc.h | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 67c0fba2470f1b..b6dd287a0be33b 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3137,7 +3137,7 @@ EXPORT_SYMBOL(vm_map_ram); static struct vm_struct *vmlist __initdata; -static inline unsigned int vm_area_page_order(struct vm_struct *vm) +static inline unsigned int vm_area_page_order(const struct vm_struct *vm) { #ifdef CONFIG_HAVE_ARCH_HUGE_VMALLOC return vm->page_order; @@ -3146,7 +3146,7 @@ static inline unsigned int vm_area_page_order(struct vm_struct *vm) #endif } -unsigned int get_vm_area_page_order(struct vm_struct *vm) +unsigned int get_vm_area_page_order(const struct vm_struct *vm) { return vm_area_page_order(vm); } diff --git a/mm/vmalloc.h b/mm/vmalloc.h index 8866ddcff6681b..211869f365095f 100644 --- a/mm/vmalloc.h +++ b/mm/vmalloc.h @@ -12,7 +12,7 @@ void __init vmalloc_init(void); int __must_check vmap_pages_range_noflush(unsigned long addr, unsigned long end, pgprot_t prot, struct page **pages, unsigned int page_shift, gfp_t gfp_mask); -unsigned int get_vm_area_page_order(struct vm_struct *vm); +unsigned int get_vm_area_page_order(const struct vm_struct *vm); #else static inline void vmalloc_init(void) {} From 2db5b0c5355f07fe90584bd7ed9a5032e2a7c3f1 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 12:28:30 +0300 Subject: [PATCH 0257/1012] mm/vmalloc: make set_area_direct_map HUGE_VMAP friendly set_area_direct_map() always updates direct map alias permissions in single page increments. For HUGE_VMAP areas it's suboptimal. Not only the loop in set_area_direct_map() needlessly has more iterations (e.g times 512 on x86), but it also causes fragmentation of the direct map that could be avoided for the HUGE_VMAP areas populated with large pages. All pages in an area are always of the same order: either same-order large pages when VM_ALLOW_HUGE_VMAP is set and all huge pages were successfully allocated, or order-0 page when VM_ALLOW_HUGE_VMAP is cleared or when huge pages allocation fails and fallback path is taken. Instead of updating the direct map permissions for every order-0 page in an area, use the area's page_order as the loop increment and update the large pages in one call to set_direct_map_{invalid,default}_noflush(). Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-4-949b64a9f755@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Dev Jain Reviewed-by: Kevin Brodsky Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: "Uladzislau Rezki (Sony)" Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon --- mm/vmalloc.c | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index b6dd287a0be33b..db357a9bdd1250 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3374,12 +3374,15 @@ static inline void set_area_direct_map(const struct vm_struct *area, int (*set_direct_map)(struct page *page, unsigned int nr)) { - unsigned long i; + unsigned int nr = (1U << vm_area_page_order(area)); + + for (unsigned long i = 0; i < area->nr_pages; i += nr) { + if (page_address(area->pages[i])) { + int err = set_direct_map(area->pages[i], nr); - /* HUGE_VMALLOC passes small pages to set_direct_map */ - for (i = 0; i < area->nr_pages; i++) - if (page_address(area->pages[i])) - set_direct_map(area->pages[i], 1); + WARN_ON_ONCE(err); + } + } } /* From 920ed835d223f737df3c0ded740af6c1b9101e95 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 12:28:31 +0300 Subject: [PATCH 0258/1012] mm/execmem: use VM_FLUSH_RESET_PERMS for ROX cache allocations Initially execmem completely removed direct map alias for the memory allocated for the ROX cache in PMD_SIZE chunks. When that memory was freed, its direct map was restored also in PMD_SIZE chunks to avoid fragmentation of the direct map caused by vmalloc::vm_reset_perms(). This required execmem to implement the wrappers for set_direct_map APIs for proper sequencing of removal and restoration of the direct map aliases. Since then x86's CPA gained support for collapsing the direct map page tables for ROX pages and execmem switched from removing ROX caches from the direct map to making them ROX there, so execmem only needs to update direct map alias permissions when freeing the ROX cache memory. vmalloc already handles those updates for areas with VM_FLUSH_RESET_PERMS set and vmalloc::vm_reset_perms() does not force split of the direct map for PMD_SIZE chunks. Set the area permissions with set_vm_flush_reset_perms() when populating the execmem cache just before flipping the area to ROX. This way freeing an allocated area on an error path won't incur two updates of the direct map alias of that area and TLB flushing in vm_reset_perms(). Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-5-949b64a9f755@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Kevin Brodsky Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Brendan Jackman Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: "Uladzislau Rezki (Sony)" Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Cc: Dev Jain --- mm/execmem.c | 40 +++++++--------------------------------- 1 file changed, 7 insertions(+), 33 deletions(-) diff --git a/mm/execmem.c b/mm/execmem.c index 74a178a87e7581..ad07cae9ed5854 100644 --- a/mm/execmem.c +++ b/mm/execmem.c @@ -113,28 +113,6 @@ static inline unsigned long mas_range_len(struct ma_state *mas) return mas->last - mas->index + 1; } -static int execmem_set_direct_map_valid(struct vm_struct *vm, bool valid) -{ - unsigned int nr = (1 << get_vm_area_page_order(vm)); - unsigned int updated = 0; - int err = 0; - - for (int i = 0; i < vm->nr_pages; i += nr) { - err = set_direct_map_valid_noflush(vm->pages[i], nr, valid); - if (err) - goto err_restore; - updated += nr; - } - - return 0; - -err_restore: - for (int i = 0; i < updated; i += nr) - set_direct_map_valid_noflush(vm->pages[i], nr, !valid); - - return err; -} - static int execmem_force_rw(void *ptr, size_t size) { unsigned int nr = PAGE_ALIGN(size) >> PAGE_SHIFT; @@ -169,9 +147,6 @@ static void execmem_cache_clean(struct work_struct *work) if (IS_ALIGNED(size, PMD_SIZE) && IS_ALIGNED(mas.index, PMD_SIZE)) { - struct vm_struct *vm = find_vm_area(area); - - execmem_set_direct_map_valid(vm, true); mas_store_gfp(&mas, NULL, GFP_KERNEL); vfree(area); } @@ -301,6 +276,8 @@ static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t si /* fill memory with instructions that will trap */ execmem_fill_trapping_insns(p, alloc_size); + set_vm_flush_reset_perms(p); + err = set_memory_rox((unsigned long)p, vm->nr_pages); if (err) goto err_free_mem; @@ -312,18 +289,15 @@ static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t si */ mutex_lock(mutex); err = execmem_cache_add_locked(p, alloc_size, GFP_KERNEL); - if (err) - goto err_reset_direct_map; - - p = execmem_cache_alloc_locked(range, size); - + if (!err) + p = execmem_cache_alloc_locked(range, size); mutex_unlock(mutex); + if (err) + goto err_free_mem; + return p; -err_reset_direct_map: - mutex_unlock(mutex); - execmem_set_direct_map_valid(vm, true); err_free_mem: vfree(p); return NULL; From 5bf66973554d1906e609bba0ea74cb8d50ecd74b Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 12:28:32 +0300 Subject: [PATCH 0259/1012] Revert "arch: introduce set_direct_map_valid_noflush()" Commit 0c6378a71574 ("arch: introduce set_direct_map_valid_noflush()") added set_direct_map_valid_noflush() to allow updating the direct map for a physically contiguous range in execmem. As Brendan recently pointed out [1], this API is confusing because on arm64 it means that is sets VALID bit in ptes, while on other architectures it is an analog of set_direct_map_default_noflush(). The only user of set_direct_map_valid_noflush() was execmem's ROX cache freeing path and it was switched to utilize VM_FLUSH_RESET_PERMS for resetting permissions of the direct map alias. With the last user gone and with set_direct_map_{invalid,default}_noflush() accepting number of pages as a parameter, set_direct_map_valid_noflush() become a copy of set_memory_valid() on arm64 and a duplicate of set_direct_map_{invalid,default}_noflush() on other architecture, it is safe to remove set_direct_map_valid_noflush(). Also drop a stale comment in arm64::__kernel_map_pages() that Linus bothered to add when merging changes containing set_direct_map_valid_noflush() to his tree. This reverts commit 0c6378a71574daa6cd1534ad42a956e3262756c7. Link: https://lore.kernel.org/20260903-execmem-set-vm-perms-v0-2-v3-6-949b64a9f755@kernel.org Link: https://lore.kernel.org/all/DJ69RCVRBO0Y.3JCYSW50IC4RC@linux.dev [1] Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Brendan Jackman Acked-by: David Hildenbrand (Arm) Reviewed-by: Kevin Brodsky Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Andy Lutomirski Cc: "Borislav Petkov (AMD)" Cc: Catalin Marinas Cc: Christian Borntraeger Cc: Dave Hansen Cc: Gerald Schaefer Cc: Heiko Carstens Cc: "H. Peter Anvin" Cc: Huacai Chen Cc: Ingo Molnar Cc: Len Brown Cc: Palmer Dabbelt Cc: Peter Zijlstra Cc: "Rafael J. Wysocki" Cc: Ryan Roberts Cc: Sven Schnelle Cc: "Uladzislau Rezki (Sony)" Cc: Vasily Gorbik Cc: WANG Xuerui Cc: Will Deacon Cc: Dev Jain --- arch/arm64/include/asm/set_memory.h | 1 - arch/arm64/mm/pageattr.c | 16 ---------------- arch/loongarch/include/asm/set_memory.h | 1 - arch/loongarch/mm/pageattr.c | 19 ------------------- arch/riscv/include/asm/set_memory.h | 1 - arch/riscv/mm/pageattr.c | 15 --------------- arch/s390/include/asm/set_memory.h | 1 - arch/s390/mm/pageattr.c | 12 ------------ arch/x86/include/asm/set_memory.h | 1 - arch/x86/mm/pat/set_memory.c | 8 -------- include/linux/set_memory.h | 6 ------ 11 files changed, 81 deletions(-) diff --git a/arch/arm64/include/asm/set_memory.h b/arch/arm64/include/asm/set_memory.h index b07fd4e026eac0..0091ba12200e68 100644 --- a/arch/arm64/include/asm/set_memory.h +++ b/arch/arm64/include/asm/set_memory.h @@ -13,7 +13,6 @@ int set_memory_valid(unsigned long addr, int numpages, int enable); int set_direct_map_invalid_noflush(struct page *page, unsigned int numpages); int set_direct_map_default_noflush(struct page *page, unsigned int numpages); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); int set_memory_encrypted(unsigned long addr, int numpages); diff --git a/arch/arm64/mm/pageattr.c b/arch/arm64/mm/pageattr.c index db8d60a84d1441..132938b32eb16f 100644 --- a/arch/arm64/mm/pageattr.c +++ b/arch/arm64/mm/pageattr.c @@ -355,23 +355,7 @@ int realm_register_memory_enc_ops(void) return arm64_mem_crypt_ops_register(&realm_crypt_ops); } -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - unsigned long addr = (unsigned long)page_address(page); - - if (!can_set_direct_map()) - return 0; - - return set_memory_valid(addr, nr, valid); -} - #ifdef CONFIG_DEBUG_PAGEALLOC -/* - * This is - apart from the return value - doing the same - * thing as the new set_direct_map_valid_noflush() function. - * - * Unify? Explain the conceptual differences? - */ void __kernel_map_pages(struct page *page, int numpages, int enable) { if (!can_set_direct_map()) diff --git a/arch/loongarch/include/asm/set_memory.h b/arch/loongarch/include/asm/set_memory.h index 563aab92896e9b..4bb01172fbc245 100644 --- a/arch/loongarch/include/asm/set_memory.h +++ b/arch/loongarch/include/asm/set_memory.h @@ -17,6 +17,5 @@ int set_memory_rw(unsigned long addr, int numpages); bool kernel_page_present(struct page *page); int set_direct_map_default_noflush(struct page *page, unsigned int nr); int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); #endif /* _ASM_LOONGARCH_SET_MEMORY_H */ diff --git a/arch/loongarch/mm/pageattr.c b/arch/loongarch/mm/pageattr.c index 43ad2a104f19df..a7dcff40f75982 100644 --- a/arch/loongarch/mm/pageattr.c +++ b/arch/loongarch/mm/pageattr.c @@ -217,22 +217,3 @@ int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) return __set_memory(addr, nr, __pgprot(0), __pgprot(_PAGE_PRESENT | _PAGE_VALID)); } - -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - unsigned long addr = (unsigned long)page_address(page); - pgprot_t set, clear; - - if (addr < vm_map_base) - return 0; - - if (valid) { - set = PAGE_KERNEL; - clear = __pgprot(0); - } else { - set = __pgprot(0); - clear = __pgprot(_PAGE_PRESENT | _PAGE_VALID); - } - - return __set_memory(addr, nr, set, clear); -} diff --git a/arch/riscv/include/asm/set_memory.h b/arch/riscv/include/asm/set_memory.h index db1d0ed82b6962..e9f9960c194772 100644 --- a/arch/riscv/include/asm/set_memory.h +++ b/arch/riscv/include/asm/set_memory.h @@ -42,7 +42,6 @@ static inline int set_kernel_memory(char *startp, char *endp, int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); int set_direct_map_default_noflush(struct page *page, unsigned int nr); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); #endif /* __ASSEMBLER__ */ diff --git a/arch/riscv/mm/pageattr.c b/arch/riscv/mm/pageattr.c index 20ef95b1d0c36e..5b3cf326455db1 100644 --- a/arch/riscv/mm/pageattr.c +++ b/arch/riscv/mm/pageattr.c @@ -386,21 +386,6 @@ int set_direct_map_default_noflush(struct page *page, unsigned int nr) PAGE_KERNEL, __pgprot(_PAGE_EXEC)); } -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - pgprot_t set, clear; - - if (valid) { - set = PAGE_KERNEL; - clear = __pgprot(_PAGE_EXEC); - } else { - set = __pgprot(0); - clear = __pgprot(_PAGE_PRESENT); - } - - return __set_memory((unsigned long)page_address(page), nr, set, clear); -} - #ifdef CONFIG_DEBUG_PAGEALLOC static int debug_pagealloc_set_page(pte_t *pte, unsigned long addr, void *data) { diff --git a/arch/s390/include/asm/set_memory.h b/arch/s390/include/asm/set_memory.h index 6b0aa9147ed8e6..e3562bf0c1aa5e 100644 --- a/arch/s390/include/asm/set_memory.h +++ b/arch/s390/include/asm/set_memory.h @@ -62,7 +62,6 @@ __SET_MEMORY_FUNC(set_memory_4k, SET_MEMORY_4K) int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); int set_direct_map_default_noflush(struct page *page, unsigned int nr); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); #endif diff --git a/arch/s390/mm/pageattr.c b/arch/s390/mm/pageattr.c index 7549543d624125..80e834e8b8e1b5 100644 --- a/arch/s390/mm/pageattr.c +++ b/arch/s390/mm/pageattr.c @@ -392,18 +392,6 @@ int set_direct_map_default_noflush(struct page *page, unsigned int nr) return __set_memory((unsigned long)page_to_virt(page), nr, SET_MEMORY_DEF); } -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - unsigned long flags; - - if (valid) - flags = SET_MEMORY_DEF; - else - flags = SET_MEMORY_INV; - - return __set_memory((unsigned long)page_to_virt(page), nr, flags); -} - bool kernel_page_present(struct page *page) { unsigned long addr; diff --git a/arch/x86/include/asm/set_memory.h b/arch/x86/include/asm/set_memory.h index 0c4235d159f483..39271a5ea92527 100644 --- a/arch/x86/include/asm/set_memory.h +++ b/arch/x86/include/asm/set_memory.h @@ -88,7 +88,6 @@ int set_pages_rw(struct page *page, int numpages); int set_direct_map_invalid_noflush(struct page *page, unsigned int nr); int set_direct_map_default_noflush(struct page *page, unsigned int nr); -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid); bool kernel_page_present(struct page *page); extern int kernel_set_to_readonly; diff --git a/arch/x86/mm/pat/set_memory.c b/arch/x86/mm/pat/set_memory.c index d36a58ae94db64..7b6d983088ea9a 100644 --- a/arch/x86/mm/pat/set_memory.c +++ b/arch/x86/mm/pat/set_memory.c @@ -2695,14 +2695,6 @@ int set_direct_map_default_noflush(struct page *page, unsigned int nr) return __set_pages_p(page, nr, 0); } -int set_direct_map_valid_noflush(struct page *page, unsigned nr, bool valid) -{ - if (valid) - return __set_pages_p(page, nr, 0); - - return __set_pages_np(page, nr, 0); -} - #ifdef CONFIG_DEBUG_PAGEALLOC void __kernel_map_pages(struct page *page, int numpages, int enable) { diff --git a/include/linux/set_memory.h b/include/linux/set_memory.h index 0b77f1d7d8b9ce..3fe293cfed8cc8 100644 --- a/include/linux/set_memory.h +++ b/include/linux/set_memory.h @@ -36,12 +36,6 @@ static inline int set_direct_map_default_noflush(struct page *page, return 0; } -static inline int set_direct_map_valid_noflush(struct page *page, - unsigned nr, bool valid) -{ - return 0; -} - static inline bool kernel_page_present(struct page *page) { return true; From 91b4ebc8b2b8cb42828c0a1481ff52e7c0fd9481 Mon Sep 17 00:00:00 2001 From: Hao Jia Date: Fri, 28 Aug 2026 16:31:49 +0800 Subject: [PATCH 0260/1012] zram: fix idle age_sec underflow in idle_store() After commit 2e8ff2f51dde ("zram: use u32 for entry ac_time tracking"), idle_store() computes the idle cutoff as: cutoff = ktime_sub((u32)ktime_get_boottime_seconds(), age_sec); Because the left operand is cast to u32, when age_sec exceeds the current uptime the subtraction wraps modulo 2^32 and the huge result is zero-extended into the s64 cutoff. mark_idle() then marks every entry as idle instead of matching nothing. For instance, running echo 86400 > /sys/block/zramX/idle on a machine up for only two minutes marks all newly written pages idle and hands them to idle writeback and recompression. No slot can have been accessed before the system booted, so an age_sec that reaches back past uptime cannot match any slot. Return early in that case, without walking the table or taking any slot locks. Track the cutoff as time64_t rather than ktime_t. Both cutoff and ac_time are boot-time values in seconds, so a plain arithmetic comparison against ac_time in mark_idle() is correct and no ktime helpers are needed. Link: https://lore.kernel.org/20260828083149.45760-1-jiahao.kernel@gmail.com Fixes: 2e8ff2f51dde ("zram: use u32 for entry ac_time tracking") Signed-off-by: Hao Jia Signed-off-by: Andrew Morton Suggested-by: Sergey Senozhatsky Cc: Brian Geffon Cc: Jens Axboe Cc: Minchan Kim Cc: --- drivers/block/zram/zram_drv.c | 21 +++++++++++++-------- 1 file changed, 13 insertions(+), 8 deletions(-) diff --git a/drivers/block/zram/zram_drv.c b/drivers/block/zram/zram_drv.c index a9b3bb1d3bef35..4ba0f77b2abd80 100644 --- a/drivers/block/zram/zram_drv.c +++ b/drivers/block/zram/zram_drv.c @@ -415,7 +415,7 @@ static ssize_t mem_used_max_store(struct device *dev, * Mark all pages which are older than or equal to cutoff as IDLE. * Callers should hold the zram init lock in read mode */ -static void mark_idle(struct zram *zram, ktime_t cutoff) +static void mark_idle(struct zram *zram, time64_t cutoff) { int is_idle = 1; unsigned long nr_pages = zram->disksize >> PAGE_SHIFT; @@ -439,7 +439,7 @@ static void mark_idle(struct zram *zram, ktime_t cutoff) #ifdef CONFIG_ZRAM_TRACK_ENTRY_ACTIME is_idle = !cutoff || - ktime_after(cutoff, zram->table[index].attr.ac_time); + cutoff > zram->table[index].attr.ac_time; #endif if (is_idle) set_slot_flag(zram, index, ZRAM_IDLE); @@ -453,21 +453,26 @@ static ssize_t idle_store(struct device *dev, struct device_attribute *attr, const char *buf, size_t len) { struct zram *zram = dev_to_zram(dev); - ktime_t cutoff = 0; + time64_t cutoff = 0; if (!sysfs_streq(buf, "all")) { /* * If it did not parse as 'all' try to treat it as an integer * when we have memory tracking enabled. */ + time64_t uptime; u32 age_sec; - if (IS_ENABLED(CONFIG_ZRAM_TRACK_ENTRY_ACTIME) && - !kstrtouint(buf, 0, &age_sec)) - cutoff = ktime_sub((u32)ktime_get_boottime_seconds(), - age_sec); - else + if (!IS_ENABLED(CONFIG_ZRAM_TRACK_ENTRY_ACTIME) || + kstrtouint(buf, 0, &age_sec)) return -EINVAL; + + /* No slot can be older than the system uptime */ + uptime = ktime_get_boottime_seconds(); + if (age_sec >= uptime) + return len; + + cutoff = uptime - age_sec; } guard(rwsem_read)(&zram->dev_lock); From c2c078effd7295b39d9bfab6552f5c28d918605b Mon Sep 17 00:00:00 2001 From: Vernon Yang Date: Wed, 9 Sep 2026 10:58:01 +0800 Subject: [PATCH 0261/1012] mm: khugepaged: fix swap entry value to folio_pfn() Patch series "mm: khugepaged: fix tracepoint UAF", v5. The khugepaged tracepoints take a folio pointer and call folio_pfn(), but by then the folio may no longer be valid: freed after folio_put(), folio_unlock() or pte_unmap_unlock(), or not a folio at all but an xarray-encoded swap entry. On classic SPARSEMEM, dereferencing it oopses khugepaged as soon as the trace event is enabled; on other memory models it merely prints a bogus pfn. Pass the pfn to the tracepoints directly, captured while the folio is still pinned, closing the use-after-free windows in mm_khugepaged_scan_file(), mm_khugepaged_scan_pmd() and mm_khugepaged_collapse_file(). This patch (of 3): When the swap entries found exceed max_ptes_swap, the loop is left via break with folio still holding the xarray value that encodes the swap entry, not valid folio pointer. That value is passed to trace_mm_khugepaged_scan_file(), which feeds it to folio_pfn(). On FLATMEM and SPARSEMEM_VMEMMAP, the page_to_pfn() is plain pointer arithmetic, so the trace event merely prints bogus scan_pfn. On classic SPARSEMEM, the page_to_pfn() reads page->flags, dereferencing the tiny encoded integer and oopsing khugepaged whenever the trace event is enabled. So when folio is the swap entry value, simply set pfn to -1, just like exhausted scan naturally. And the folio_put() has maybe dropped the last reference of folio. The trace_mm_khugepaged_scan_file() is left with a dangling folio pointer. so using the folio_pfn() before dropping the reference, closing use-after-free window. Link: https://lore.kernel.org/20260909025804.3233645-1-vernon2gm@gmail.com Link: https://lore.kernel.org/20260909025804.3233645-2-vernon2gm@gmail.com Fixes: d41fd2016ed0 ("mm/khugepaged: add tracepoint to hpage_collapse_scan_file()") Signed-off-by: Vernon Yang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Barry Song Cc: Dev Jain Cc: Lance Yang Cc: Lorenzo Stoakes Cc: Ryan Roberts Cc: Zach O'Keefe Cc: --- include/trace/events/huge_memory.h | 6 +++--- mm/khugepaged.c | 7 ++++++- 2 files changed, 9 insertions(+), 4 deletions(-) diff --git a/include/trace/events/huge_memory.h b/include/trace/events/huge_memory.h index 5a48c5406cce49..7b526528f85b80 100644 --- a/include/trace/events/huge_memory.h +++ b/include/trace/events/huge_memory.h @@ -178,10 +178,10 @@ TRACE_EVENT(mm_collapse_huge_page_swapin, TRACE_EVENT(mm_khugepaged_scan_file, - TP_PROTO(struct mm_struct *mm, struct folio *folio, struct file *file, + TP_PROTO(struct mm_struct *mm, unsigned long pfn, struct file *file, int present, int swap, int result), - TP_ARGS(mm, folio, file, present, swap, result), + TP_ARGS(mm, pfn, file, present, swap, result), TP_STRUCT__entry( __field(struct mm_struct *, mm) @@ -194,7 +194,7 @@ TRACE_EVENT(mm_khugepaged_scan_file, TP_fast_assign( __entry->mm = mm; - __entry->pfn = folio ? folio_pfn(folio) : -1; + __entry->pfn = pfn; __assign_str(filename); __entry->present = present; __entry->swap = swap; diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 75639298efc271..6380a12b8eee05 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -2683,6 +2683,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, int present, swap; int node = NUMA_NO_NODE; enum scan_result result = SCAN_SUCCEED; + unsigned long failed_pfn = -1; present = 0; swap = 0; @@ -2715,6 +2716,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, if (is_pmd_order(folio_order(folio))) { result = SCAN_PTE_MAPPED_HUGEPAGE; + failed_pfn = folio_pfn(folio); /* * PMD-sized THP implies that we can only try * retracting the PTE table. @@ -2726,6 +2728,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, node = folio_nid(folio); if (collapse_scan_abort(node, cc)) { result = SCAN_SCAN_ABORT; + failed_pfn = folio_pfn(folio); folio_put(folio); break; } @@ -2733,12 +2736,14 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, if (!folio_test_lru(folio)) { result = SCAN_PAGE_LRU; + failed_pfn = folio_pfn(folio); folio_put(folio); break; } if (folio_expected_ref_count(folio) + 1 != folio_ref_count(folio)) { result = SCAN_PAGE_COUNT; + failed_pfn = folio_pfn(folio); folio_put(folio); break; } @@ -2773,7 +2778,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, } } - trace_mm_khugepaged_scan_file(mm, folio, file, present, swap, result); + trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present, swap, result); return result; } From b65f6d1d72c3db9f9534f905891a31a8b8360d78 Mon Sep 17 00:00:00 2001 From: Vernon Yang Date: Wed, 9 Sep 2026 10:58:02 +0800 Subject: [PATCH 0262/1012] mm: khugepaged: fix folio is used after pte_unmap_unlock() After the page table lock has dropped, the folio can be freed concurrently. The trace_mm_khugepaged_scan_pmd() is left with a dangling folio pointer. So using the folio_pfn() before dropping the page table lock, closing use-after-free window. And other pre-existing bug, When the `for (i = 0; i < HPAGE_PMD_NR; i++)` iteration to terminate and the folio operation preceding is normal, but pfn will be incorrect. so we really only trace the PFN if it really was problematic. Link: https://lore.kernel.org/20260909025804.3233645-3-vernon2gm@gmail.com Fixes: 7d2eba0557c1 ("mm: add tracepoint for scanning pages") Signed-off-by: Vernon Yang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Lorenzo Stoakes (ARM) Cc: Barry Song Cc: Dev Jain Cc: Lance Yang Cc: Ryan Roberts Cc: Zach O'Keefe Cc: --- include/trace/events/huge_memory.h | 6 +++--- mm/khugepaged.c | 10 +++++++++- 2 files changed, 12 insertions(+), 4 deletions(-) diff --git a/include/trace/events/huge_memory.h b/include/trace/events/huge_memory.h index 7b526528f85b80..fa828967e1fb0e 100644 --- a/include/trace/events/huge_memory.h +++ b/include/trace/events/huge_memory.h @@ -55,10 +55,10 @@ SCAN_STATUS TRACE_EVENT(mm_khugepaged_scan_pmd, - TP_PROTO(struct mm_struct *mm, struct folio *folio, + TP_PROTO(struct mm_struct *mm, unsigned long pfn, int referenced, int none_or_zero, int status, int unmapped), - TP_ARGS(mm, folio, referenced, none_or_zero, status, unmapped), + TP_ARGS(mm, pfn, referenced, none_or_zero, status, unmapped), TP_STRUCT__entry( __field(struct mm_struct *, mm) @@ -71,7 +71,7 @@ TRACE_EVENT(mm_khugepaged_scan_pmd, TP_fast_assign( __entry->mm = mm; - __entry->pfn = folio ? folio_pfn(folio) : -1; + __entry->pfn = pfn; __entry->referenced = referenced; __entry->none_or_zero = none_or_zero; __entry->status = status; diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 6380a12b8eee05..730829947b4ce2 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1612,6 +1612,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, enum scan_result result = SCAN_FAIL; struct page *page = NULL; struct folio *folio = NULL; + unsigned long failed_pfn = -1; unsigned long addr; unsigned long enabled_orders; spinlock_t *ptl; @@ -1706,11 +1707,13 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, if (cc->is_khugepaged && !(vma->vm_flags & VM_DROPPABLE) && folio_test_lazyfree(folio) && !pte_dirty(pteval)) { result = SCAN_PAGE_LAZYFREE; + failed_pfn = folio_pfn(folio); goto out_unmap; } if (!folio_test_anon(folio)) { result = SCAN_PAGE_ANON; + failed_pfn = folio_pfn(folio); goto out_unmap; } @@ -1721,6 +1724,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, if (folio_maybe_mapped_shared(folio)) { if (++shared > max_ptes_shared) { result = SCAN_EXCEED_SHARED_PTE; + failed_pfn = folio_pfn(folio); count_collapse_event(HPAGE_PMD_ORDER, THP_SCAN_EXCEED_SHARED_PTE, MTHP_STAT_COLLAPSE_EXCEED_SHARED); goto out_unmap; @@ -1738,15 +1742,18 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, node = folio_nid(folio); if (collapse_scan_abort(node, cc)) { result = SCAN_SCAN_ABORT; + failed_pfn = folio_pfn(folio); goto out_unmap; } cc->node_load[node]++; if (!folio_test_lru(folio)) { result = SCAN_PAGE_LRU; + failed_pfn = folio_pfn(folio); goto out_unmap; } if (folio_test_locked(folio)) { result = SCAN_PAGE_LOCK; + failed_pfn = folio_pfn(folio); goto out_unmap; } @@ -1759,6 +1766,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, */ if (folio_expected_ref_count(folio) != folio_ref_count(folio)) { result = SCAN_PAGE_COUNT; + failed_pfn = folio_pfn(folio); goto out_unmap; } @@ -1784,7 +1792,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, *lock_dropped = true; } out: - trace_mm_khugepaged_scan_pmd(mm, folio, referenced, + trace_mm_khugepaged_scan_pmd(mm, failed_pfn, referenced, none_or_zero, result, unmapped); return result; } From f9e657f4fe249d917187d09e88ab747d7687f6fd Mon Sep 17 00:00:00 2001 From: Vernon Yang Date: Wed, 9 Sep 2026 10:58:03 +0800 Subject: [PATCH 0263/1012] mm: khugepaged: fix folio is used after folio_put/unlock() On the rollback path, folio_put() has already dropped the last reference of new_folio. On the success path, new_folio is already unlocked and can be freed concurrently. The trace_mm_khugepaged_collapse_file() is left with a dangling folio pointer. So using the folio_pfn() before dropping the reference, closing use-after-free window. Link: https://lore.kernel.org/20260909025804.3233645-4-vernon2gm@gmail.com Fixes: 4c9473e87e75 ("mm/khugepaged: add tracepoint to collapse_file()") Signed-off-by: Vernon Yang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Lorenzo Stoakes (ARM) Cc: Barry Song Cc: Dev Jain Cc: Lance Yang Cc: Ryan Roberts Cc: Zach O'Keefe Cc: --- include/trace/events/huge_memory.h | 6 +++--- mm/khugepaged.c | 4 +++- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/include/trace/events/huge_memory.h b/include/trace/events/huge_memory.h index fa828967e1fb0e..5fb4d92cfd8408 100644 --- a/include/trace/events/huge_memory.h +++ b/include/trace/events/huge_memory.h @@ -211,10 +211,10 @@ TRACE_EVENT(mm_khugepaged_scan_file, ); TRACE_EVENT(mm_khugepaged_collapse_file, - TP_PROTO(struct mm_struct *mm, struct folio *new_folio, pgoff_t index, + TP_PROTO(struct mm_struct *mm, unsigned long new_pfn, pgoff_t index, unsigned long addr, bool is_shmem, struct file *file, int nr, int result), - TP_ARGS(mm, new_folio, index, addr, is_shmem, file, nr, result), + TP_ARGS(mm, new_pfn, index, addr, is_shmem, file, nr, result), TP_STRUCT__entry( __field(struct mm_struct *, mm) __field(unsigned long, hpfn) @@ -228,7 +228,7 @@ TRACE_EVENT(mm_khugepaged_collapse_file, TP_fast_assign( __entry->mm = mm; - __entry->hpfn = new_folio ? folio_pfn(new_folio) : -1; + __entry->hpfn = new_pfn; __entry->index = index; __entry->addr = addr; __entry->is_shmem = is_shmem; diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 730829947b4ce2..792166950bba81 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -2253,6 +2253,7 @@ static enum scan_result collapse_file(struct mm_struct *mm, unsigned long addr, struct address_space *mapping = file->f_mapping; struct page *dst; struct folio *folio, *tmp, *new_folio; + unsigned long new_pfn = -1; pgoff_t index = 0, end = start + HPAGE_PMD_NR; LIST_HEAD(pagelist); XA_STATE_ORDER(xas, &mapping->i_pages, start, HPAGE_PMD_ORDER); @@ -2272,6 +2273,7 @@ static enum scan_result collapse_file(struct mm_struct *mm, unsigned long addr, result = alloc_charge_folio(&new_folio, mm, cc, HPAGE_PMD_ORDER); if (result != SCAN_SUCCEED) goto out; + new_pfn = folio_pfn(new_folio); mapping_set_update(&xas, mapping); @@ -2675,7 +2677,7 @@ static enum scan_result collapse_file(struct mm_struct *mm, unsigned long addr, folio_put(new_folio); out: VM_BUG_ON(!list_empty(&pagelist)); - trace_mm_khugepaged_collapse_file(mm, new_folio, index, addr, is_shmem, file, HPAGE_PMD_NR, result); + trace_mm_khugepaged_collapse_file(mm, new_pfn, index, addr, is_shmem, file, HPAGE_PMD_NR, result); return result; } From 2065113e6cbe77cc60f46e76d5748def8d6cb505 Mon Sep 17 00:00:00 2001 From: Qi Zheng Date: Mon, 17 Aug 2026 17:03:26 +0800 Subject: [PATCH 0264/1012] mm: memcontrol: make obj_cgroup_memcg() handle NULL objcg Patch series "make unused huge shrinker memcg aware", v4. The shmem unused huge shrinker maintains a per-superblock list of inodes whose tail huge folio extends beyond i_size. Because this list is not memcg aware, reclaim triggered by memcg A can scan inodes across the entire superblock and split huge folios charged to unrelated memcg B, causing unexpected impact on it. In the worst case, memcg A has no reclaimable shmem at all, making the reclaim entirely useless and incurring unnecessary latency. We observed this in production, where page lock contention during split caused multi-hundred-millisecond stalls: tid 11340 comm scanner locked a page for 182264 us! kstack: unlock_page+1 split_huge_page_to_list+3135 shmem_unused_huge_shrink+767 super_cache_scan+329 do_shrink_slab+291 shrink_slab+533 shrink_node+400 do_try_to_free_pages+206 try_to_free_mem_cgroup_pages+262 try_charge_memcg+591 mem_cgroup_charge+136 __handle_mm_fault+2431 handle_mm_fault+194 do_user_addr_fault+462 __do_page_fault+176 do_page_fault+48 page_fault+62 Usama's recent patch [1] prevents the shmem unused shrinker from being invoked during memcg-level reclaim altogether, but this is overly conservative: we can do better by reclaiming only the shmem charged to the reclaiming memcg. This series converts the shrinker list to a memcg-aware list_lru, so that non-root memcg reclaim walks only candidates charged to the reclaiming memcg. Global reclaim, root memcg reclaim and shmem quota reclaim retain their existing global semantics. To avoid pinning a dying memcg through a long-lived CSS reference, each inode stores an obj_cgroup reference instead of a mem_cgroup reference. The list_lru add/delete paths resolve the current memcg from the objcg under RCU, staying consistent with list_lru's own memcg migration on offline. This patch (of 3): obj_cgroup_memcg() currently requires a non-NULL objcg, so callers that may hold a NULL objcg must guard the call with an explicit NULL check. This pattern is duplicated in folio_memcg(), folio_memcg_check(), mm/page_owner.c, and mm/zswap.c. Teach obj_cgroup_memcg() to accept NULL and return NULL in that case, then remove the redundant NULL checks at the call sites. Also remove the mem_cgroup_from_entry() wrapper in zswap, which existed solely to provide this NULL-safe behaviour, and replace its two callers with direct obj_cgroup_memcg() calls. No functional change intended. Link: https://lore.kernel.org/cover.1786955972.git.zhengqi.arch@bytedance.com Link: https://lore.kernel.org/09bcf74312246a6e4146be8a0cb9787f8beddb28.1786955972.git.zhengqi.arch@bytedance.com Signed-off-by: Qi Zheng Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Cc: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 11 ++++++++--- mm/page_owner.c | 2 +- mm/zswap.c | 17 ++--------------- 3 files changed, 11 insertions(+), 19 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 7d1c0ce189a887..da625d2edb3bab 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -380,7 +380,7 @@ enum objext_flags { static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) { lockdep_assert_once(rcu_read_lock_held() || lockdep_is_held(&cgroup_mutex)); - return READ_ONCE(objcg->memcg); + return objcg ? READ_ONCE(objcg->memcg) : NULL; } /* @@ -433,7 +433,7 @@ static inline struct mem_cgroup *folio_memcg(struct folio *folio) { struct obj_cgroup *objcg = folio_objcg(folio); - return objcg ? obj_cgroup_memcg(objcg) : NULL; + return obj_cgroup_memcg(objcg); } /* @@ -476,7 +476,7 @@ static inline struct mem_cgroup *folio_memcg_check(struct folio *folio) objcg = (void *)(memcg_data & ~OBJEXTS_FLAGS_MASK); - return objcg ? obj_cgroup_memcg(objcg) : NULL; + return obj_cgroup_memcg(objcg); } static inline struct mem_cgroup *page_memcg_check(struct page *page) @@ -1050,6 +1050,11 @@ void mem_cgroup_flush_workqueue(void); extern int mem_cgroup_init(void); #else /* CONFIG_MEMCG */ +static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) +{ + return NULL; +} + #define MEM_CGROUP_ID_SHIFT 0 #define root_mem_cgroup (NULL) diff --git a/mm/page_owner.c b/mm/page_owner.c index fbbda7ba914ba5..3fc37d9b908ef0 100644 --- a/mm/page_owner.c +++ b/mm/page_owner.c @@ -575,7 +575,7 @@ static inline int print_page_owner_memcg(char *kbuf, size_t count, int ret, } objcg = (void *)(memcg_data & ~OBJEXTS_FLAGS_MASK); - memcg = objcg ? obj_cgroup_memcg(objcg) : NULL; + memcg = obj_cgroup_memcg(objcg); if (!memcg) goto out_unlock; diff --git a/mm/zswap.c b/mm/zswap.c index 37f34e406c8e3b..c1dc60926bad99 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -647,19 +647,6 @@ static int zswap_enabled_param_set(const char *val, * lru functions **********************************/ -/* should be called under RCU */ -#ifdef CONFIG_MEMCG -static inline struct mem_cgroup *mem_cgroup_from_entry(struct zswap_entry *entry) -{ - return entry->objcg ? obj_cgroup_memcg(entry->objcg) : NULL; -} -#else -static inline struct mem_cgroup *mem_cgroup_from_entry(struct zswap_entry *entry) -{ - return NULL; -} -#endif - static inline int entry_to_nid(struct zswap_entry *entry) { return page_to_nid(virt_to_page(entry)); @@ -682,7 +669,7 @@ static void zswap_lru_add(struct zswap_entry *entry) * Similar reasoning holds for list_lru_del(). */ rcu_read_lock(); - memcg = mem_cgroup_from_entry(entry); + memcg = obj_cgroup_memcg(entry->objcg); /* will always succeed */ list_lru_add(&zswap_list_lru, &entry->lru, nid, memcg); rcu_read_unlock(); @@ -694,7 +681,7 @@ static void zswap_lru_del(struct zswap_entry *entry) struct mem_cgroup *memcg; rcu_read_lock(); - memcg = mem_cgroup_from_entry(entry); + memcg = obj_cgroup_memcg(entry->objcg); /* will always succeed */ list_lru_del(&zswap_list_lru, &entry->lru, nid, memcg); rcu_read_unlock(); From 18af6c21815b65f105dbec32b412ada7a2c56a8a Mon Sep 17 00:00:00 2001 From: Qi Zheng Date: Mon, 17 Aug 2026 17:03:27 +0800 Subject: [PATCH 0265/1012] mm: shmem: move unused huge shrinklist queuing past the truncation check The shmem_get_folio_gfp() adds the inode to the unused huge shrinker list at the alloced label, but a subsequent truncation check may still fail and remove the folio, leaving the inode on the list with a stale folio. The original code works because the shrinker re-looks-up the folio and drops stale entries, but it is cleaner to queue the inode only after all checks that might remove the folio have passed. So just make the pure structural move with no functional change, and it serves as preparation for the memcg-aware shrinker conversion. Link: https://lore.kernel.org/17fcf64faec0dfbbe6cf8a97924e3f39cd3c51a4.1786955972.git.zhengqi.arch@bytedance.com Signed-off-by: Qi Zheng Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt --- mm/shmem.c | 48 +++++++++++++++++++++++++++--------------------- 1 file changed, 27 insertions(+), 21 deletions(-) diff --git a/mm/shmem.c b/mm/shmem.c index 848316eaa7f4fb..9f2585d6dbfb06 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -2539,27 +2539,6 @@ static int shmem_get_folio_gfp(struct inode *inode, pgoff_t index, alloced: alloced = true; - if (folio_test_large(folio) && - DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE) < - folio_next_index(folio)) { - struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); - struct shmem_inode_info *info = SHMEM_I(inode); - /* - * Part of the large folio is beyond i_size: subject - * to shrink under memory pressure. - */ - spin_lock(&sbinfo->shrinklist_lock); - /* - * _careful to defend against unlocked access to - * ->shrink_list in shmem_unused_huge_shrink() - */ - if (list_empty_careful(&info->shrinklist)) { - list_add_tail(&info->shrinklist, - &sbinfo->shrinklist); - sbinfo->shrinklist_len++; - } - spin_unlock(&sbinfo->shrinklist_lock); - } if (sgp == SGP_WRITE) folio_set_referenced(folio); @@ -2589,6 +2568,33 @@ static int shmem_get_folio_gfp(struct inode *inode, pgoff_t index, error = -EINVAL; goto unlock; } + + /* + * Queue the inode on the shrink list only after all checks that might + * remove the folio have passed. Otherwise the inode could be left on + * the shrinker list with a stale folio. + */ + if (alloced && folio_test_large(folio) && + DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE) < folio_next_index(folio)) { + struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); + struct shmem_inode_info *info = SHMEM_I(inode); + /* + * Part of the large folio is beyond i_size: subject + * to shrink under memory pressure. + */ + spin_lock(&sbinfo->shrinklist_lock); + /* + * _careful to defend against unlocked access to + * ->shrink_list in shmem_unused_huge_shrink() + */ + if (list_empty_careful(&info->shrinklist)) { + list_add_tail(&info->shrinklist, + &sbinfo->shrinklist); + sbinfo->shrinklist_len++; + } + spin_unlock(&sbinfo->shrinklist_lock); + } + out: *foliop = folio; return 0; From bf112239ff03e527cac724f7449ca7381e2c1e84 Mon Sep 17 00:00:00 2001 From: Qi Zheng Date: Mon, 17 Aug 2026 17:03:28 +0800 Subject: [PATCH 0266/1012] mm: shmem: make unused huge shrinker memcg aware The shmem unused huge shrinker keeps a per-superblock list of inodes whose tail huge folio extends beyond i_size. Since that list is not memcg aware, reclaim triggered by one memcg can scan inodes from the whole superblock and split shmem huge folios charged to unrelated memcgs. Convert the shrink list to a memcg-aware list_lru. Queue each inode on the list_lru sublist matching the memcg and node of the current tail huge folio, so non-root memcg reclaim only walks candidates charged to the reclaiming memcg. Global reclaim, root memcg reclaim and shmem quota reclaim keep global semantics. Rather than pinning a struct mem_cgroup reference in shmem_inode_info, store a struct obj_cgroup reference instead. The list_lru add and delete paths resolve the current memcg from the objcg under RCU, so that memcg offline and list_lru entry migration remain consistent: list_lru migrates entries to the parent memcg sublist on offline, and obj_cgroup_memcg() follows the same reparenting, ensuring the correct sublist is always found at delete time. This avoids pinning a dying memcg through a long-lived CSS reference. The list_lru still tracks inodes while the actual split target is the current tail huge folio, so validate the folio memcg/node during scan. If the folio no longer matches the reclaim context or splitting cannot proceed, requeue the inode according to the current tail folio; if the inode is no longer shrinkable, drop the scan entry. This can be tested with the shrinker debugfs interface by allocating 32 tmpfs tail THPs in each of two memcgs, then scanning the sb-tmpfs shrinker with memcg A's cgroup id: before A scan after A scan base A=64M, B=64M A=64M, B=64M (per-memcg count is skipped) patched A=64M, B=64M A=0, B=64M Link: https://lore.kernel.org/94cc7fe1fd645254ae90effc1c0687678438d940.1786955972.git.zhengqi.arch@bytedance.com Signed-off-by: Qi Zheng Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt Cc: Qinyun Tan --- include/linux/shmem_fs.h | 12 +- mm/shmem.c | 362 ++++++++++++++++++++++++++++++--------- 2 files changed, 289 insertions(+), 85 deletions(-) diff --git a/include/linux/shmem_fs.h b/include/linux/shmem_fs.h index 5663dff53186e2..a7c7a96a7cbf90 100644 --- a/include/linux/shmem_fs.h +++ b/include/linux/shmem_fs.h @@ -11,6 +11,7 @@ #include #include #include +#include /* inode in-kernel data */ @@ -54,6 +55,11 @@ struct shmem_inode_info { struct dquot __rcu *i_dquot[MAXQUOTAS]; #endif struct inode vfs_inode; + +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + struct obj_cgroup *shrinklist_objcg; + int shrinklist_nid; +#endif }; #define SHMEM_FL_USER_VISIBLE (FS_FL_USER_VISIBLE | FS_CASEFOLD_FL) @@ -83,9 +89,9 @@ struct shmem_sb_info { ino_t next_ino; /* The next per-sb inode number to use */ ino_t __percpu *ino_batch; /* The next per-cpu inode number to use */ struct mempolicy *mpol; /* default memory policy for mappings */ - spinlock_t shrinklist_lock; /* Protects shrinklist */ - struct list_head shrinklist; /* List of shinkable inodes */ - unsigned long shrinklist_len; /* Length of shrinklist */ +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + struct list_lru shrinklist; /* List of shrinkable inodes */ +#endif struct shmem_quota_limits qlimits; /* Default quota limits */ struct simple_xattr_cache xa_cache; }; diff --git a/mm/shmem.c b/mm/shmem.c index 9f2585d6dbfb06..ae39966fc7304e 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -725,51 +725,258 @@ static const char *shmem_format_huge(int huge) } #endif -static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, - struct shrink_control *sc, unsigned long nr_to_free) +static bool is_shmem_unused_huge_isolated(struct shmem_inode_info *info) { - LIST_HEAD(list), *pos, *next; - struct inode *inode; + + return info->shrinklist_nid == -1; +} + +static void set_shmem_unused_huge_isolated(struct shmem_inode_info *info) +{ + info->shrinklist_nid = -1; +} + +static struct obj_cgroup *shmem_get_and_clear_objcg(struct shmem_inode_info *info) +{ + struct obj_cgroup *objcg = info->shrinklist_objcg; + + info->shrinklist_objcg = NULL; + + return objcg; +} + +#ifdef CONFIG_MEMCG +static struct obj_cgroup * +shmem_unused_huge_alloc_lru(struct shmem_sb_info *sbinfo, struct folio *folio, + gfp_t gfp) +{ + int ret; + + ret = folio_memcg_list_lru_alloc(folio, &sbinfo->shrinklist, gfp); + if (ret) + return ERR_PTR(ret); + + return get_obj_cgroup_from_folio(folio); +} +#else +static struct obj_cgroup * +shmem_unused_huge_alloc_lru(struct shmem_sb_info *sbinfo, struct folio *folio, + gfp_t gfp) +{ + return NULL; +} +#endif + +static void shmem_unused_huge_lru_add(struct shmem_sb_info *sbinfo, + struct list_head *item, int nid, + struct obj_cgroup *objcg) +{ + struct mem_cgroup *memcg; + + rcu_read_lock(); + memcg = obj_cgroup_memcg(objcg); + list_lru_add(&sbinfo->shrinklist, item, nid, memcg); + rcu_read_unlock(); +} + +static void shmem_unused_huge_lru_del(struct shmem_sb_info *sbinfo, + struct list_head *item, int nid, + struct obj_cgroup *objcg) +{ + struct mem_cgroup *memcg; + + rcu_read_lock(); + memcg = obj_cgroup_memcg(objcg); + list_lru_del(&sbinfo->shrinklist, item, nid, memcg); + rcu_read_unlock(); +} + +static void shmem_unused_huge_add(struct inode *inode, struct folio *folio, + gfp_t gfp) +{ + struct shmem_inode_info *info = SHMEM_I(inode); + struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); + int nid = folio_nid(folio); + struct obj_cgroup *objcg = NULL, *old_objcg = NULL; + + objcg = shmem_unused_huge_alloc_lru(sbinfo, folio, gfp); + if (IS_ERR(objcg)) + return; + + spin_lock(&info->lock); + if (!list_empty(&info->shrinklist)) { + /* isolated on scan list, let shrink handle it */ + if (is_shmem_unused_huge_isolated(info)) + goto unlock; + + if (info->shrinklist_nid == nid && + info->shrinklist_objcg == objcg) + goto unlock; + + shmem_unused_huge_lru_del(sbinfo, &info->shrinklist, + info->shrinklist_nid, + info->shrinklist_objcg); + old_objcg = shmem_get_and_clear_objcg(info); + } + + info->shrinklist_objcg = objcg; + info->shrinklist_nid = nid; + shmem_unused_huge_lru_add(sbinfo, &info->shrinklist, nid, objcg); + objcg = NULL; +unlock: + spin_unlock(&info->lock); + obj_cgroup_put(old_objcg); + obj_cgroup_put(objcg); +} + +static void shmem_unused_huge_del(struct inode *inode) +{ + struct shmem_inode_info *info = SHMEM_I(inode); + struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); + struct obj_cgroup *objcg = NULL; + + spin_lock(&info->lock); + if (!list_empty(&info->shrinklist)) { + shmem_unused_huge_lru_del(sbinfo, &info->shrinklist, + info->shrinklist_nid, + info->shrinklist_objcg); + objcg = shmem_get_and_clear_objcg(info); + } + spin_unlock(&info->lock); + + obj_cgroup_put(objcg); +} + +struct shmem_unused_huge_scan { + struct list_head list; + struct shrink_control *sc; +}; + +static enum lru_status shmem_unused_huge_isolate(struct list_head *item, + struct list_lru_one *lru, + void *arg) +{ + struct shmem_unused_huge_scan *scan = arg; struct shmem_inode_info *info; - struct folio *folio; - unsigned long batch = sc ? sc->nr_to_scan : 128; - unsigned long split = 0, freed = 0; + struct inode *inode; + struct obj_cgroup *objcg = NULL; - if (list_empty(&sbinfo->shrinklist)) - return SHRINK_STOP; + info = list_entry(item, struct shmem_inode_info, shrinklist); - spin_lock(&sbinfo->shrinklist_lock); - list_for_each_safe(pos, next, &sbinfo->shrinklist) { - info = list_entry(pos, struct shmem_inode_info, shrinklist); + /* + * Use trylock to avoid ABBA deadlock: add/del path takes info->lock + * before the list_lru bucket lock, while here the order is reversed. + */ + if (!spin_trylock(&info->lock)) + return LRU_SKIP; - /* pin the inode */ - inode = igrab(&info->vfs_inode); + /* pin the inode */ + inode = igrab(&info->vfs_inode); + /* inode is about to be evicted */ + if (!inode) { + list_lru_isolate(lru, item); + objcg = shmem_get_and_clear_objcg(info); + spin_unlock(&info->lock); + obj_cgroup_put(objcg); + return LRU_REMOVED; + } - /* inode is about to be evicted */ - if (!inode) { - list_del_init(&info->shrinklist); - goto next; - } + list_lru_isolate(lru, item); + objcg = shmem_get_and_clear_objcg(info); + set_shmem_unused_huge_isolated(info); + list_add_tail(&info->shrinklist, &scan->list); + spin_unlock(&info->lock); + obj_cgroup_put(objcg); - list_move(&info->shrinklist, &list); -next: - sbinfo->shrinklist_len--; - if (!--batch) - break; + return LRU_REMOVED; +} + +static bool is_shmem_unused_huge_match(struct folio *folio, + struct shrink_control *sc) +{ + struct mem_cgroup *memcg = NULL; + bool match; + + /* shmem quota reclaim has no NUMA node or memcg restriction */ + if (!sc) + return true; + + if (folio_nid(folio) != sc->nid) + return false; + + /* + * Only non-root memcg reclaim needs to match the folio charge against + * sc->memcg. Skip the folio memcg check for global shrinker reclaim and + * root memcg reclaim. + */ + if (!sc->memcg || mem_cgroup_is_root(sc->memcg)) + return true; + + memcg = get_mem_cgroup_from_folio(folio); + match = memcg == sc->memcg; + mem_cgroup_put(memcg); + + return match; +} + +static void shmem_unused_huge_drop(struct inode *inode) +{ + struct shmem_inode_info *info = SHMEM_I(inode); + + spin_lock(&info->lock); + list_del_init(&info->shrinklist); + spin_unlock(&info->lock); +} + +static void shmem_unused_huge_requeue(struct inode *inode, struct folio *folio) +{ + struct shmem_inode_info *info = SHMEM_I(inode); + struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); + struct obj_cgroup *objcg; + int nid = folio_nid(folio); + + objcg = shmem_unused_huge_alloc_lru(sbinfo, folio, GFP_NOWAIT); + if (IS_ERR(objcg)) { + shmem_unused_huge_drop(inode); + return; } - spin_unlock(&sbinfo->shrinklist_lock); - list_for_each_safe(pos, next, &list) { - pgoff_t next, end; + spin_lock(&info->lock); + /* Requeue the inode to shrinklist */ + list_del_init(&info->shrinklist); + shmem_unused_huge_lru_add(sbinfo, &info->shrinklist, nid, objcg); + info->shrinklist_objcg = objcg; + info->shrinklist_nid = nid; + spin_unlock(&info->lock); +} + +static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, + struct shrink_control *sc, unsigned long nr_to_free) +{ + struct shmem_unused_huge_scan scan; + struct inode *inode; + struct shmem_inode_info *info; + struct folio *folio; + struct list_head *pos, *next; + unsigned long split = 0, freed = 0; + + INIT_LIST_HEAD(&scan.list); + scan.sc = sc; + if (sc) + list_lru_shrink_walk(&sbinfo->shrinklist, sc, + shmem_unused_huge_isolate, &scan); + else + list_lru_walk(&sbinfo->shrinklist, shmem_unused_huge_isolate, + &scan, 128); + + list_for_each_safe(pos, next, &scan.list) { + pgoff_t folio_end, end; loff_t i_size; int ret; info = list_entry(pos, struct shmem_inode_info, shrinklist); inode = &info->vfs_inode; - if (nr_to_free && freed >= nr_to_free) - goto move_back; - i_size = i_size_read(inode); folio = filemap_get_entry(inode->i_mapping, i_size / PAGE_SIZE); if (!folio || xa_is_value(folio)) @@ -782,13 +989,19 @@ static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, } /* Check if there is anything to gain from splitting */ - next = folio_next_index(folio); + folio_end = folio_next_index(folio); end = shmem_fallocend(inode, DIV_ROUND_UP(i_size, PAGE_SIZE)); - if (end <= folio->index || end >= next) { + if (end <= folio->index || end >= folio_end) { folio_put(folio); goto drop; } + if (!is_shmem_unused_huge_match(folio, scan.sc)) + goto move_back; + + if (nr_to_free && freed >= nr_to_free) + goto move_back; + /* * Move the inode on the list back to shrinklist if we failed * to lock the page at this time. @@ -796,35 +1009,30 @@ static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, * Waiting for the lock may lead to deadlock in the * reclaim path. */ - if (!folio_trylock(folio)) { - folio_put(folio); + if (!folio_trylock(folio)) + goto move_back; + + if (!is_shmem_unused_huge_match(folio, scan.sc)) { + folio_unlock(folio); goto move_back; } ret = split_folio(folio); folio_unlock(folio); - folio_put(folio); /* If split failed move the inode on the list back to shrinklist */ if (ret) goto move_back; - freed += next - end; + freed += folio_end - end; split++; + folio_put(folio); drop: - list_del_init(&info->shrinklist); + shmem_unused_huge_drop(inode); goto put; move_back: - /* - * Make sure the inode is either on the global list or deleted - * from any local list before iput() since it could be deleted - * in another thread once we put the inode (then the local list - * is corrupted). - */ - spin_lock(&sbinfo->shrinklist_lock); - list_move(&info->shrinklist, &sbinfo->shrinklist); - sbinfo->shrinklist_len++; - spin_unlock(&sbinfo->shrinklist_lock); + shmem_unused_huge_requeue(inode, folio); + folio_put(folio); put: iput(inode); } @@ -837,7 +1045,7 @@ static long shmem_unused_huge_scan(struct super_block *sb, { struct shmem_sb_info *sbinfo = SHMEM_SB(sb); - if (!READ_ONCE(sbinfo->shrinklist_len)) + if (!list_lru_shrink_count(&sbinfo->shrinklist, sc)) return SHRINK_STOP; return shmem_unused_huge_shrink(sbinfo, sc, 0); @@ -848,21 +1056,21 @@ static long shmem_unused_huge_count(struct super_block *sb, { struct shmem_sb_info *sbinfo = SHMEM_SB(sb); - /* - * The per-superblock shrinklist is filesystem-global and does not - * honour sc->memcg, so it is only meaningful on the global (kswapd or - * root direct reclaim) shrink path. Skip the per-memcg iterations of - * shrink_slab_memcg() to avoid queueing duplicate global work. - */ - if (!mem_cgroup_shrink_is_root(sc)) - return 0; - - return READ_ONCE(sbinfo->shrinklist_len); + return list_lru_shrink_count(&sbinfo->shrinklist, sc); } #else /* !CONFIG_TRANSPARENT_HUGEPAGE */ #define shmem_huge SHMEM_HUGE_DENY +static void shmem_unused_huge_add(struct inode *inode, struct folio *folio, + gfp_t gfp) +{ +} + +static void shmem_unused_huge_del(struct inode *inode) +{ +} + static unsigned long shmem_unused_huge_shrink(struct shmem_sb_info *sbinfo, struct shrink_control *sc, unsigned long nr_to_free) { @@ -1418,14 +1626,7 @@ static void shmem_evict_inode(struct inode *inode) inode->i_size = 0; mapping_set_exiting(inode->i_mapping); shmem_truncate_range(inode, 0, (loff_t)-1); - if (!list_empty(&info->shrinklist)) { - spin_lock(&sbinfo->shrinklist_lock); - if (!list_empty(&info->shrinklist)) { - list_del_init(&info->shrinklist); - sbinfo->shrinklist_len--; - } - spin_unlock(&sbinfo->shrinklist_lock); - } + shmem_unused_huge_del(inode); while (!list_empty(&info->swaplist)) { /* Wait while shmem_unuse() is scanning this inode... */ wait_var_event(&info->stop_eviction, @@ -2576,25 +2777,12 @@ static int shmem_get_folio_gfp(struct inode *inode, pgoff_t index, */ if (alloced && folio_test_large(folio) && DIV_ROUND_UP(i_size_read(inode), PAGE_SIZE) < folio_next_index(folio)) { - struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); - struct shmem_inode_info *info = SHMEM_I(inode); /* * Part of the large folio is beyond i_size: subject * to shrink under memory pressure. */ - spin_lock(&sbinfo->shrinklist_lock); - /* - * _careful to defend against unlocked access to - * ->shrink_list in shmem_unused_huge_shrink() - */ - if (list_empty_careful(&info->shrinklist)) { - list_add_tail(&info->shrinklist, - &sbinfo->shrinklist); - sbinfo->shrinklist_len++; - } - spin_unlock(&sbinfo->shrinklist_lock); + shmem_unused_huge_add(inode, folio, gfp); } - out: *foliop = folio; return 0; @@ -3071,6 +3259,10 @@ static struct inode *__shmem_get_inode(struct mnt_idmap *idmap, if (info->fsflags) shmem_set_inode_flags(inode, info->fsflags, NULL); INIT_LIST_HEAD(&info->shrinklist); +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + info->shrinklist_objcg = NULL; + info->shrinklist_nid = -1; +#endif INIT_LIST_HEAD(&info->swaplist); cache_no_acl(inode); if (sbinfo->noswap) @@ -4945,6 +5137,9 @@ static void shmem_put_super(struct super_block *sb) #endif free_percpu(sbinfo->ino_batch); percpu_counter_destroy(&sbinfo->used_blocks); +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + list_lru_destroy(&sbinfo->shrinklist); +#endif mpol_put(sbinfo->mpol); #ifdef CONFIG_TMPFS_XATTR simple_xattr_cache_cleanup(&sbinfo->xa_cache); @@ -5038,8 +5233,11 @@ static int shmem_fill_super(struct super_block *sb, struct fs_context *fc) raw_spin_lock_init(&sbinfo->stat_lock); if (percpu_counter_init(&sbinfo->used_blocks, 0, GFP_KERNEL)) goto failed; - spin_lock_init(&sbinfo->shrinklist_lock); - INIT_LIST_HEAD(&sbinfo->shrinklist); + +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + if (list_lru_init_memcg(&sbinfo->shrinklist, sb->s_shrink)) + goto failed; +#endif sb->s_maxbytes = MAX_LFS_FILESIZE; sb->s_blocksize = PAGE_SIZE; From ceec2ba4c7739d5e2239a90ebd4aaa0fede3bd37 Mon Sep 17 00:00:00 2001 From: Qinyun Tan Date: Wed, 2 Sep 2026 17:32:02 +0800 Subject: [PATCH 0267/1012] mm/list_lru: disable memcg awareness under cgroup_disable=memory __list_lru_init() only collapses a memcg-aware list_lru into plain per-node lists when kmem accounting is disabled (cgroup.memory=nokmem). When the memory controller is disabled entirely (cgroup_disable=memory), mem_cgroup_kmem_disabled() is false, so the lru stays memcg aware even though no object will ever be charged to a memcg. This is more than a semantic inconsistency. folio_memcg_list_lru_alloc() trusts list_lru_memcg_aware() and dereferences the folio's memcg, which is always NULL with the controller disabled. The only mainline caller, folio_memcg_alloc_deferred(), papers over this with an explicit mem_cgroup_disabled() check. The shmem unused-huge shrinker conversion ("mm: shmem: make unused huge shrinker memcg aware") adds a second caller without such a guard, so booting with cgroup_disable=memory and writing to a huge=always tmpfs oopses: BUG: unable to handle page fault for address: 0000000000000488 RIP: 0010:folio_memcg_list_lru_alloc+0x41/0xf0 Call Trace: shmem_get_folio_gfp+0x1cd/0x7c0 shmem_write_begin+0x5d/0x100 generic_perform_write+0x89/0x2a0 shmem_file_write_iter+0x82/0x90 vfs_write+0x256/0x410 ksys_write+0x61/0xe0 do_syscall_64+0x8d/0x460 entry_SYSCALL_64_after_hwframe+0x76/0x7e The faulting address is the offset of mem_cgroup->kmemcg_id, dereferenced on a NULL memcg in memcg_list_lru_allocated(): folio_memcg_list_lru_alloc() list_lru_memcg_aware() <- true, only nokmem checked memcg = folio_memcg(folio) <- NULL memcg_list_lru_allocated(memcg, lru) memcg->kmemcg_id <- NULL pointer dereference Check mem_cgroup_disabled() in __list_lru_init() so that all list_lrus fall back to plain per-node lists when the controller is disabled, matching what the shrinker side already does (shrinker_memcg_alloc() bails out on mem_cgroup_disabled()). This makes the mem_cgroup_disabled() check in callers unnecessary rather than mandatory. Link: https://lore.kernel.org/20260902093202.609559-1-qinyuntan@linux.alibaba.com Signed-off-by: Qinyun Tan Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Cc: Christian Brauner Cc: David Hildenbrand Cc: Hugh Dickins Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Qi Zheng Cc: Roman Gushchin Cc: Shakeel Butt --- mm/list_lru.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/list_lru.c b/mm/list_lru.c index 36662d02ff9631..a4522ca93ebcb9 100644 --- a/mm/list_lru.c +++ b/mm/list_lru.c @@ -671,7 +671,7 @@ int __list_lru_init(struct list_lru *lru, bool memcg_aware, struct shrinker *shr else lru->shrinker_id = -1; - if (mem_cgroup_kmem_disabled()) + if (mem_cgroup_disabled() || mem_cgroup_kmem_disabled()) memcg_aware = false; #endif From 5524dbf3e3e0a00beecae3cfe30c25e32899ca1f Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Fri, 28 Aug 2026 03:37:31 +0000 Subject: [PATCH 0268/1012] selftests/cgroup: test_zswap: wait for cgroup to unpopulate in test_zswap_writeback MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "selftests/cgroup: fixes for test_zswap on single core VM", v4. This series fixes two test failures in test_zswap observed when running on a single-core VM (-smp 1) with 4GB of RAM. Patch 1 addresses a race condition in test_zswap_writeback() where waitpid() returns before the exiting child process is switched away by the kernel, causing an immediate write of "+memory" to cgroup.subtree_control to fail with -EBUSY. We fix this by waiting for cgroup.events to report "populated 0". Patch 2 fixes an implicit unsigned conversion bug in test_no_kmem_bypass() where small negative timing differences between debugfs stored_pages and cgroup zswapped bytes caused the comparison to falsely fail due to unsigned promotion. This patch (of 2): When running test_zswap on a single-core VM (-smp 1) with 4GB of RAM, test_zswap_writeback intermittently fails on the initial run after boot. In test_zswap_writeback(), after waitpid() reaps the child process created by test_zswap_writeback_one(), writing "+memory" to cgroup.subtree_control can fail with -EBUSY. Under cgroup v2, enabling domain subtree controllers is forbidden while any tasks remain in cgroup.procs. When a child process exits, exit_notify() wakes the parent process, allowing waitpid() to return immediately. However, the cgroup populated task count (nr_populated_csets) is only decremented when the exiting task is switched away via finish_task_switch() -> cgroup_task_dead(). On single-core systems, the parent runs before the dead child has been switched out, causing "+memory" to fail with -EBUSY if written immediately after waitpid() returns. Fix this by waiting for cgroup.events to report "populated 0\n" via cg_read_strcmp_wait() before enabling subtree control. Link: https://lore.kernel.org/20260828033741.2184560-1-wfelipe@google.com Link: https://lore.kernel.org/20260828033741.2184560-2-wfelipe@google.com Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton Acked-by: Michal Koutný Cc: Chengming Zhou Cc: Johannes Weiner Cc: Nhat Pham Cc: Shuah Khan Cc: Tejun Heo --- tools/testing/selftests/cgroup/test_zswap.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tools/testing/selftests/cgroup/test_zswap.c b/tools/testing/selftests/cgroup/test_zswap.c index 8df54b59513a8c..b1abc94317e4f7 100644 --- a/tools/testing/selftests/cgroup/test_zswap.c +++ b/tools/testing/selftests/cgroup/test_zswap.c @@ -409,6 +409,8 @@ static int test_zswap_writeback(const char *root, bool wb) * Thus, the parent's setting shall be what's in effect. */ if (cg_write(test_group, "memory.zswap.max", "max")) goto out; + if (cg_read_strcmp_wait(test_group, "cgroup.events", "populated 0\n")) + goto out; if (cg_write(test_group, "cgroup.subtree_control", "+memory")) goto out; From 9d3ae3340e0ba611f71b898e6e40f2ceb232d504 Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Fri, 28 Aug 2026 03:37:32 +0000 Subject: [PATCH 0269/1012] selftests/cgroup: test_zswap: fix implicit unsigned promotion bug in test_no_kmem_bypass MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit In test_no_kmem_bypass(), delta (stored_pages * page_size - zswapped) is checked against stored_pages * page_size / 4 to verify that the pages pushed to zswap belong to the test memory cgroup. Due to slight stat update timing differences, delta can evaluate to a small negative number (e.g. -5MB out of 1GB). Because delta is declared as a signed int and stored_pages is an unsigned size_t, C's usual arithmetic conversions implicitly promote a negative delta to a large unsigned 64-bit integer, causing `delta < stored_pages * page_size / 4` to falsely evaluate to 0 and fail the test. Fix this by declaring zswapped and delta as signed long long and comparing against a signed threshold, ensuring negative deltas correctly evaluate to true. Link: https://lore.kernel.org/20260828033741.2184560-3-wfelipe@google.com Fixes: a549f9f31561a ("selftests: cgroup: add test_zswap with no kmem bypass test") Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton Acked-by: Michal Koutný Cc: Chengming Zhou Cc: Johannes Weiner Cc: Nhat Pham Cc: Shuah Khan Cc: Tejun Heo --- tools/testing/selftests/cgroup/test_zswap.c | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/tools/testing/selftests/cgroup/test_zswap.c b/tools/testing/selftests/cgroup/test_zswap.c index b1abc94317e4f7..1ac77907277570 100644 --- a/tools/testing/selftests/cgroup/test_zswap.c +++ b/tools/testing/selftests/cgroup/test_zswap.c @@ -654,11 +654,14 @@ static int test_no_kmem_bypass(const char *root) break; /* If memory was pushed to zswap, verify it belongs to memcg */ if (stored_pages > stored_pages_threshold) { - int zswapped = cg_read_key_long(test_group, "memory.stat", "zswapped "); - int delta = stored_pages * page_size - zswapped; - int result_ok = delta < stored_pages * page_size / 4; - - ret = result_ok ? KSFT_PASS : KSFT_FAIL; + long zswapped = cg_read_key_long( + test_group, "memory.stat", "zswapped "); + long long delta = + (long long)stored_pages * page_size - zswapped; + long long max_delta = + (long long)stored_pages * page_size / 4; + + ret = (delta < max_delta) ? KSFT_PASS : KSFT_FAIL; break; } } From ca7db66856be9813dcd1c63736fae6086a8fc5c1 Mon Sep 17 00:00:00 2001 From: Rik van Riel Date: Fri, 28 Aug 2026 13:50:36 -0400 Subject: [PATCH 0270/1012] mm/memcontrol: fix stuck FLUSHING_CACHED_CHARGE bit on isolated cpus When drain_all_stock() sets FLUSHING_CACHED_CHARGE before checking isolation, schedule_drain_work() can drop the work in a separate RCU critical section, and housekeeping_update()'s synchronize_rcu() can race that second check, leaving the flag set. drain_local_stock() only clears the bit for work that ran, so the flag remains set and the stock is never drained again. Have schedule_drain_work() return whether the work was queued, and clear FLUSHING_CACHED_CHARGE in drain_all_stock() when the remote CPU is isolated, so future drains can retry. Link: https://lore.kernel.org/20260828135036.7d44361f@fangorn Fixes: 6a792697a53a ("memcg: do not drain charge pcp caches on remote isolated cpus") Suggested-by: Michal Hocko Suggested-by: Shakeel Butt Signed-off-by: Rik van Riel Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Acked-by: Michal Hocko Cc: Johannes Weiner Cc: Muchun Song Cc: Roman Gushchin Cc: --- mm/memcontrol.c | 19 ++++++++++++------- 1 file changed, 12 insertions(+), 7 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 1709ac96bbdec5..c6b85e3a5a0b3e 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2306,7 +2306,7 @@ static bool is_memcg_drain_needed(struct memcg_stock_pcp *stock, return flush; } -static void schedule_drain_work(int cpu, struct work_struct *work) +static bool schedule_drain_work(int cpu, struct work_struct *work) { /* * Protect housekeeping cpumask read and work enqueue together @@ -2315,8 +2315,11 @@ static void schedule_drain_work(int cpu, struct work_struct *work) * pending work on newly isolated CPUs. */ guard(rcu)(); - if (!cpu_is_isolated(cpu)) - queue_work_on(cpu, memcg_wq, work); + if (cpu_is_isolated(cpu)) + return false; + + queue_work_on(cpu, memcg_wq, work); + return true; } /* @@ -2348,8 +2351,9 @@ void drain_all_stock(struct mem_cgroup *root_memcg) &memcg_st->flags)) { if (cpu == curcpu) drain_local_memcg_stock(&memcg_st->work); - else - schedule_drain_work(cpu, &memcg_st->work); + else if (!schedule_drain_work(cpu, &memcg_st->work)) + clear_bit(FLUSHING_CACHED_CHARGE, + &memcg_st->flags); } if (!test_bit(FLUSHING_CACHED_CHARGE, &obj_st->flags) && @@ -2358,8 +2362,9 @@ void drain_all_stock(struct mem_cgroup *root_memcg) &obj_st->flags)) { if (cpu == curcpu) drain_local_obj_stock(&obj_st->work); - else - schedule_drain_work(cpu, &obj_st->work); + else if (!schedule_drain_work(cpu, &obj_st->work)) + clear_bit(FLUSHING_CACHED_CHARGE, + &obj_st->flags); } } migrate_enable(); From b432fdc885755c046b1d9778c0f489db8ae6d352 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 28 Aug 2026 12:24:19 -0700 Subject: [PATCH 0271/1012] memcg: clear FLUSHING_CACHED_CHARGE on cpu offline Sashiko [1] reported that memcg_hotplug_cpu_dead() drains the stocks of the CPU which went away but leaves FLUSHING_CACHED_CHARGE alone. The flag can be set at that point: drain_all_stock() may have claimed the stock and queued the drain work shortly before the CPU went down. workqueue_offline_cpu() unbinds the per-cpu workers, so such a pending work item is executed by an unbound worker on some other CPU, where drain_local_memcg_stock() operates on this_cpu_ptr() and thus drains and clears the flag of that other CPU instead. Nothing clears the flag of the dead CPU, so drain_all_stock() would skip its stock forever once the CPU comes back online. Clear the flag of both stocks after draining them. Link: https://lore.kernel.org/20260828192419.3057939-1-shakeel.butt@linux.dev Link: https://sashiko.dev/#/patchset/20260828135036.7d44361f%40fangorn [1] Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Reviewed-by: Rik van Riel Reported-by: Sashiko Acked-by: Michal Hocko Cc: Johannes Weiner Cc: Muchun Song Cc: Roman Gushchin --- mm/memcontrol.c | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index c6b85e3a5a0b3e..05f5338765a995 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2373,9 +2373,21 @@ void drain_all_stock(struct mem_cgroup *root_memcg) static int memcg_hotplug_cpu_dead(unsigned int cpu) { + struct memcg_stock_pcp *memcg_st = &per_cpu(memcg_stock, cpu); + struct obj_stock_pcp *obj_st = &per_cpu(obj_stock, cpu); + /* no need for the local lock */ - drain_obj_stock(&per_cpu(obj_stock, cpu)); - drain_stock_fully(&per_cpu(memcg_stock, cpu)); + drain_obj_stock(obj_st); + drain_stock_fully(memcg_st); + + /* + * A drain work queued before the CPU went away is executed by an + * unbound worker on some other CPU and clears that CPU's flag, so + * clear the flags here to make these stocks drainable again once + * the CPU comes back online. + */ + clear_bit(FLUSHING_CACHED_CHARGE, &memcg_st->flags); + clear_bit(FLUSHING_CACHED_CHARGE, &obj_st->flags); return 0; } From 036865640eddd9c6457b73f0af043150b8d11733 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Fri, 28 Aug 2026 15:31:11 -0400 Subject: [PATCH 0272/1012] mm/mempolicy: take a cpuset cookie for the interleave node count alloc_pages_bulk_interleave() counts pol->nodes without a cpuset cookie: nodes = nodes_weight(pol->nodes); nr_pages_per_node = nr_pages / nodes; nodemask_t spans several words once MAX_NUMNODES exceeds BITS_PER_LONG, so a concurrent cpuset rebind can tear that read and yield an empty mask even though neither version of it was empty. The call then allocates nothing and returns 0. Some compilers will hoist the loop entry test above the division, because nr_pages_per_node is dead when the loop does not run. 682e: call ... <- nodes_weight() 6838: test %eax,%eax 683a: jle 692d <- nodes <= 0 skips the loop 684a: div %rcx So in most deployments, this div/0 is unreachable - but nothing in the source guarantees that, it's just not easily exercised. Take the cookie around the count and bail if the mask really is empty. Only the count needs it, interleave_nodes() takes the cookie itself so so a torn read there is already retried. A rebind landing mid-loop can still leave the count disagreeing with the mask, so the loop may revisit a node or skip one - but a rebind where nodes change causes migration, so a handful of misplaced pages isn't catastrophic in any sense. Measured on a 72 node VM (NODES_SHIFT=10) with a cgroup v2 cpuset flipping cpuset.mems between a word 0 and a word 1 node set, and the two word read artificially widened: 330 zero counts in 130414 calls without the cookie, and 401 retries with it. Link: https://lore.kernel.org/20260828193111.1023497-1-gourry@gourry.net Fixes: c00b6b961099 ("mm/vmalloc: introduce alloc_pages_bulk_array_mempolicy to accelerate memory allocation") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Reported-by: Chelsy Ratnawat Closes: https://lore.kernel.org/all/20250907160829.91628-1-chelsyratnawat2001@gmail.com/ Reviewed-by: Huang Ying Assisted-by: Claude:claude-opus-5 Cc: Alistair Popple Cc: Byungchul Park Cc: Chenwandun Cc: David Hildenbrand Cc: Joshua Hahn Cc: Matthew Brost Cc: Rakie Kim Cc: "Uladzislau Rezki (Sony)" Cc: Zi Yan Cc: --- mm/mempolicy.c | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 79053ece02cd48..060a0eb2691709 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -2592,6 +2592,7 @@ static unsigned long alloc_pages_bulk_interleave(gfp_t gfp, struct mempolicy *pol, unsigned long nr_pages, struct page **page_array) { + unsigned int cpuset_mems_cookie; int nodes; unsigned long nr_pages_per_node; int delta; @@ -2599,7 +2600,16 @@ static unsigned long alloc_pages_bulk_interleave(gfp_t gfp, unsigned long nr_allocated; unsigned long total_allocated = 0; - nodes = nodes_weight(pol->nodes); + /* count the nodes, retry if a rebind happened during the read */ + do { + cpuset_mems_cookie = read_mems_allowed_begin(); + nodes = nodes_weight(pol->nodes); + } while (read_mems_allowed_retry(cpuset_mems_cookie)); + + /* if the nodemask has become invalid, we cannot do anything */ + if (!nodes) + return 0; + nr_pages_per_node = nr_pages / nodes; delta = nr_pages - nodes * nr_pages_per_node; From ddea04157a73e39aa14c344eae3f0bb1d40531c9 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Wed, 19 Aug 2026 10:16:09 +0800 Subject: [PATCH 0273/1012] tools/mm/page_owner_sort: fix --sort option being silently ignored Patch series "tools/mm/page_owner_sort: fix --sort, add module filter, improve usage", v3. This series improves the page_owner_sort tool with a bug fix, a new module-name feature, and better usage text. Patch 1 fixes a long-standing bug where --sort was silently ignored when used without a short option (-a, -m, -p, etc.). The COMP_NO_FLAG case fell through to COMP_NUM and overwrote the sort conditions configured by parse_sort_args(). Patch 2 adds kernel module name support for sort, cull, and filter operations. Page owner stack traces already contain module names in the "function+0xNN/0xNN [module]" format produced by %pS, but page_owner_sort had no way to use them. Records without module frames are assigned "vmlinux". # Aggregate page usage per module ./page_owner_sort input.txt output.txt --cull=mod # Filter to records from xfs module only ./page_owner_sort input.txt output.txt --module xfs # Sort by module name, then by pid descending ./page_owner_sort input.txt output.txt --sort=mod,-pid Patch 3 lists all available sort keys with abbreviations and examples directly in the --sort help section so users no longer need to read the source to discover valid keys. This patch (of 3): When --sort is used without any short option (-a, -m, -p, etc.), compare_flag remains COMP_NO_FLAG. The switch (compare_flag) then falls through to the COMP_NUM case and calls set_single_cmp(), which unconditionally overwrites the sort conditions that parse_sort_args() already configured. This makes --sort silently ineffective unless a short option is also supplied. Split COMP_NO_FLAG out of the COMP_NUM fallthrough so that --sort is respected when no short option is present. Reproduction: # Before fix: ascending order (ignored --sort=-pid) ./page_owner_sort --sort=-pid input.txt output.txt # After fix: descending order as expected Link: https://lore.kernel.org/20260819021611.2910835-1-ye.liu@linux.dev Link: https://lore.kernel.org/20260819021611.2910835-2-ye.liu@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Liu Jing Cc: Yichong Chen --- tools/mm/page_owner_sort.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tools/mm/page_owner_sort.c b/tools/mm/page_owner_sort.c index 22b3b500d33a47..f1dc0763c25cef 100644 --- a/tools/mm/page_owner_sort.c +++ b/tools/mm/page_owner_sort.c @@ -821,6 +821,9 @@ int main(int argc, char **argv) set_single_cmp(compare_stacktrace, SORT_ASC); break; case COMP_NO_FLAG: + if (sc.size > 0) + break; + /* fallthrough */ case COMP_NUM: set_single_cmp(compare_num, SORT_DESC); break; From 503583fb808291e10ff7b3d356715cfa3a941f61 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Wed, 19 Aug 2026 10:16:10 +0800 Subject: [PATCH 0274/1012] tools/mm/page_owner_sort: add module name sort/cull/filter support Page owner stack traces already contain kernel module names in the "[module]" format produced by %pS, but page_owner_sort has no way to sort, cull, or filter by module. Extract the first module name from each record's stack trace using the regex \[([a-zA-Z0-9_]+)\]. Records whose stack traces contain no module frames are assigned "vmlinux". New options: -M Sort by module name --sort=mod Sort by module name (supports +/- prefix) --cull=mod Cull (aggregate) by module name --module Filter to records matching the given module(s) The module field is also printed in cull output when relevant. Link: https://lore.kernel.org/20260819021611.2910835-3-ye.liu@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Cc: David Hildenbrand (Arm) Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Liu Jing Cc: Yichong Chen --- Documentation/mm/page_owner.rst | 8 ++- tools/mm/page_owner_sort.c | 122 +++++++++++++++++++++++++++----- 2 files changed, 110 insertions(+), 20 deletions(-) diff --git a/Documentation/mm/page_owner.rst b/Documentation/mm/page_owner.rst index a6bd3fe6423ad6..bd027377dff6ee 100644 --- a/Documentation/mm/page_owner.rst +++ b/Documentation/mm/page_owner.rst @@ -199,6 +199,7 @@ Usage -p Sort by pid. -P Sort by tgid. -n Sort by task command name. + -M Sort by module name. -r Sort by memory release time. -s Sort by stack trace. -t Sort by times (default). @@ -240,8 +241,10 @@ Usage group ID numbers appear in . --name Select by task command name. This selects the blocks whose task command name appear in . + --module Select by module name. This selects the blocks whose + module name appear in . - , , are single arguments in the form of a comma-separated list, + , , , are single arguments in the form of a comma-separated list, which offers a way to specify individual selecting rules. @@ -249,6 +252,7 @@ Usage ./page_owner_sort --pid=1 ./page_owner_sort --tgid=1,2,3 ./page_owner_sort --name name1,name2 + ./page_owner_sort --module xfs,ext4 STANDARD FORMAT SPECIFIERS ========================== @@ -265,6 +269,7 @@ STANDARD FORMAT SPECIFIERS ft free_ts timestamp of the page when it was released at alloc_ts timestamp of the page when it was allocated ator allocator memory allocator for pages + mod module kernel module name For --cull option: @@ -275,6 +280,7 @@ STANDARD FORMAT SPECIFIERS f free whether the page has been released or not st stacktrace stack trace of the page allocation ator allocator memory allocator for pages + mod module kernel module name Filtering page_owner output ============================ diff --git a/tools/mm/page_owner_sort.c b/tools/mm/page_owner_sort.c index f1dc0763c25cef..a5b61b4abc2f08 100644 --- a/tools/mm/page_owner_sort.c +++ b/tools/mm/page_owner_sort.c @@ -25,11 +25,13 @@ #include #define TASK_COMM_LEN 16 +#define MODULE_NAME_LEN 64 struct block_list { char *txt; char *comm; // task command name char *stacktrace; + char *module; // kernel module name __u64 ts_nsec; int len; int num; @@ -41,7 +43,8 @@ struct block_list { enum FILTER_BIT { FILTER_PID = 1<<1, FILTER_TGID = 1<<2, - FILTER_COMM = 1<<3 + FILTER_COMM = 1<<3, + FILTER_MODULE = 1<<4 }; enum FILTER_RESULT { @@ -55,7 +58,8 @@ enum CULL_BIT { CULL_TGID = 1<<2, CULL_COMM = 1<<3, CULL_STACKTRACE = 1<<4, - CULL_ALLOCATOR = 1<<5 + CULL_ALLOCATOR = 1<<5, + CULL_MODULE = 1<<6 }; enum ALLOCATOR_BIT { ALLOCATOR_CMA = 1<<1, @@ -65,7 +69,8 @@ enum ALLOCATOR_BIT { }; enum ARG_TYPE { ARG_TXT, ARG_COMM, ARG_STACKTRACE, ARG_ALLOC_TS, ARG_CULL_TIME, - ARG_PAGE_NUM, ARG_PID, ARG_TGID, ARG_UNKNOWN, ARG_ALLOCATOR + ARG_PAGE_NUM, ARG_PID, ARG_TGID, ARG_UNKNOWN, ARG_ALLOCATOR, + ARG_MODULE }; enum SORT_ORDER { SORT_ASC = 1, @@ -79,15 +84,18 @@ enum COMP_FLAG { COMP_STACK = 1<<3, COMP_NUM = 1<<4, COMP_TGID = 1<<5, - COMP_COMM = 1<<6 + COMP_COMM = 1<<6, + COMP_MODULE = 1<<7 }; struct filter_condition { pid_t *pids; pid_t *tgids; char **comms; + char **modules; int pids_size; int tgids_size; int comms_size; + int modules_size; }; struct sort_condition { int (**cmps)(const void *, const void *); @@ -101,6 +109,7 @@ static regex_t pid_pattern; static regex_t tgid_pattern; static regex_t comm_pattern; static regex_t ts_nsec_pattern; +static regex_t module_pattern; static struct block_list *list; static int list_size; static int max_size; @@ -184,6 +193,13 @@ static int compare_comm(const void *p1, const void *p2) return strcmp(l1->comm, l2->comm); } +static int compare_module(const void *p1, const void *p2) +{ + const struct block_list *l1 = p1, *l2 = p2; + + return strcmp(l1->module, l2->module); +} + static int compare_ts(const void *p1, const void *p2) { const struct block_list *l1 = p1, *l2 = p2; @@ -207,6 +223,8 @@ static int compare_cull_condition(const void *p1, const void *p2) return compare_tgid(p1, p2); if ((cull & CULL_COMM) && compare_comm(p1, p2)) return compare_comm(p1, p2); + if ((cull & CULL_MODULE) && compare_module(p1, p2)) + return compare_module(p1, p2); if ((cull & CULL_ALLOCATOR) && compare_allocator(p1, p2)) return compare_allocator(p1, p2); return 0; @@ -411,9 +429,33 @@ static char *get_comm(char *buf) return comm_str; } +static char *get_module(char *buf) +{ + char *module_str = malloc(MODULE_NAME_LEN); + regmatch_t pmatch[2]; + int val_len; + + if (!module_str) + return NULL; + memset(module_str, 0, MODULE_NAME_LEN); + if (regexec(&module_pattern, buf, 2, pmatch, REG_NOTBOL) != 0 || pmatch[1].rm_so == -1) { + strcpy(module_str, "vmlinux"); + return module_str; + } + + val_len = pmatch[1].rm_eo - pmatch[1].rm_so; + if ((size_t)val_len >= MODULE_NAME_LEN) + val_len = MODULE_NAME_LEN - 1; + memcpy(module_str, buf + pmatch[1].rm_so, val_len); + module_str[val_len] = '\0'; + + return module_str; +} + static void free_block_list(struct block_list *block) { free(block->comm); + free(block->module); free(block->txt); } @@ -433,6 +475,8 @@ static int get_arg_type(const char *arg) return ARG_ALLOC_TS; else if (!strcmp(arg, "allocator") || !strcmp(arg, "ator")) return ARG_ALLOCATOR; + else if (!strcmp(arg, "module") || !strcmp(arg, "mod")) + return ARG_MODULE; else { return ARG_UNKNOWN; } @@ -483,25 +527,36 @@ static bool match_str_list(const char *str, char **list, int list_size) static enum FILTER_RESULT filter_record(char *buf) { - char *comm; + char *comm, *module; if ((filter & FILTER_PID) && !match_num_list(get_pid(buf), fc.pids, fc.pids_size)) return FILTER_SKIP; if ((filter & FILTER_TGID) && !match_num_list(get_tgid(buf), fc.tgids, fc.tgids_size)) return FILTER_SKIP; - if (!(filter & FILTER_COMM)) + if (!(filter & (FILTER_COMM | FILTER_MODULE))) return FILTER_MATCH; - comm = get_comm(buf); - if (!comm) - return FILTER_ERROR; - - if (!match_str_list(comm, fc.comms, fc.comms_size)) { + if (filter & FILTER_COMM) { + comm = get_comm(buf); + if (!comm) + return FILTER_ERROR; + if (!match_str_list(comm, fc.comms, fc.comms_size)) { + free(comm); + return FILTER_SKIP; + } free(comm); - return FILTER_SKIP; } - free(comm); + if (filter & FILTER_MODULE) { + module = get_module(buf); + if (!module) + return FILTER_ERROR; + if (!match_str_list(module, fc.modules, fc.modules_size)) { + free(module); + return FILTER_SKIP; + } + free(module); + } return FILTER_MATCH; } @@ -547,6 +602,12 @@ static bool add_list(char *buf, int len, char *ext_buf) list[list_size].stacktrace++; list[list_size].ts_nsec = get_ts_nsec(buf); list[list_size].allocator = get_allocator(buf, ext_buf); + list[list_size].module = get_module(buf); + if (!list[list_size].module) { + fprintf(stderr, "Out of memory\n"); + free_block_list(&list[list_size]); + return false; + } list_size++; if (list_size % 1000 == 0) { printf("loaded %d\r", list_size); @@ -573,6 +634,8 @@ static bool parse_cull_args(const char *arg_str) cull |= CULL_STACKTRACE; else if (arg_type == ARG_ALLOCATOR) cull |= CULL_ALLOCATOR; + else if (arg_type == ARG_MODULE) + cull |= CULL_MODULE; else { free_explode(args, size); return false; @@ -635,6 +698,8 @@ static bool parse_sort_args(const char *arg_str) sc.cmps[i] = compare_txt; else if (arg_type == ARG_ALLOCATOR) sc.cmps[i] = compare_allocator; + else if (arg_type == ARG_MODULE) + sc.cmps[i] = compare_module; else { free_explode(args, size); sc.size = 0; @@ -691,7 +756,8 @@ static void usage(void) "-p\t\t\tSort by pid.\n" "-P\t\t\tSort by tgid.\n" "-s\t\t\tSort by the stacktrace.\n" - "-t\t\t\tSort by number of times record is seen (default).\n\n" + "-t\t\t\tSort by number of times record is seen (default).\n" + "-M\t\t\tSort by module name.\n\n" "--pid \t\tSelect by pid. This selects the information" " of\n\t\t\tblocks whose process ID numbers appear in .\n" "--tgid \tSelect by tgid. This selects the information" @@ -700,10 +766,11 @@ static void usage(void) "--name \tSelect by command name. This selects the" " information\n\t\t\tof blocks whose command name appears in" " .\n" - "--cull \t\tCull by user-defined rules. is a " - "single\n\t\t\targument in the form of a comma-separated list " - "with some\n\t\t\tcommon fields predefined (pid, tgid, comm, " - "stacktrace, allocator)\n" + "--module \tSelect by module name. This selects the information\n" + "\t\t\tof blocks whose module name appears in .\n" + "--cull \t\tCull by user-defined rules. is a single\n" + "\t\t\targument in the form of a comma-separated list with some\n" + "\t\t\tcommon fields predefined (pid, tgid, comm, stacktrace, allocator, module)\n" "--sort \t\tSpecify sort order as: [+|-]key[,[+|-]key[,...]]\n" ); } @@ -721,13 +788,14 @@ int main(int argc, char **argv) { "name", required_argument, NULL, 3 }, { "cull", required_argument, NULL, 4 }, { "sort", required_argument, NULL, 5 }, + { "module", required_argument, NULL, 6 }, { "help", no_argument, NULL, 'h' }, { 0, 0, 0, 0}, }; compare_flag = COMP_NO_FLAG; - while ((opt = getopt_long(argc, argv, "admnpstPh", longopts, NULL)) != -1) + while ((opt = getopt_long(argc, argv, "admnpstPMh", longopts, NULL)) != -1) switch (opt) { case 'a': compare_flag |= COMP_ALLOC; @@ -753,6 +821,9 @@ int main(int argc, char **argv) case 'n': compare_flag |= COMP_COMM; break; + case 'M': + compare_flag |= COMP_MODULE; + break; case 'h': usage(); exit(0); @@ -792,6 +863,10 @@ int main(int argc, char **argv) exit(1); } break; + case 6: + filter = filter | FILTER_MODULE; + fc.modules = explode(',', optarg, &fc.modules_size); + break; default: usage(); exit(1); @@ -833,6 +908,9 @@ int main(int argc, char **argv) case COMP_COMM: set_single_cmp(compare_comm, SORT_ASC); break; + case COMP_MODULE: + set_single_cmp(compare_module, SORT_ASC); + break; default: usage(); exit(1); @@ -855,6 +933,8 @@ int main(int argc, char **argv) goto out_comm; if (!check_regcomp(&ts_nsec_pattern, "ts\\s*([0-9]*)\\s*ns")) goto out_ts; + if (!check_regcomp(&module_pattern, "\\+0x[0-9a-f]+/0x[0-9a-f]+\\s*\\[([a-zA-Z0-9_-]+)\\]")) + goto out_module; fstat(fileno(fin), &st); max_size = st.st_size / 100; /* hack ... */ @@ -920,6 +1000,8 @@ int main(int argc, char **argv) fprintf(fout, ", TGID %d", list[i].tgid); if (cull & CULL_COMM || filter & FILTER_COMM) fprintf(fout, ", task_comm_name: %s", list[i].comm); + if (cull & CULL_MODULE || filter & FILTER_MODULE) + fprintf(fout, ", module: %s", list[i].module); if (cull & CULL_ALLOCATOR) { fprintf(fout, ", "); print_allocator(fout, list[i].allocator); @@ -940,6 +1022,8 @@ int main(int argc, char **argv) free_block_list(&list[i]); free(list); } +out_module: + regfree(&module_pattern); out_ts: regfree(&ts_nsec_pattern); out_comm: From 4acacc9e22e46a1b553f86ac86d29ccbe90f703f Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Wed, 19 Aug 2026 10:16:11 +0800 Subject: [PATCH 0275/1012] tools/mm/page_owner_sort: show available sort keys in usage text The --sort option accepts abbreviated or complete key names, but the usage text never listed them. Users had to read the source or the documentation to discover valid keys. List all available keys (full form and abbreviation) with a brief description and examples directly in the --sort help section. Link: https://lore.kernel.org/20260819021611.2910835-4-ye.liu@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Yichong Chen Cc: Liu Jing --- tools/mm/page_owner_sort.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/tools/mm/page_owner_sort.c b/tools/mm/page_owner_sort.c index a5b61b4abc2f08..6c2ac5d9dc4b6e 100644 --- a/tools/mm/page_owner_sort.c +++ b/tools/mm/page_owner_sort.c @@ -772,6 +772,15 @@ static void usage(void) "\t\t\targument in the form of a comma-separated list with some\n" "\t\t\tcommon fields predefined (pid, tgid, comm, stacktrace, allocator, module)\n" "--sort \t\tSpecify sort order as: [+|-]key[,[+|-]key[,...]]\n" + "\t\t\tAvailable keys:\n" + "\t\t\t pid(p), tgid(tg), name(n), stacktrace(st),\n" + "\t\t\t txt(T), alloc_ts(at), allocator(ator), module(mod)\n" + "\t\t\tThe \"+\" is optional since default direction is\n" + "\t\t\tincreasing numerical or lexicographic order.\n" + "\t\t\tMixed use of abbreviated and complete-form is allowed.\n" + "\t\t\tExamples:\n" + "\t\t\t --sort=n,+pid,-tgid\n" + "\t\t\t --sort=mod,at\n" ); } From f14b6fd22f2772fe83609864ce88c785b4a8ffb9 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 19 Aug 2026 18:20:10 -0700 Subject: [PATCH 0276/1012] memcg: trim the per-cpu charge stock instead of draining it Joy reported that an application generating a request/response traffic pattern spends 44.6% to 57.0% of CPU in the memcg charge/uncharge path for a range of message sizes, against 0.27% to 0.71% outside that range. Running from the root memcg, where socket memory accounting is skipped, recovers the performance. Tracing the charge path showed that the application generates a pattern where the write syscall charges one page and the read syscall uncharges two pages on the same CPU. This hits a corner case in the memcg percpu stock code that thrashes the stock continuously. In the memcg percpu stock code, MEMCG_CHARGE_BATCH (64) is both the high watermark and the emptying target, i.e. on a request to charge one page the kernel charges MEMCG_CHARGE_BATCH pages and caches (MEMCG_CHARGE_BATCH - 1) of them in the percpu stock. The following uncharge of 2 pages takes the cached count to (MEMCG_CHARGE_BATCH + 1), and refill_stock() then empties the cache completely. With such a pattern the percpu stock becomes completely ineffective. Instead of a single boundary point for charges, use the technique the page allocator uses for its own percpu caches, which keeps the watermark and the emptying target apart: nr_pcp_free() frees between batch and high - batch pages, leaving at least pcp->batch on the list. Add a high watermark MEMCG_STOCK_HIGH and, once the cached count goes over it, return only the pages above MEMCG_STOCK_LOW. The watermarks are MEMCG_CHARGE_BATCH apart, so a page_counter update still covers a full batch. For now, keep MEMCG_STOCK_HIGH same as MEMCG_CHARGE_BATCH and in future we will reevaluate if it makes sense to increase it. Link: https://lore.kernel.org/20260820012010.2016086-1-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Reported-by: Joy Chaoyue Xiong Acked-by: Michal Hocko Cc: Jakub Kacinski Cc: Johannes Weiner Cc: Joshua Hahn Cc: Muchun Song Cc: Roman Gushchin --- mm/memcontrol.c | 25 +++++++++++++++++++------ 1 file changed, 19 insertions(+), 6 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 05f5338765a995..bfd0a74fac9239 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2048,6 +2048,15 @@ void mem_cgroup_print_oom_group(struct mem_cgroup *memcg) * nr_pages in a single cacheline. This may change in future. */ #define NR_MEMCG_STOCK 7 + +/* + * Watermarks for a charge stock slot, in the spirit of pcp->high and + * pcp->batch: MEMCG_STOCK_HIGH is the high watermark at which a slot is + * trimmed, and it is trimmed down to MEMCG_STOCK_LOW rather than emptied. + */ +#define MEMCG_STOCK_LOW (MEMCG_CHARGE_BATCH / 2) +#define MEMCG_STOCK_HIGH (MEMCG_CHARGE_BATCH) + #define FLUSHING_CACHED_CHARGE 0 struct memcg_stock_pcp { local_trylock_t lock; @@ -2228,17 +2237,18 @@ static void refill_stock(struct mem_cgroup *memcg, unsigned int nr_pages) { struct memcg_stock_pcp *stock; struct mem_cgroup *cached; - uint8_t stock_pages; + unsigned int stock_pages; bool success = false; int empty_slot = -1; int i; /* - * For now limit MEMCG_CHARGE_BATCH to 127 and less. In future if we - * decide to increase it more than 127 then we will need more careful - * handling of nr_pages[] in struct memcg_stock_pcp. + * nr_pages[] is a uint8_t and a slot's count is capped at + * MEMCG_STOCK_HIGH. Raising MEMCG_CHARGE_BATCH beyond 127 would need + * more careful handling of nr_pages[] in struct memcg_stock_pcp. */ BUILD_BUG_ON(MEMCG_CHARGE_BATCH > S8_MAX); + BUILD_BUG_ON(MEMCG_STOCK_HIGH > U8_MAX); VM_WARN_ON_ONCE(mem_cgroup_is_root(memcg)); @@ -2259,9 +2269,12 @@ static void refill_stock(struct mem_cgroup *memcg, unsigned int nr_pages) empty_slot = i; if (memcg == READ_ONCE(stock->cached[i])) { stock_pages = READ_ONCE(stock->nr_pages[i]) + nr_pages; + if (stock_pages > MEMCG_STOCK_HIGH) { + memcg_uncharge(memcg, + stock_pages - MEMCG_STOCK_LOW); + stock_pages = MEMCG_STOCK_LOW; + } WRITE_ONCE(stock->nr_pages[i], stock_pages); - if (stock_pages > MEMCG_CHARGE_BATCH) - drain_stock(stock, i); success = true; break; } From 5d1ada96797be76b8a53d132fcd3e9ff267c7660 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:04 -0700 Subject: [PATCH 0277/1012] memcg: remove v1 soft limit reclaim Patch series "memcg: remove the v1 soft limit", v2. The v1 soft limit was deprecated in v6.12 by commit 569c4f62d84a ("memcg: initiate deprecation of v1 soft limit") and nobody has reported depending on it in the ~21 months since. memory.low and memory.min in v2 have covered the same ground for far longer. The knob has since been made inert by "memcg: make the v1 soft limit knob inert", already queued in mm-hotfixes as a backportable fix for a syzbot report [1]. Nothing can enter the soft limit rbtree anymore, so this series just deletes the machinery that is now dead: the reclaim pass in kswapd and direct reclaim, mem_cgroup_shrink_node() and its tracepoints, the per-node rbtree, lru_gen_soft_reclaim() and the MEMCG_LRU_HEAD op, the per-node tree fields, mem_cgroup->soft_limit, and finally the v1 event ratelimiting which is now down to a single target. memory.soft_limit_in_bytes itself is untouched: writes stay ignored and reads keep returning the maximum value. This patch (of 8): Nothing can put a cgroup on the soft limit rbtree anymore, so the tree is always empty and both callers of memcg1_soft_limit_reclaim() are guaranteed no-ops. Remove the reclaim pass from direct reclaim and from kswapd, along with its implementation. In shrink_zones() this leaves the global reclaim branch with a last_pgdat check that is now redundant with the identical check right below it, so drop it and move the explaining comment down to the check that remains. That check could only ever fire once last_pgdat was set, which implies first_pgdat had already been assigned, so skipping it does not change which node consider_reclaim_throttle() gets. Link: https://lore.kernel.org/20260902174311.1772372-1-shakeel.butt@linux.dev Link: https://lore.kernel.org/20260902174311.1772372-2-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- include/linux/memcontrol.h | 12 --- mm/memcontrol-v1.c | 175 ------------------------------------- mm/vmscan.c | 39 ++------- 3 files changed, 6 insertions(+), 220 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index da625d2edb3bab..11c1fa88d6fd0c 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -1924,10 +1924,6 @@ static inline bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg) /* Cgroup v1-related declarations */ #ifdef CONFIG_MEMCG_V1 -unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order, - gfp_t gfp_mask, - unsigned long *total_scanned); - bool mem_cgroup_oom_synchronize(bool wait); static inline bool task_in_memcg_oom(struct task_struct *p) @@ -1948,14 +1944,6 @@ static inline void mem_cgroup_exit_user_fault(void) } #else /* CONFIG_MEMCG_V1 */ -static inline -unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order, - gfp_t gfp_mask, - unsigned long *total_scanned) -{ - return 0; -} - static inline bool task_in_memcg_oom(struct task_struct *p) { return false; diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index 05ef55cae4dc61..b38b8d0f7f51c5 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -34,13 +34,6 @@ struct mem_cgroup_tree { static struct mem_cgroup_tree soft_limit_tree __read_mostly; -/* - * Maximum loops in mem_cgroup_soft_reclaim(), used for soft - * limit reclaim to prevent infinite loops, if they ever occur. - */ -#define MEM_CGROUP_MAX_RECLAIM_LOOPS 100 -#define MEM_CGROUP_MAX_SOFT_LIMIT_RECLAIM_LOOPS 2 - /* for OOM */ struct mem_cgroup_eventfd_list { struct list_head list; @@ -233,174 +226,6 @@ void memcg1_remove_from_trees(struct mem_cgroup *memcg) } } -static struct mem_cgroup_per_node * -__mem_cgroup_largest_soft_limit_node(struct mem_cgroup_tree_per_node *mctz) -{ - struct mem_cgroup_per_node *mz; - -retry: - mz = NULL; - if (!mctz->rb_rightmost) - goto done; /* Nothing to reclaim from */ - - mz = rb_entry(mctz->rb_rightmost, - struct mem_cgroup_per_node, tree_node); - /* - * Remove the node now but someone else can add it back, - * we will to add it back at the end of reclaim to its correct - * position in the tree. - */ - __mem_cgroup_remove_exceeded(mz, mctz); - if (!soft_limit_excess(mz->memcg) || - !css_tryget(&mz->memcg->css)) - goto retry; -done: - return mz; -} - -static struct mem_cgroup_per_node * -mem_cgroup_largest_soft_limit_node(struct mem_cgroup_tree_per_node *mctz) -{ - struct mem_cgroup_per_node *mz; - - spin_lock_irq(&mctz->lock); - mz = __mem_cgroup_largest_soft_limit_node(mctz); - spin_unlock_irq(&mctz->lock); - return mz; -} - -static int mem_cgroup_soft_reclaim(struct mem_cgroup *root_memcg, - pg_data_t *pgdat, - gfp_t gfp_mask, - unsigned long *total_scanned) -{ - struct mem_cgroup *victim = NULL; - int total = 0; - int loop = 0; - unsigned long excess; - unsigned long nr_scanned; - struct mem_cgroup_reclaim_cookie reclaim = { - .pgdat = pgdat, - }; - - excess = soft_limit_excess(root_memcg); - - while (1) { - victim = mem_cgroup_iter(root_memcg, victim, &reclaim); - if (!victim) { - loop++; - if (loop >= 2) { - /* - * If we have not been able to reclaim - * anything, it might because there are - * no reclaimable pages under this hierarchy - */ - if (!total) - break; - /* - * We want to do more targeted reclaim. - * excess >> 2 is not to excessive so as to - * reclaim too much, nor too less that we keep - * coming back to reclaim from this cgroup - */ - if (total >= (excess >> 2) || - (loop > MEM_CGROUP_MAX_RECLAIM_LOOPS)) - break; - } - continue; - } - total += mem_cgroup_shrink_node(victim, gfp_mask, false, - pgdat, &nr_scanned); - *total_scanned += nr_scanned; - if (!soft_limit_excess(root_memcg)) - break; - } - mem_cgroup_iter_break(root_memcg, victim); - return total; -} - -unsigned long memcg1_soft_limit_reclaim(pg_data_t *pgdat, int order, - gfp_t gfp_mask, - unsigned long *total_scanned) -{ - unsigned long nr_reclaimed = 0; - struct mem_cgroup_per_node *mz, *next_mz = NULL; - unsigned long reclaimed; - int loop = 0; - struct mem_cgroup_tree_per_node *mctz; - unsigned long excess; - - if (lru_gen_enabled()) - return 0; - - if (order > 0) - return 0; - - mctz = soft_limit_tree.rb_tree_per_node[pgdat->node_id]; - - /* - * Do not even bother to check the largest node if the root - * is empty. Do it lockless to prevent lock bouncing. Races - * are acceptable as soft limit is best effort anyway. - */ - if (!mctz || RB_EMPTY_ROOT(&mctz->rb_root)) - return 0; - - /* - * This loop can run a while, specially if mem_cgroup's continuously - * keep exceeding their soft limit and putting the system under - * pressure - */ - do { - if (next_mz) - mz = next_mz; - else - mz = mem_cgroup_largest_soft_limit_node(mctz); - if (!mz) - break; - - reclaimed = mem_cgroup_soft_reclaim(mz->memcg, pgdat, - gfp_mask, total_scanned); - nr_reclaimed += reclaimed; - spin_lock_irq(&mctz->lock); - - /* - * If we failed to reclaim anything from this memory cgroup - * it is time to move on to the next cgroup - */ - next_mz = NULL; - if (!reclaimed) - next_mz = __mem_cgroup_largest_soft_limit_node(mctz); - - excess = soft_limit_excess(mz->memcg); - /* - * One school of thought says that we should not add - * back the node to the tree if reclaim returns 0. - * But our reclaim could return 0, simply because due - * to priority we are exposing a smaller subset of - * memory to reclaim from. Consider this as a longer - * term TODO. - */ - /* If excess == 0, no tree ops */ - __mem_cgroup_insert_exceeded(mz, mctz, excess); - spin_unlock_irq(&mctz->lock); - css_put(&mz->memcg->css); - loop++; - /* - * Could not reclaim anything and there are no more - * mem cgroups to try or we seem to be looping without - * reclaiming anything. - */ - if (!nr_reclaimed && - (next_mz == NULL || - loop > MEM_CGROUP_MAX_SOFT_LIMIT_RECLAIM_LOOPS)) - break; - } while (!nr_reclaimed); - if (next_mz) - css_put(&next_mz->memcg->css); - return nr_reclaimed; -} - static u64 mem_cgroup_move_charge_read(struct cgroup_subsys_state *css, struct cftype *cft) { diff --git a/mm/vmscan.c b/mm/vmscan.c index fdd13299a04a93..0e04eaf64af3ae 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -6439,8 +6439,6 @@ static void shrink_zones(struct zonelist *zonelist, struct scan_control *sc) { struct zoneref *z; struct zone *zone; - unsigned long nr_soft_reclaimed; - unsigned long nr_soft_scanned; gfp_t orig_mask; pg_data_t *last_pgdat = NULL; pg_data_t *first_pgdat = NULL; @@ -6482,35 +6480,17 @@ static void shrink_zones(struct zonelist *zonelist, struct scan_control *sc) sc->compaction_ready = true; continue; } - - /* - * Shrink each node in the zonelist once. If the - * zonelist is ordered by zone (not the default) then a - * node may be shrunk multiple times but in that case - * the user prefers lower zones being preserved. - */ - if (zone->zone_pgdat == last_pgdat) - continue; - - /* - * This steals pages from memory cgroups over softlimit - * and returns the number of reclaimed pages and - * scanned pages. This works for global memory pressure - * and balancing, not for a memcg's limit. - */ - nr_soft_scanned = 0; - nr_soft_reclaimed = memcg1_soft_limit_reclaim(zone->zone_pgdat, - sc->order, sc->gfp_mask, - &nr_soft_scanned); - sc->nr_reclaimed += nr_soft_reclaimed; - sc->nr_scanned += nr_soft_scanned; - /* need some check for avoid more shrink_zone() */ } if (!first_pgdat) first_pgdat = zone->zone_pgdat; - /* See comment about same check for global reclaim above */ + /* + * Shrink each node in the zonelist once. If the zonelist is + * ordered by zone (not the default) then a node may be shrunk + * multiple times but in that case the user prefers lower zones + * being preserved. + */ if (zone->zone_pgdat == last_pgdat) continue; last_pgdat = zone->zone_pgdat; @@ -7171,8 +7151,6 @@ clear_reclaim_active(pg_data_t *pgdat, int highest_zoneidx) static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx) { int i; - unsigned long nr_soft_reclaimed; - unsigned long nr_soft_scanned; unsigned long pflags; unsigned long nr_boost_reclaim; unsigned long zone_boosts[MAX_NR_ZONES] = { 0, }; @@ -7278,12 +7256,7 @@ static int balance_pgdat(pg_data_t *pgdat, int order, int highest_zoneidx) */ kswapd_age_node(pgdat, &sc); - /* Call soft limit reclaim before calling shrink_node. */ sc.nr_scanned = 0; - nr_soft_scanned = 0; - nr_soft_reclaimed = memcg1_soft_limit_reclaim(pgdat, sc.order, - sc.gfp_mask, &nr_soft_scanned); - sc.nr_reclaimed += nr_soft_reclaimed; /* * There should be no need to raise the scanning priority if From cf5abba32367f770fbb348197e54cddc51a24fe7 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:05 -0700 Subject: [PATCH 0278/1012] memcg: remove mem_cgroup_shrink_node() Its only caller was soft limit reclaim, which is gone. Link: https://lore.kernel.org/20260902174311.1772372-3-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- mm/internal.h | 4 ---- mm/vmscan.c | 41 ----------------------------------------- 2 files changed, 45 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 5cc220db907668..e16f1250b25c80 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -78,10 +78,6 @@ unsigned long try_to_free_mem_cgroup_pages(struct mem_cgroup *memcg, gfp_t gfp_mask, unsigned int reclaim_options, int *swappiness); -unsigned long mem_cgroup_shrink_node(struct mem_cgroup *memcg, - gfp_t gfp_mask, bool noswap, - pg_data_t *pgdat, - unsigned long *nr_scanned); #ifdef CONFIG_NUMA extern int sysctl_min_unmapped_ratio; diff --git a/mm/vmscan.c b/mm/vmscan.c index 0e04eaf64af3ae..d66b5cd167d6f2 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -6805,47 +6805,6 @@ unsigned long try_to_free_pages(struct zonelist *zonelist, int order, #ifdef CONFIG_MEMCG -/* Only used by soft limit reclaim. Do not reuse for anything else. */ -unsigned long mem_cgroup_shrink_node(struct mem_cgroup *memcg, - gfp_t gfp_mask, bool noswap, - pg_data_t *pgdat, - unsigned long *nr_scanned) -{ - struct lruvec *lruvec = mem_cgroup_lruvec(memcg, pgdat); - struct scan_control sc = { - .nr_to_reclaim = SWAP_CLUSTER_MAX, - .target_mem_cgroup = memcg, - .may_writepage = 1, - .may_unmap = 1, - .reclaim_idx = MAX_NR_ZONES - 1, - .may_swap = !noswap, - }; - - WARN_ON_ONCE(!current->reclaim_state); - - sc.gfp_mask = (gfp_mask & GFP_RECLAIM_MASK) | - (GFP_HIGHUSER_MOVABLE & ~GFP_RECLAIM_MASK); - - trace_mm_vmscan_memcg_softlimit_reclaim_begin(sc.gfp_mask, - sc.order, - memcg); - - /* - * NOTE: Although we can get the priority field, using it - * here is not a good idea, since it limits the pages we can scan. - * if we don't reclaim here, the shrink_node from balance_pgdat - * will pick up pages from other mem cgroup's as well. We hack - * the priority and make it zero. - */ - shrink_lruvec(lruvec, &sc); - - trace_mm_vmscan_memcg_softlimit_reclaim_end(sc.nr_reclaimed, memcg); - - *nr_scanned = sc.nr_scanned; - - return sc.nr_reclaimed; -} - unsigned long try_to_free_mem_cgroup_pages(struct mem_cgroup *memcg, unsigned long nr_pages, gfp_t gfp_mask, From 3acedb5103c2319b7b5b5519f89620fdaffd7156 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:06 -0700 Subject: [PATCH 0279/1012] memcg: remove the soft limit reclaim tracepoints mm_vmscan_memcg_softlimit_reclaim_begin and mm_vmscan_memcg_softlimit_reclaim_end were only emitted by mem_cgroup_shrink_node(), which is gone, so they can never fire again. Link: https://lore.kernel.org/20260902174311.1772372-4-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- include/trace/events/vmscan.h | 14 -------------- 1 file changed, 14 deletions(-) diff --git a/include/trace/events/vmscan.h b/include/trace/events/vmscan.h index b4bf7b8def1f5f..8a872990b4bee6 100644 --- a/include/trace/events/vmscan.h +++ b/include/trace/events/vmscan.h @@ -214,13 +214,6 @@ DEFINE_EVENT(mm_vmscan_direct_reclaim_begin_template, mm_vmscan_memcg_reclaim_be TP_ARGS(gfp_flags, order, memcg) ); - -DEFINE_EVENT(mm_vmscan_direct_reclaim_begin_template, mm_vmscan_memcg_softlimit_reclaim_begin, - - TP_PROTO(gfp_t gfp_flags, int order, struct mem_cgroup *memcg), - - TP_ARGS(gfp_flags, order, memcg) -); #endif /* CONFIG_MEMCG */ DECLARE_EVENT_CLASS(mm_vmscan_direct_reclaim_end_template, @@ -260,13 +253,6 @@ DEFINE_EVENT(mm_vmscan_direct_reclaim_end_template, mm_vmscan_memcg_reclaim_end, TP_ARGS(nr_reclaimed, memcg) ); - -DEFINE_EVENT(mm_vmscan_direct_reclaim_end_template, mm_vmscan_memcg_softlimit_reclaim_end, - - TP_PROTO(unsigned long nr_reclaimed, struct mem_cgroup *memcg), - - TP_ARGS(nr_reclaimed, memcg) -); #endif /* CONFIG_MEMCG */ TRACE_EVENT(mm_shrink_slab_start, From 64e215e9b86247aa9f3f20d87734a0874f3c955d Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:07 -0700 Subject: [PATCH 0280/1012] memcg: remove the soft limit rbtree With soft limit reclaim gone, the per-node rbtree of cgroups in excess has no readers left. Remove the tree, the helpers maintaining it, and the subsys_initcall that existed only to allocate it. memcg1_check_events() no longer needs to feed it, which also drops the last caller of lru_gen_soft_reclaim(). Link: https://lore.kernel.org/20260902174311.1772372-5-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- mm/memcontrol-v1.c | 176 +-------------------------------------------- mm/memcontrol-v1.h | 2 - mm/memcontrol.c | 1 - 3 files changed, 2 insertions(+), 177 deletions(-) diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index b38b8d0f7f51c5..475f998b764318 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -17,23 +17,6 @@ #include "swap_table.h" #include "memcontrol-v1.h" -/* - * Cgroups above their limits are maintained in a RB-Tree, independent of - * their hierarchy representation - */ - -struct mem_cgroup_tree_per_node { - struct rb_root rb_root; - struct rb_node *rb_rightmost; - spinlock_t lock; -}; - -struct mem_cgroup_tree { - struct mem_cgroup_tree_per_node *rb_tree_per_node[MAX_NUMNODES]; -}; - -static struct mem_cgroup_tree soft_limit_tree __read_mostly; - /* for OOM */ struct mem_cgroup_eventfd_list { struct list_head list; @@ -99,133 +82,6 @@ static struct lockdep_map memcg_oom_lock_dep_map = { DEFINE_SPINLOCK(memcg_oom_lock); -static void __mem_cgroup_insert_exceeded(struct mem_cgroup_per_node *mz, - struct mem_cgroup_tree_per_node *mctz, - unsigned long new_usage_in_excess) -{ - struct rb_node **p = &mctz->rb_root.rb_node; - struct rb_node *parent = NULL; - struct mem_cgroup_per_node *mz_node; - bool rightmost = true; - - if (mz->on_tree) - return; - - mz->usage_in_excess = new_usage_in_excess; - if (!mz->usage_in_excess) - return; - while (*p) { - parent = *p; - mz_node = rb_entry(parent, struct mem_cgroup_per_node, - tree_node); - if (mz->usage_in_excess < mz_node->usage_in_excess) { - p = &(*p)->rb_left; - rightmost = false; - } else { - p = &(*p)->rb_right; - } - } - - if (rightmost) - mctz->rb_rightmost = &mz->tree_node; - - rb_link_node(&mz->tree_node, parent, p); - rb_insert_color(&mz->tree_node, &mctz->rb_root); - mz->on_tree = true; -} - -static void __mem_cgroup_remove_exceeded(struct mem_cgroup_per_node *mz, - struct mem_cgroup_tree_per_node *mctz) -{ - if (!mz->on_tree) - return; - - if (&mz->tree_node == mctz->rb_rightmost) - mctz->rb_rightmost = rb_prev(&mz->tree_node); - - rb_erase(&mz->tree_node, &mctz->rb_root); - mz->on_tree = false; -} - -static void mem_cgroup_remove_exceeded(struct mem_cgroup_per_node *mz, - struct mem_cgroup_tree_per_node *mctz) -{ - unsigned long flags; - - spin_lock_irqsave(&mctz->lock, flags); - __mem_cgroup_remove_exceeded(mz, mctz); - spin_unlock_irqrestore(&mctz->lock, flags); -} - -static unsigned long soft_limit_excess(struct mem_cgroup *memcg) -{ - unsigned long nr_pages = page_counter_read(&memcg->memory); - unsigned long soft_limit = READ_ONCE(memcg->soft_limit); - unsigned long excess = 0; - - if (nr_pages > soft_limit) - excess = nr_pages - soft_limit; - - return excess; -} - -static void memcg1_update_tree(struct mem_cgroup *memcg, int nid) -{ - unsigned long excess; - struct mem_cgroup_per_node *mz; - struct mem_cgroup_tree_per_node *mctz; - - if (lru_gen_enabled()) { - if (soft_limit_excess(memcg)) - lru_gen_soft_reclaim(memcg, nid); - return; - } - - mctz = soft_limit_tree.rb_tree_per_node[nid]; - if (!mctz) - return; - /* - * Necessary to update all ancestors when hierarchy is used. - * because their event counter is not touched. - */ - for (; memcg; memcg = parent_mem_cgroup(memcg)) { - mz = memcg->nodeinfo[nid]; - excess = soft_limit_excess(memcg); - /* - * We have to update the tree if mz is on RB-tree or - * mem is over its softlimit. - */ - if (excess || mz->on_tree) { - unsigned long flags; - - spin_lock_irqsave(&mctz->lock, flags); - /* if on-tree, remove it */ - if (mz->on_tree) - __mem_cgroup_remove_exceeded(mz, mctz); - /* - * Insert again. mz->usage_in_excess will be updated. - * If excess is 0, no tree ops. - */ - __mem_cgroup_insert_exceeded(mz, mctz, excess); - spin_unlock_irqrestore(&mctz->lock, flags); - } - } -} - -void memcg1_remove_from_trees(struct mem_cgroup *memcg) -{ - struct mem_cgroup_tree_per_node *mctz; - struct mem_cgroup_per_node *mz; - int nid; - - for_each_node(nid) { - mz = memcg->nodeinfo[nid]; - mctz = soft_limit_tree.rb_tree_per_node[nid]; - if (mctz) - mem_cgroup_remove_exceeded(mz, mctz); - } -} - static u64 mem_cgroup_move_charge_read(struct cgroup_subsys_state *css, struct cftype *cft) { @@ -336,7 +192,7 @@ static void mem_cgroup_threshold(struct mem_cgroup *memcg) } } -/* Cgroup1: threshold notifications & softlimit tree updates */ +/* Cgroup1: threshold notifications */ /* * Per memcg event counter is incremented at every pagein/pageout. With THP, @@ -405,17 +261,8 @@ static void memcg1_check_events(struct mem_cgroup *memcg, int nid) if (IS_ENABLED(CONFIG_PREEMPT_RT)) return; - /* threshold event is triggered in finer grain than soft limit */ - if (unlikely(memcg1_event_ratelimit(memcg, - MEM_CGROUP_TARGET_THRESH))) { - bool do_softlimit; - - do_softlimit = memcg1_event_ratelimit(memcg, - MEM_CGROUP_TARGET_SOFTLIMIT); + if (unlikely(memcg1_event_ratelimit(memcg, MEM_CGROUP_TARGET_THRESH))) mem_cgroup_threshold(memcg); - if (unlikely(do_softlimit)) - memcg1_update_tree(memcg, nid); - } } void memcg1_commit_charge(struct folio *folio, struct mem_cgroup *memcg) @@ -2391,22 +2238,3 @@ void memcg1_free_events(struct mem_cgroup *memcg) { free_percpu(memcg->events_percpu); } - -static int __init memcg1_init(void) -{ - int node; - - for_each_node(node) { - struct mem_cgroup_tree_per_node *rtpn; - - rtpn = kzalloc_node(sizeof(*rtpn), GFP_KERNEL, node); - - rtpn->rb_root = RB_ROOT; - rtpn->rb_rightmost = NULL; - spin_lock_init(&rtpn->lock); - soft_limit_tree.rb_tree_per_node[node] = rtpn; - } - - return 0; -} -subsys_initcall(memcg1_init); diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index 1e394269c613d8..fd611e66859a32 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -41,7 +41,6 @@ bool memcg1_alloc_events(struct mem_cgroup *memcg); void memcg1_free_events(struct mem_cgroup *memcg); void memcg1_memcg_init(struct mem_cgroup *memcg); -void memcg1_remove_from_trees(struct mem_cgroup *memcg); static inline void memcg1_soft_limit_reset(struct mem_cgroup *memcg) { @@ -98,7 +97,6 @@ static inline bool memcg1_alloc_events(struct mem_cgroup *memcg) { return true; static inline void memcg1_free_events(struct mem_cgroup *memcg) {} static inline void memcg1_memcg_init(struct mem_cgroup *memcg) {} -static inline void memcg1_remove_from_trees(struct mem_cgroup *memcg) {} static inline void memcg1_soft_limit_reset(struct mem_cgroup *memcg) {} static inline void memcg1_css_offline(struct mem_cgroup *memcg) {} diff --git a/mm/memcontrol.c b/mm/memcontrol.c index bfd0a74fac9239..29def037681943 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4429,7 +4429,6 @@ static void mem_cgroup_css_free(struct cgroup_subsys_state *css) vmpressure_cleanup(&memcg->vmpressure); cancel_work_sync(&memcg->high_work); - memcg1_remove_from_trees(memcg); free_shrinker_info(memcg); mem_cgroup_free(memcg); } From 129a0ad2e96886209d8e49faeebc1e730cc22a1c Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:08 -0700 Subject: [PATCH 0281/1012] memcg: remove lru_gen_soft_reclaim() The soft limit rbtree was the only caller. Dropping it leaves MEMCG_LRU_HEAD unreachable, since nothing else ever rotates a memcg with that op, so remove the op too and update the memcg LRU comment. Link: https://lore.kernel.org/20260902174311.1772372-6-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Reviewed-by: T.J. Mercier Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/mmzone.h | 30 +++++++++++------------------- mm/vmscan.c | 16 ++-------------- 2 files changed, 13 insertions(+), 33 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 6acc14b169bbe6..c070b867e2f349 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -638,35 +638,32 @@ struct lru_gen_mm_walk { * For each node, memcgs are divided into two generations: the old and the * young. For each generation, memcgs are randomly sharded into multiple bins * to improve scalability. For each bin, the hlist_nulls is virtually divided - * into three segments: the head, the tail and the default. + * into two segments: the tail and the default. * * An onlining memcg is added to the tail of a random bin in the old generation. * The eviction starts at the head of a random bin in the old generation. The * per-node memcg generation counter, whose reminder (mod MEMCG_NR_GENS) indexes * the old generation, is incremented when all its bins become empty. * - * There are four operations: - * 1. MEMCG_LRU_HEAD, which moves a memcg to the head of a random bin in its - * current generation (old or young) and updates its "seg" to "head"; - * 2. MEMCG_LRU_TAIL, which moves a memcg to the tail of a random bin in its + * There are three operations: + * 1. MEMCG_LRU_TAIL, which moves a memcg to the tail of a random bin in its * current generation (old or young) and updates its "seg" to "tail"; - * 3. MEMCG_LRU_OLD, which moves a memcg to the head of a random bin in the old + * 2. MEMCG_LRU_OLD, which moves a memcg to the head of a random bin in the old * generation, updates its "gen" to "old" and resets its "seg" to "default"; - * 4. MEMCG_LRU_YOUNG, which moves a memcg to the tail of a random bin in the + * 3. MEMCG_LRU_YOUNG, which moves a memcg to the tail of a random bin in the * young generation, updates its "gen" to "young" and resets its "seg" to * "default". * * The events that trigger the above operations are: - * 1. Exceeding the soft limit, which triggers MEMCG_LRU_HEAD; - * 2. The first attempt to reclaim a memcg below low, which triggers + * 1. The first attempt to reclaim a memcg below low, which triggers * MEMCG_LRU_TAIL; - * 3. The first attempt to reclaim a memcg offlined or below reclaimable size + * 2. The first attempt to reclaim a memcg offlined or below reclaimable size * threshold, which triggers MEMCG_LRU_TAIL; - * 4. The second attempt to reclaim a memcg offlined or below reclaimable size + * 3. The second attempt to reclaim a memcg offlined or below reclaimable size * threshold, which triggers MEMCG_LRU_YOUNG; - * 5. Attempting to reclaim a memcg below min, which triggers MEMCG_LRU_YOUNG; - * 6. Finishing the aging on the eviction path, which triggers MEMCG_LRU_YOUNG; - * 7. Offlining a memcg, which triggers MEMCG_LRU_OLD. + * 4. Attempting to reclaim a memcg below min, which triggers MEMCG_LRU_YOUNG; + * 5. Finishing the aging on the eviction path, which triggers MEMCG_LRU_YOUNG; + * 6. Offlining a memcg, which triggers MEMCG_LRU_OLD. * * Notes: * 1. Memcg LRU only applies to global reclaim, and the round-robin incrementing @@ -699,7 +696,6 @@ void lru_gen_exit_memcg(struct mem_cgroup *memcg); void lru_gen_online_memcg(struct mem_cgroup *memcg); void lru_gen_offline_memcg(struct mem_cgroup *memcg); void lru_gen_release_memcg(struct mem_cgroup *memcg); -void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid); void max_lru_gen_memcg(struct mem_cgroup *memcg, int nid); bool recheck_lru_gen_max_memcg(struct mem_cgroup *memcg, int nid); void lru_gen_reparent_memcg(struct mem_cgroup *memcg, struct mem_cgroup *parent, int nid); @@ -740,10 +736,6 @@ static inline void lru_gen_release_memcg(struct mem_cgroup *memcg) { } -static inline void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid) -{ -} - static inline void max_lru_gen_memcg(struct mem_cgroup *memcg, int nid) { } diff --git a/mm/vmscan.c b/mm/vmscan.c index d66b5cd167d6f2..deb087c57007db 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4373,7 +4373,6 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) /* see the comment on MEMCG_NR_GENS */ enum { MEMCG_LRU_NOP, - MEMCG_LRU_HEAD, MEMCG_LRU_TAIL, MEMCG_LRU_OLD, MEMCG_LRU_YOUNG, @@ -4395,9 +4394,7 @@ static void lru_gen_rotate_memcg(struct lruvec *lruvec, int op) new = old = lruvec->lrugen.gen; /* see the comment on MEMCG_NR_GENS */ - if (op == MEMCG_LRU_HEAD) - seg = MEMCG_LRU_HEAD; - else if (op == MEMCG_LRU_TAIL) + if (op == MEMCG_LRU_TAIL) seg = MEMCG_LRU_TAIL; else if (op == MEMCG_LRU_OLD) new = get_memcg_gen(pgdat->memcg_lru.seq); @@ -4411,7 +4408,7 @@ static void lru_gen_rotate_memcg(struct lruvec *lruvec, int op) hlist_nulls_del_rcu(&lruvec->lrugen.list); - if (op == MEMCG_LRU_HEAD || op == MEMCG_LRU_OLD) + if (op == MEMCG_LRU_OLD) hlist_nulls_add_head_rcu(&lruvec->lrugen.list, &pgdat->memcg_lru.fifo[new][bin]); else hlist_nulls_add_tail_rcu(&lruvec->lrugen.list, &pgdat->memcg_lru.fifo[new][bin]); @@ -4489,15 +4486,6 @@ void lru_gen_release_memcg(struct mem_cgroup *memcg) } } -void lru_gen_soft_reclaim(struct mem_cgroup *memcg, int nid) -{ - struct lruvec *lruvec = get_lruvec(memcg, nid); - - /* see the comment on MEMCG_NR_GENS */ - if (READ_ONCE(lruvec->lrugen.seg) != MEMCG_LRU_HEAD) - lru_gen_rotate_memcg(lruvec, MEMCG_LRU_HEAD); -} - bool recheck_lru_gen_max_memcg(struct mem_cgroup *memcg, int nid) { struct lruvec *lruvec = get_lruvec(memcg, nid); From 63413f8e9caaa4cd83da4df6c02a12ee8e004c5b Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:09 -0700 Subject: [PATCH 0282/1012] memcg: remove the per-node soft limit tree fields tree_node, usage_in_excess and on_tree only existed for the soft limit rbtree. They also doubled as the buffer between the read-mostly head of struct mem_cgroup_per_node and its update-often tail, so replace them with the explicit padding that CONFIG_MEMCG_V1=n already used. Link: https://lore.kernel.org/20260902174311.1772372-7-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- include/linux/memcontrol.h | 13 ------------- 1 file changed, 13 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 11c1fa88d6fd0c..1eababed16f532 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -95,20 +95,7 @@ struct mem_cgroup_per_node { struct lruvec_stats *lruvec_stats; struct shrinker_info __rcu *shrinker_info; -#ifdef CONFIG_MEMCG_V1 - /* - * Memcg-v1 only stuff in middle as buffer between read mostly fields - * and update often fields to avoid false sharing. If v1 stuff is - * not present, an explicit padding is needed. - */ - - struct rb_node tree_node; /* RB tree node */ - unsigned long usage_in_excess;/* Set to the value by which */ - /* the soft limit is exceeded*/ - bool on_tree; -#else CACHELINE_PADDING(_pad1_); -#endif /* Fields which get updated often at the end. */ struct lruvec lruvec; From 5593b4fc46219b6042ab45058b5c04f1cc41b173 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:10 -0700 Subject: [PATCH 0283/1012] memcg: remove mem_cgroup->soft_limit Nothing reads it anymore, so the field and the helper that reset it on css alloc and css reset can go. Link: https://lore.kernel.org/20260902174311.1772372-8-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- include/linux/memcontrol.h | 2 -- mm/memcontrol-v1.h | 6 ------ mm/memcontrol.c | 2 -- 3 files changed, 10 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 1eababed16f532..c799926435560f 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -280,8 +280,6 @@ struct mem_cgroup { struct memcg1_events_percpu __percpu *events_percpu; - unsigned long soft_limit; - /* protected by memcg_oom_lock */ bool oom_lock; int under_oom; diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index fd611e66859a32..f48d0e22e615b6 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -42,11 +42,6 @@ void memcg1_free_events(struct mem_cgroup *memcg); void memcg1_memcg_init(struct mem_cgroup *memcg); -static inline void memcg1_soft_limit_reset(struct mem_cgroup *memcg) -{ - WRITE_ONCE(memcg->soft_limit, PAGE_COUNTER_MAX); -} - struct cgroup_taskset; void memcg1_css_offline(struct mem_cgroup *memcg); @@ -97,7 +92,6 @@ static inline bool memcg1_alloc_events(struct mem_cgroup *memcg) { return true; static inline void memcg1_free_events(struct mem_cgroup *memcg) {} static inline void memcg1_memcg_init(struct mem_cgroup *memcg) {} -static inline void memcg1_soft_limit_reset(struct mem_cgroup *memcg) {} static inline void memcg1_css_offline(struct mem_cgroup *memcg) {} static inline bool memcg1_oom_prepare(struct mem_cgroup *memcg, bool *locked) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 29def037681943..bce3962dba5752 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4257,7 +4257,6 @@ mem_cgroup_css_alloc(struct cgroup_subsys_state *parent_css) return ERR_CAST(memcg); page_counter_set_high(&memcg->memory, PAGE_COUNTER_MAX); - memcg1_soft_limit_reset(memcg); #ifdef CONFIG_ZSWAP memcg->zswap_max = PAGE_COUNTER_MAX; WRITE_ONCE(memcg->zswap_writeback, true); @@ -4464,7 +4463,6 @@ static void mem_cgroup_css_reset(struct cgroup_subsys_state *css) page_counter_set_min(&memcg->memory, 0); page_counter_set_low(&memcg->memory, 0); page_counter_set_high(&memcg->memory, PAGE_COUNTER_MAX); - memcg1_soft_limit_reset(memcg); page_counter_set_high(&memcg->swap, PAGE_COUNTER_MAX); memcg_wb_domain_size_changed(memcg); } From 46cce9b485853e0f97167238848785aec64c237b Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Wed, 2 Sep 2026 10:43:11 -0700 Subject: [PATCH 0284/1012] memcg: simplify v1 event ratelimiting Thresholds are the only periodic v1 event left, so the target enum, the per-cpu target array and the switch in memcg1_event_ratelimit() all collapse to a single counter. memcg1_check_events() no longer needs a node id either, which lets memcg1_uncharge_batch() drop its nid argument and struct uncharge_gather drop the field feeding it. Link: https://lore.kernel.org/20260902174311.1772372-9-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Muchun Song Cc: Roman Gushchin Cc: T.J. Mercier --- mm/memcontrol-v1.c | 43 +++++++++++-------------------------------- mm/memcontrol-v1.h | 4 ++-- mm/memcontrol.c | 4 +--- 3 files changed, 14 insertions(+), 37 deletions(-) diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index 475f998b764318..bf2c7d53b01b1c 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -200,15 +200,9 @@ static void mem_cgroup_threshold(struct mem_cgroup *memcg) * to trigger some periodic events. This is straightforward and better * than using jiffies etc. to handle periodic memcg event. */ -enum mem_cgroup_events_target { - MEM_CGROUP_TARGET_THRESH, - MEM_CGROUP_TARGET_SOFTLIMIT, - MEM_CGROUP_NTARGETS, -}; - struct memcg1_events_percpu { unsigned long nr_page_events; - unsigned long targets[MEM_CGROUP_NTARGETS]; + unsigned long threshold_target; }; static void memcg1_charge_statistics(struct mem_cgroup *memcg, int nr_pages) @@ -225,43 +219,28 @@ static void memcg1_charge_statistics(struct mem_cgroup *memcg, int nr_pages) } #define THRESHOLDS_EVENTS_TARGET 128 -#define SOFTLIMIT_EVENTS_TARGET 1024 -static bool memcg1_event_ratelimit(struct mem_cgroup *memcg, - enum mem_cgroup_events_target target) +static bool memcg1_event_ratelimit(struct mem_cgroup *memcg) { unsigned long val, next; val = __this_cpu_read(memcg->events_percpu->nr_page_events); - next = __this_cpu_read(memcg->events_percpu->targets[target]); + next = __this_cpu_read(memcg->events_percpu->threshold_target); /* from time_after() in jiffies.h */ if ((long)(next - val) < 0) { - switch (target) { - case MEM_CGROUP_TARGET_THRESH: - next = val + THRESHOLDS_EVENTS_TARGET; - break; - case MEM_CGROUP_TARGET_SOFTLIMIT: - next = val + SOFTLIMIT_EVENTS_TARGET; - break; - default: - break; - } - __this_cpu_write(memcg->events_percpu->targets[target], next); + __this_cpu_write(memcg->events_percpu->threshold_target, + val + THRESHOLDS_EVENTS_TARGET); return true; } return false; } -/* - * Check events in order. - * - */ -static void memcg1_check_events(struct mem_cgroup *memcg, int nid) +static void memcg1_check_events(struct mem_cgroup *memcg) { if (IS_ENABLED(CONFIG_PREEMPT_RT)) return; - if (unlikely(memcg1_event_ratelimit(memcg, MEM_CGROUP_TARGET_THRESH))) + if (unlikely(memcg1_event_ratelimit(memcg))) mem_cgroup_threshold(memcg); } @@ -271,7 +250,7 @@ void memcg1_commit_charge(struct folio *folio, struct mem_cgroup *memcg) local_irq_save(flags); memcg1_charge_statistics(memcg, folio_nr_pages(folio)); - memcg1_check_events(memcg, folio_nid(folio)); + memcg1_check_events(memcg); local_irq_restore(flags); } @@ -344,7 +323,7 @@ void __memcg1_swapout(struct folio *folio, struct swap_cluster_info *ci) VM_WARN_ON_IRQS_ENABLED(); memcg1_charge_statistics(memcg, -folio_nr_pages(folio)); preempt_enable_nested(); - memcg1_check_events(memcg, folio_nid(folio)); + memcg1_check_events(memcg); rcu_read_unlock(); obj_cgroup_put(objcg); @@ -398,14 +377,14 @@ void memcg1_swapin(struct folio *folio) #endif void memcg1_uncharge_batch(struct mem_cgroup *memcg, unsigned long pgpgout, - unsigned long nr_memory, int nid) + unsigned long nr_memory) { unsigned long flags; local_irq_save(flags); count_memcg_events(memcg, PGPGOUT, pgpgout); __this_cpu_add(memcg->events_percpu->nr_page_events, nr_memory); - memcg1_check_events(memcg, nid); + memcg1_check_events(memcg); local_irq_restore(flags); } diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index f48d0e22e615b6..b9a21f0fd2c3ac 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -59,7 +59,7 @@ void memcg1_oom_recover(struct mem_cgroup *memcg); void memcg1_commit_charge(struct folio *folio, struct mem_cgroup *memcg); void memcg1_uncharge_batch(struct mem_cgroup *memcg, unsigned long pgpgout, - unsigned long nr_memory, int nid); + unsigned long nr_memory); void memcg1_stat_format(struct mem_cgroup *memcg, struct seq_buf *s); void reparent_memcg1_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent); @@ -107,7 +107,7 @@ static inline void memcg1_commit_charge(struct folio *folio, static inline void memcg1_uncharge_batch(struct mem_cgroup *memcg, unsigned long pgpgout, - unsigned long nr_memory, int nid) {} + unsigned long nr_memory) {} static inline void memcg1_stat_format(struct mem_cgroup *memcg, struct seq_buf *s) {} diff --git a/mm/memcontrol.c b/mm/memcontrol.c index bce3962dba5752..30636b9d96739e 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5328,7 +5328,6 @@ struct uncharge_gather { unsigned long nr_memory; unsigned long pgpgout; unsigned long nr_kmem; - int nid; }; static inline void uncharge_gather_clear(struct uncharge_gather *ug) @@ -5351,7 +5350,7 @@ static void uncharge_batch(const struct uncharge_gather *ug) memcg1_oom_recover(memcg); } - memcg1_uncharge_batch(memcg, ug->pgpgout, ug->nr_memory, ug->nid); + memcg1_uncharge_batch(memcg, ug->pgpgout, ug->nr_memory); rcu_read_unlock(); /* drop reference from uncharge_folio */ @@ -5380,7 +5379,6 @@ static void uncharge_folio(struct folio *folio, struct uncharge_gather *ug) uncharge_gather_clear(ug); } ug->objcg = objcg; - ug->nid = folio_nid(folio); /* pairs with obj_cgroup_put in uncharge_batch */ obj_cgroup_get(objcg); From 43a77f9d579f34233a7d334cd26f600324aa6c3d Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:00 -0400 Subject: [PATCH 0285/1012] mm/page_io: convert write completion handlers to folios Patch series "mm/page_io: folio conversion cleanups", v2. Convert the remaining struct page usage in mm/page_io.c to folios. This removes one of the last callers of end_page_writeback(), along with the last caller of ClearPageReclaim(). This allows us to remove the PG_reclaim page accessors entirely. Clean up a few other stale references to pages throughout while at it. The rename of mm/page_io.c to mm/swap_io.c is deferred to a separate series. This patch (of 6): Convert swap_write_end() and swap_fs_write_complete() to operate on folios directly instead of going through the folio-compat page APIs. This removes calls to end_page_writeback() and set_page_dirty(), and the last caller of ClearPageReclaim(), saving two calls to compound_head() per folio on the write error path. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-0-649728091117@columbia.edu Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-1-649728091117@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Baoquan He Cc: Barry Song Cc: Chengming Zhou Cc: Chris Li Cc: Christoph Hellwig Cc: David Hildenbrand Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/page_io.c | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index 88962571cb931e..fbcf58ff292d8b 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -497,13 +497,13 @@ static void swap_write_end(struct swap_iocb *sio, bool failed) int p; for (p = 0; p < sio->nr_bvecs; p++) { - struct page *page = sio->bvecs[p].bv_page; + struct folio *folio = bvec_folio(&sio->bvecs[p]); if (failed) { - set_page_dirty(page); - ClearPageReclaim(page); + folio_mark_dirty(folio); + folio_clear_reclaim(folio); } - end_page_writeback(page); + folio_end_writeback(folio); } mempool_free(sio, sio_pool); } @@ -514,16 +514,16 @@ static void swap_fs_write_complete(struct kiocb *iocb, long ret) bool failed = ret != sio->len; if (failed) { - struct page *page = sio->bvecs[0].bv_page; + struct folio *folio = bvec_folio(&sio->bvecs[0]); /* * In the case of swap-over-nfs, this can be a temporary failure * if the system has limited memory for allocating transmit - * buffers. Mark the page dirty and avoid + * buffers. Mark the folio dirty and avoid * folio_rotate_reclaimable but rate-limit the messages. */ pr_err_ratelimited("Write error %ld on dio swapfile (%llu)\n", - ret, swap_dev_pos(page_swap_entry(page))); + ret, swap_dev_pos(folio->swap)); } swap_write_end(sio, failed); From 864dfaf43d6cde62cc06d8e92bcba1f485008a03 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:01 -0400 Subject: [PATCH 0286/1012] mm: remove PageReclaim This flag is now only used on folios, so we can remove all the page accessors. folio_test_clear_reclaim() is not used, so don't add FOLIO_TEST_CLEAR_FLAG() for it. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-2-649728091117@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) --- include/linux/page-flags.h | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index 7a863572adce79..ae2ebaed6d4d96 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -593,8 +593,7 @@ TESTPAGEFLAG(Writeback, writeback, PF_NO_TAIL) FOLIO_FLAG(mappedtodisk, FOLIO_HEAD_PAGE) /* PG_readahead is only used for reads; PG_reclaim is only for writes */ -PAGEFLAG(Reclaim, reclaim, PF_NO_TAIL) - TESTCLEARFLAG(Reclaim, reclaim, PF_NO_TAIL) +FOLIO_FLAG(reclaim, FOLIO_HEAD_PAGE) FOLIO_FLAG(readahead, FOLIO_HEAD_PAGE) FOLIO_TEST_CLEAR_FLAG(readahead, FOLIO_HEAD_PAGE) From eb8c5f4e1e3755df1b494edaee5182ffaab1fef0 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:02 -0400 Subject: [PATCH 0287/1012] mm/page_io: use swap entries directly in zeromap helpers Increment swp_entry_t::val directly instead of recomputing each entry with page_swap_entry(). This removes the last struct page usage in page_io.c and saves one call to compound_head() per page. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-3-649728091117@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) --- mm/page_io.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index fbcf58ff292d8b..dd95cdb0cd7a18 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -160,7 +160,7 @@ static void swap_zeromap_folio_set(struct folio *folio) struct obj_cgroup *objcg = get_obj_cgroup_from_folio(folio); int nr_pages = folio_nr_pages(folio); struct swap_cluster_info *ci; - swp_entry_t entry; + swp_entry_t entry = folio->swap; unsigned int i; VM_WARN_ON_ONCE_FOLIO(!folio_test_swapcache(folio), folio); @@ -168,8 +168,8 @@ static void swap_zeromap_folio_set(struct folio *folio) ci = swap_cluster_get_and_lock(folio); for (i = 0; i < folio_nr_pages(folio); i++) { - entry = page_swap_entry(folio_page(folio, i)); __swap_table_set_zero(ci, swp_cluster_offset(entry)); + entry.val++; } swap_cluster_unlock(ci); @@ -183,7 +183,7 @@ static void swap_zeromap_folio_set(struct folio *folio) static void swap_zeromap_folio_clear(struct folio *folio) { struct swap_cluster_info *ci; - swp_entry_t entry; + swp_entry_t entry = folio->swap; unsigned int i; VM_WARN_ON_ONCE_FOLIO(!folio_test_swapcache(folio), folio); @@ -191,8 +191,8 @@ static void swap_zeromap_folio_clear(struct folio *folio) ci = swap_cluster_get_and_lock(folio); for (i = 0; i < folio_nr_pages(folio); i++) { - entry = page_swap_entry(folio_page(folio, i)); __swap_table_clear_zero(ci, swp_cluster_offset(entry)); + entry.val++; } swap_cluster_unlock(ci); } From 05768b7ebb745b227370e152cdd149e6b9634efe Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:03 -0400 Subject: [PATCH 0288/1012] mm/page_io: rename bio_associate_blkg_from_page() This function takes a folio. Rename it to bio_associate_blkg_from_folio() accordingly. While at it, convert the macro in the !CONFIG_MEMCG || !CONFIG_BLK_CGROUP case to a function. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-4-649728091117@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) --- mm/page_io.c | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index dd95cdb0cd7a18..295cc6ac6244af 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -277,7 +277,7 @@ static bool folio_blkg_can_merge(struct folio *folio, struct folio *prev_folio) return can_merge; } -static void bio_associate_blkg_from_page(struct bio *bio, struct folio *folio) +static void bio_associate_blkg_from_folio(struct bio *bio, struct folio *folio) { struct cgroup_subsys_state *css; @@ -298,7 +298,9 @@ static bool folio_blkg_can_merge(struct folio *folio, struct folio *prev_folio) { return true; } -#define bio_associate_blkg_from_page(bio, folio) do { } while (0) +static void bio_associate_blkg_from_folio(struct bio *bio, struct folio *folio) +{ +} #endif /* CONFIG_MEMCG && CONFIG_BLK_CGROUP */ static mempool_t *sio_pool; @@ -596,7 +598,7 @@ static void swap_bdev_submit_write(struct swap_io_ctx *ctx) REQ_OP_WRITE | REQ_SWAP); bio->bi_iter.bi_size = sio->len; bio->bi_iter.bi_sector = swap_folio_sector(bio_first_folio_all(bio)); - bio_associate_blkg_from_page(bio, bio_first_folio_all(bio)); + bio_associate_blkg_from_folio(bio, bio_first_folio_all(bio)); if (ctx->sis->flags & SWP_SYNCHRONOUS_IO) { submit_bio_wait(bio); From c972362b30932d17fd4f933dd753b825e13ad29b Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:04 -0400 Subject: [PATCH 0289/1012] mm/page_io: refer to folios in swap_writeout() comments swap_writeout() operates on folios, not pages. Update its comments accordingly. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-5-649728091117@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) --- mm/page_io.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index 295cc6ac6244af..36466159cf191e 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -209,7 +209,7 @@ int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) goto out_unlock; /* - * Arch code may have to preserve more data than just the page + * Arch code may have to preserve more data than just the folio * contents, e.g. memory tags. */ ret = arch_prepare_to_swap(folio); @@ -220,7 +220,7 @@ int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) /* * Use the swap table zero mark to avoid doing IO for zero-filled - * pages. The zero mark is protected by the cluster lock, which is + * folios. The zero mark is protected by the cluster lock, which is * acquired internally by swap_zeromap_folio_set/clear. */ if (is_folio_zero_filled(folio)) { From b6a192b7c84e5ba5fbe260f647e2acb1b7384350 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 29 Aug 2026 15:09:05 -0400 Subject: [PATCH 0290/1012] mm/swap: rename __swap_writepage() to __swap_writeout() Commit 84798514db50 ("mm: Remove swap_writepage() and shmem_writepage()") renamed swap_writepage() to swap_writeout(). Rename __swap_writepage(), which operates on a folio, to match its caller. Update a stale reference to swap_writepage() in swapfile.c as well. Link: https://lore.kernel.org/20260829-b4-page_io-folios-v2-6-649728091117@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Matthew Wilcox (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) --- mm/page_io.c | 4 ++-- mm/swap.h | 2 +- mm/swapfile.c | 2 +- mm/zswap.c | 2 +- 4 files changed, 5 insertions(+), 5 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index 36466159cf191e..1da4ff484f0971 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -248,7 +248,7 @@ int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) } rcu_read_unlock(); - __swap_writepage(ctx, folio); + __swap_writeout(ctx, folio); return 0; out_unlock: folio_unlock(folio); @@ -369,7 +369,7 @@ static void swap_add_folio(struct swap_io_ctx *ctx, struct folio *folio, int rw) } } -void __swap_writepage(struct swap_io_ctx *ctx, struct folio *folio) +void __swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) { VM_BUG_ON_FOLIO(!folio_test_swapcache(folio), folio); diff --git a/mm/swap.h b/mm/swap.h index fddba7a87500a4..0b5d507739bcb6 100644 --- a/mm/swap.h +++ b/mm/swap.h @@ -258,7 +258,7 @@ void swap_read_folio(struct swap_io_ctx *ctx, struct folio *folio); void swap_read_submit(struct swap_io_ctx *ctx); void swap_write_submit(struct swap_io_ctx *ctx); int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio); -void __swap_writepage(struct swap_io_ctx *ctx, struct folio *folio); +void __swap_writeout(struct swap_io_ctx *ctx, struct folio *folio); /* linux/mm/swap_state.c */ extern struct address_space swap_space __read_mostly; diff --git a/mm/swapfile.c b/mm/swapfile.c index 601979b97f95b2..3b2279a16d1cf8 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -2928,7 +2928,7 @@ EXPORT_SYMBOL_GPL(add_swap_extent); /* * A `swap extent' is a simple thing which maps a contiguous range of pages * onto a contiguous range of disk blocks. A rbtree of swap extents is - * built at swapon time and is then used at swap_writepage/swap_read_folio + * built at swapon time and is then used at swap_writeout/swap_read_folio * time for locating where on disk a page belongs. * * If the swapfile is an S_ISBLK block device, a single extent is installed. diff --git a/mm/zswap.c b/mm/zswap.c index c1dc60926bad99..fc869d60ef1b48 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1037,7 +1037,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry, folio_set_reclaim(folio); /* start writeback */ - __swap_writepage(&ctx, folio); + __swap_writeout(&ctx, folio); swap_write_submit(&ctx); out: From 2365961cfd7cedceee01a8c9f6ad640cbc580e10 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Sat, 29 Aug 2026 15:42:03 +0800 Subject: [PATCH 0291/1012] mm/mglru: make type fallback logic explicit in isolate_folios() Patch series "mm/mglru: clean up isolate_folios for readability and clarity", v2. Right now, isolate_folios() is quite difficult to follow: 1. It uses for_each_evictable_type(i, swappiness) to iterate over the types, but 'i' is not actually used as the type within the loop body. 2. It retries the same type when folios were scanned but none could be isolated, but the retry is implemented in a rather subtle way that is difficult to understand. This patchset makes both behaviors explicit and much easier to follow. There are no functional changes for swappiness values from 1 to 200. There is a slight functional change for 0 and 201: with the existing code, there is no chance to retry for these values because for_each_evictable_type() only iterates once. After this patch, 0 and 201 have behavior that is more consistent with the 1-200 range. This patch (of 2): The for_each_evictable_type() loop in isolate_folios() is misleading: it does not actually iterate over each evictable type. Instead, get_type_to_scan() selects the type to scan, while the iterator `i` merely bounds the number of attempts. Make the fallback behavior explicit in the code and remove the opaque for_each_evictable_type(i, swappiness). Link: https://lore.kernel.org/20260829074204.45304-1-baohua@kernel.org Link: https://lore.kernel.org/20260829074204.45304-2-baohua@kernel.org Signed-off-by: Ridong Chen Co-developed-by: Barry Song (Xiaomi) Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reviewed-by: Lian Wang Reviewed-by: Baoquan He Cc: Axel Rasmussen Cc: Baolin Wang Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 46 ++++++++++++++++++++++++++-------------------- 1 file changed, 26 insertions(+), 20 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index deb087c57007db..1b40c63706f2ed 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4826,35 +4826,41 @@ static int get_type_to_scan(struct lruvec *lruvec, int swappiness) return positive_ctrl_err(&sp, &pv); } +static inline bool is_single_type_reclaim(int swappiness) +{ + return swappiness == MIN_SWAPPINESS || + swappiness == SWAPPINESS_ANON_ONLY; +} + static int isolate_folios(unsigned long nr_to_scan, struct lruvec *lruvec, struct scan_control *sc, int swappiness, struct list_head *list, int *isolated, int *isolate_type, int *isolate_scanned) { - int i; - int total_scanned = 0; + bool type_fallback_allowed = !is_single_type_reclaim(swappiness); int type = get_type_to_scan(lruvec, swappiness); + int total_scanned = 0, scanned, tier; - for_each_evictable_type(i, swappiness) { - int scanned; - int tier = get_tier_idx(lruvec, type); +retry: + tier = get_tier_idx(lruvec, type); + scanned = scan_folios(nr_to_scan, lruvec, sc, + type, tier, list, isolated); - scanned = scan_folios(nr_to_scan, lruvec, sc, - type, tier, list, isolated); + total_scanned += scanned; + if (*isolated) { + *isolate_type = type; + *isolate_scanned = scanned; + return total_scanned; + } - total_scanned += scanned; - if (*isolated) { - *isolate_type = type; - *isolate_scanned = scanned; - break; - } - /* - * If scanned > 0 and isolated == 0, avoid falling back to the - * other type, as this type remains sufficient. Falling back - * too readily can disrupt the positive_ctrl_err() bias. - */ - if (!scanned) - type = !type; + /* + * We are running out of the current reclaim type. Fall back to + * the other type if allowed. + */ + if (!scanned && type_fallback_allowed) { + type = !type; + type_fallback_allowed = false; + goto retry; } return total_scanned; From a9cfc27062db6be42b03b83b245a11fe284d3558 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Sat, 29 Aug 2026 15:42:04 +0800 Subject: [PATCH 0292/1012] mm/mglru: make retry logic explicit in isolate_folios() The existing mainline code retries the same type once in a rather subtle way. `for_each_evictable_type()` may provide one more iteration, allowing the same type to be retried if we scanned some folios but failed to isolate any due to protections, promotions, or races. This patch makes the retry behavior explicit. Link: https://lore.kernel.org/20260829074204.45304-3-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/mm/vmscan.c b/mm/vmscan.c index 1b40c63706f2ed..413efe44d1f695 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4840,6 +4840,7 @@ static int isolate_folios(unsigned long nr_to_scan, struct lruvec *lruvec, bool type_fallback_allowed = !is_single_type_reclaim(swappiness); int type = get_type_to_scan(lruvec, swappiness); int total_scanned = 0, scanned, tier; + bool tried = false; retry: tier = get_tier_idx(lruvec, type); @@ -4859,9 +4860,18 @@ static int isolate_folios(unsigned long nr_to_scan, struct lruvec *lruvec, */ if (!scanned && type_fallback_allowed) { type = !type; + tried = true; type_fallback_allowed = false; goto retry; } + /* + * We scanned some folios but failed to isolate any due to promotions, + * protections, or races. Retry once to avoid a larger loop. + */ + if (scanned && !tried) { + tried = true; + goto retry; + } return total_scanned; } From 4605dfff80a3bbd3b7cab76fe06d4a8dc54c988a Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Thu, 3 Sep 2026 14:48:52 +0800 Subject: [PATCH 0293/1012] mm: revert slight behavior change for swappiness 1-200 Baoquan's review found that we unexpectedly introduced a slight behavior change for swappiness 1-200. We could now have a case like: 1. First scan -> `scanned != 0` 2. Second scan -> `scanned = 0` 3. Type fallback Step 3 was impossible before. Let's remove this possibility. Link: https://lore.kernel.org/20260903070500.76379-1-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reported-by: Baoquan He Closes: https://lore.kernel.org/linux-mm/apfZQE1X6zGAsBb_@fedora/ Cc: Axel Rasmussen Cc: Baolin Wang Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 413efe44d1f695..42fcdcd3d2e49d 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4858,7 +4858,7 @@ static int isolate_folios(unsigned long nr_to_scan, struct lruvec *lruvec, * We are running out of the current reclaim type. Fall back to * the other type if allowed. */ - if (!scanned && type_fallback_allowed) { + if (!scanned && !tried && type_fallback_allowed) { type = !type; tried = true; type_fallback_allowed = false; From 5bb59730278e6047c5d58bdc9edcc33cc51032b5 Mon Sep 17 00:00:00 2001 From: Avi Weiss Date: Sat, 29 Aug 2026 20:11:12 +0300 Subject: [PATCH 0294/1012] mm/memory: simplify error handling in insert_pages() Patch series "mm/memory: improve insert_pages() error handling", v3. Improve insert_pages() error handling. The first patch simplifies error handling by initializing the error status to zero and assigning error codes at their respective failure sites. The second patch returns -ENOMEM when walk_to_pmd() fails. A NULL return from walk_to_pmd() indicates failure to allocate an upper page-table level, so -ENOMEM is more appropriate than -EFAULT and is consistent with the subsequent pte_alloc() failure. This patch (of 2): Initialize error return status to zero and then set it as needed at each point of failure. Assign -ENOMEM explicitly when pte_alloc() fails as the pte_alloc() macro returns a boolean. Link: https://lore.kernel.org/cover.1788022178.git.thnkslprpt@gmail.com Link: https://lore.kernel.org/dd3a672c858b38c7525541b19a919e120c4e5a0e.1788022178.git.thnkslprpt@gmail.com Signed-off-by: Avi Weiss Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/memory.c | 22 +++++++++++----------- 1 file changed, 11 insertions(+), 11 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 9cbce5c90bffde..2561dc6bdde635 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2565,20 +2565,22 @@ static int insert_pages(struct vm_area_struct *vma, unsigned long addr, unsigned long curr_page_idx = 0; unsigned long remaining_pages_total = *num; unsigned long pages_to_write_in_pmd; - int ret; + int err = 0; more: - ret = -EFAULT; pmd = walk_to_pmd(mm, addr); - if (!pmd) + if (!pmd) { + err = -EFAULT; goto out; + } pages_to_write_in_pmd = min_t(unsigned long, remaining_pages_total, PTRS_PER_PTE - pte_index(addr)); /* Allocate the PTE if necessary; takes PMD lock once only. */ - ret = -ENOMEM; - if (pte_alloc(mm, pmd)) + if (pte_alloc(mm, pmd)) { + err = -ENOMEM; goto out; + } while (pages_to_write_in_pmd) { int pte_idx = 0; @@ -2586,15 +2588,14 @@ static int insert_pages(struct vm_area_struct *vma, unsigned long addr, start_pte = pte_offset_map_lock(mm, pmd, addr, &pte_lock); if (!start_pte) { - ret = -EFAULT; + err = -EFAULT; goto out; } for (pte = start_pte; pte_idx < batch_size; ++pte, ++pte_idx) { - int err = insert_page_in_batch_locked(vma, pte, - addr, pages[curr_page_idx], prot); + err = insert_page_in_batch_locked(vma, pte, addr, + pages[curr_page_idx], prot); if (unlikely(err)) { pte_unmap_unlock(start_pte, pte_lock); - ret = err; remaining_pages_total -= pte_idx; goto out; } @@ -2607,10 +2608,9 @@ static int insert_pages(struct vm_area_struct *vma, unsigned long addr, } if (remaining_pages_total) goto more; - ret = 0; out: *num = remaining_pages_total; - return ret; + return err; } /** From 4f3c8d2e9c99acbe15c2c5d778e351f10daa90f7 Mon Sep 17 00:00:00 2001 From: Avi Weiss Date: Sat, 29 Aug 2026 20:11:13 +0300 Subject: [PATCH 0295/1012] mm/memory: return -ENOMEM for page-table allocation failure in insert_pages() walk_to_pmd() returns NULL only when p4d_alloc(), pud_alloc(), or pmd_alloc() fails. These are page-table allocation failures, but insert_pages() currently reports them as -EFAULT. Return -ENOMEM instead, consistent with the subsequent pte_alloc() failure and with the single-page insert_page() path, which reports failure of the same page-table allocation chain as -ENOMEM. Address and range validation failures in vm_insert_pages() continue to return -EFAULT. Keep the later -EFAULT return for pte_offset_map_lock(), which is not an allocation failure. Link: https://lore.kernel.org/9d990c3ed43608e674d4b12a8c221a09fd200f49.1788022178.git.thnkslprpt@gmail.com Signed-off-by: Avi Weiss Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/memory.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/memory.c b/mm/memory.c index 2561dc6bdde635..27ffe1a99a08b0 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2569,7 +2569,7 @@ static int insert_pages(struct vm_area_struct *vma, unsigned long addr, more: pmd = walk_to_pmd(mm, addr); if (!pmd) { - err = -EFAULT; + err = -ENOMEM; goto out; } From 0d2b9cbce31d8511f9d490b2f54a4b641c170eb0 Mon Sep 17 00:00:00 2001 From: Wei Yang Date: Sat, 29 Aug 2026 02:58:47 +0000 Subject: [PATCH 0296/1012] mm: adjust out-dated document of __GFP_NOFAIL Commit ee040cbd6e48 ("mm/page_alloc: don't warn about large allocations with __GFP_NOFAIL") remove a warning on allocating large folio with __GFP_NOFAIL, which was placed there by commit 903edea6c53f ("mm: warn about illegal __GFP_NOFAIL usage in a more appropriate location and manner"). While in that commit, it also documented this behavior which is out-dated now. Adjust the document to align to current code, and adjust the comment while at it. Link: https://lore.kernel.org/20260829025847.26779-1-richard.weiyang@gmail.com Signed-off-by: Wei Yang Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Zi Yan Reviewed-by: Vlastimil Babka (SUSE) Cc: Brendan Jackman Cc: David Hildenbrand Cc: Johannes Weiner Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan --- include/linux/gfp_types.h | 2 +- mm/page_alloc.c | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/include/linux/gfp_types.h b/include/linux/gfp_types.h index 190191411009f2..bfd4c43ed77794 100644 --- a/include/linux/gfp_types.h +++ b/include/linux/gfp_types.h @@ -244,7 +244,7 @@ enum { * definitely preferable to use the flag rather than opencode endless * loop around allocator. * Allocating pages from the buddy with __GFP_NOFAIL and order > 1 is - * not supported. Please consider using kvmalloc() instead. + * discouraged. Please consider using kvmalloc() instead if possible. */ #define __GFP_IO ((__force gfp_t)___GFP_IO) #define __GFP_FS ((__force gfp_t)___GFP_FS) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 104a07031f30a1..aaeb30a7524b1b 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -4837,7 +4837,7 @@ __alloc_pages_slowpath(gfp_t gfp_mask, unsigned int order, if (unlikely(nofail)) { /* - * Also we don't support __GFP_NOFAIL without __GFP_DIRECT_RECLAIM, + * We don't support __GFP_NOFAIL without __GFP_DIRECT_RECLAIM, * otherwise, we may result in lockup. */ WARN_ON_ONCE(!can_direct_reclaim); From 03d2fb7b626b6a427309435301e9d1f96c1a7bc1 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Thu, 17 Sep 2026 20:12:02 -0400 Subject: [PATCH 0297/1012] mm/mempolicy: use SRCU for the weighted interleave state Patch series "mm/mempolicy: stop copying state in the interleave paths", v2. The interleave node selectors and bulk allocators take copies of nodemasks and node weights (for weighted interleave) in the fault path. Both of these copies can be entirely eliminated. For node weights, use SRCU to pin the weights in place. This eliminates a copy and a kmalloc from the bulk allocator path. For nodemasks, we can operate directly on pol->nodes as long as we bounds check the walk. A concurrent rebind can shrink the mask, or tear the read of it so the mask appears empty. - The interleave node selectors fall back to numa_node_id() when that happens, which is what they already did when a copy came back empty. - The bulk allocator simply returns what it managed to allocate. The node count and weight totals are read separately from the nodemask walk that consumes them - creating a time-of-check / time-of-use race. Just clamp the walk to a single pass (number of nodes), and clamp each bulk allocation chunk to the space left in the request. The cost is distribution accuracy during a rebind. The copies never corrected for that either - they only kept the code from dividing by zero and overrunning the allocation request. This patch (of 2): alloc_pages_bulk_weighted_interleave() copies iw_table into a scratch array on every call so it can walk the weights outside of RCU. The copy exists only because the loop may sleep in the page allocator and so cannot hold rcu_read_lock(). Use SRCU to pin the global iw_table object and use it in-place instead. Retire through both flavors - call_srcu() for the sleeping readers, then kfree_rcu() for the reference-less ones - so writers no longer block on synchronize_rcu() either. Tested in a VM with KASAN, PROVE_LOCKING and DEBUG_OBJECTS_RCU_HEAD, with a udelay() injected into the read section to widen the race against concurrent sysfs weight writers, and placement checked against the configured weights. Every retired state reached its callback. Swapping the deferred free for a bare kfree() in the same test reports a use-after-free immediately. Link: https://lore.kernel.org/20260918001203.3389165-1-gourry@gourry.net Link: https://lore.kernel.org/20260918001203.3389165-2-gourry@gourry.net Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Suggested-by: Andrew Morton Suggested-by: Matthew Wilcox Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Alistair Popple Cc: Byungchul Park Cc: "Huang, Ying" Cc: Joshua Hahn Cc: Matthew Brost Cc: Rakie Kim Cc: Zi Yan --- mm/mempolicy.c | 70 ++++++++++++++++++++++++-------------------------- 1 file changed, 34 insertions(+), 36 deletions(-) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 060a0eb2691709..2643915dc96699 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -112,6 +112,7 @@ #include #include #include +#include #include #include @@ -157,6 +158,7 @@ static const int weightiness = 32; */ struct weighted_interleave_state { bool mode_auto; + struct rcu_head rcu; u8 iw_table[]; }; static struct weighted_interleave_state __rcu *wi_state; @@ -168,6 +170,24 @@ static unsigned int *node_bw_table; */ static DEFINE_MUTEX(wi_state_lock); +/* Readers that sleep while walking iw_table hold this instead */ +DEFINE_STATIC_SRCU_FAST(wi_srcu); + +static void wi_state_free_rcu(struct rcu_head *head) +{ + struct weighted_interleave_state *state = + container_of(head, struct weighted_interleave_state, rcu); + + kfree_rcu(state, rcu); +} + +/* Retire through both flavors: sleeping readers use SRCU, the rest RCU */ +static void wi_state_retire(struct weighted_interleave_state *state) +{ + if (state) + call_srcu(&wi_srcu, &state->rcu, wi_state_free_rcu); +} + static u8 get_il_weight(int node) { struct weighted_interleave_state *state; @@ -266,10 +286,7 @@ int mempolicy_set_node_perf(unsigned int node, struct access_coordinate *coords) rcu_assign_pointer(wi_state, new_wi_state); mutex_unlock(&wi_state_lock); - if (old_wi_state) { - synchronize_rcu(); - kfree(old_wi_state); - } + wi_state_retire(old_wi_state); out: kfree(old_bw); return 0; @@ -2644,7 +2661,8 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, unsigned long nr_allocated = 0; unsigned long rounds; unsigned long node_pages, delta; - u8 *weights, weight; + struct srcu_ctr __percpu *scp; + u8 *table, weight; unsigned int weight_total = 0; unsigned long rem_pages = nr_pages; nodemask_t nodes; @@ -2688,25 +2706,14 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, me->il_weight = 0; prev_node = node; - /* create a local copy of node weights to operate on outside rcu */ - weights = kmalloc(nr_node_ids, gfp & GFP_RECLAIM_MASK); - if (!weights) - return total_allocated; - - rcu_read_lock(); - state = rcu_dereference(wi_state); - if (state) { - memcpy(weights, state->iw_table, nr_node_ids * sizeof(u8)); - rcu_read_unlock(); - } else { - rcu_read_unlock(); - for (i = 0; i < nr_node_ids; i++) - weights[i] = 1; - } + /* The page allocator may sleep, pin the weight table with SRCU */ + scp = srcu_read_lock_fast(&wi_srcu); + state = srcu_dereference(wi_state, &wi_srcu); + table = state ? state->iw_table : NULL; /* calculate total, detect system default usage */ for_each_node_mask(node, nodes) - weight_total += weights[node]; + weight_total += table ? table[node] : 1; /* * Calculate rounds/partial rounds to minimize __alloc_pages_bulk calls. @@ -2718,10 +2725,10 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, rounds = rem_pages / weight_total; delta = rem_pages % weight_total; resume_node = next_node_in(prev_node, nodes); - resume_weight = weights[resume_node]; + resume_weight = table ? table[resume_node] : 1; for (i = 0; i < nnodes; i++) { node = next_node_in(prev_node, nodes); - weight = weights[node]; + weight = table ? table[node] : 1; node_pages = weight * rounds; /* If a delta exists, add this node's portion of the delta */ if (delta > weight) { @@ -2747,7 +2754,7 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, } me->il_prev = resume_node; me->il_weight = resume_weight; - kfree(weights); + srcu_read_unlock_fast(&wi_srcu, scp); return total_allocated; } @@ -3673,10 +3680,7 @@ static ssize_t node_store(struct kobject *kobj, struct kobj_attribute *attr, rcu_assign_pointer(wi_state, new_wi_state); mutex_unlock(&wi_state_lock); - if (old_wi_state) { - synchronize_rcu(); - kfree(old_wi_state); - } + wi_state_retire(old_wi_state); return count; } @@ -3742,10 +3746,7 @@ static ssize_t weighted_interleave_auto_store(struct kobject *kobj, update_wi_state: rcu_assign_pointer(wi_state, new_wi_state); mutex_unlock(&wi_state_lock); - if (old_wi_state) { - synchronize_rcu(); - kfree(old_wi_state); - } + wi_state_retire(old_wi_state); return count; } @@ -3789,10 +3790,7 @@ static void wi_state_free(void) rcu_assign_pointer(wi_state, NULL); mutex_unlock(&wi_state_lock); - if (old_wi_state) { - synchronize_rcu(); - kfree(old_wi_state); - } + wi_state_retire(old_wi_state); } static struct kobj_attribute wi_auto_attr = { From 1c068efa7dde485a7cec0512ab1340886bee38d4 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Thu, 17 Sep 2026 20:12:03 -0400 Subject: [PATCH 0298/1012] mm/mempolicy: stop copying the nodemask in the interleave paths The interleave node selectors copy pol->nodes onto the stack so the mask cannot change while they walk it. nodemask_t is 128 bytes at MAX_NUMNODES=1024, and two of the three run per folio fault. The copy only buys consistency between the node count and the walk. Drop the consistency and just bounds check the walk instead. The nodelist access is racy by design, but safe as long as we handle the scenario where a cpuset rebind causes a torn nodemask read to perceive the nodemask as empty (weight_total == 0). If an empty nodelist or weight is perceived, fall back to numa_node_id(), which is what the functions already did when the copy came back empty Otherwise, iterating the nodelist during the actual allocation loop is perfectly safe - a concurrent rebind may simply cause a skew in in the distribution of memory (or fail and fall back the same as any other error condition). weighted_interleave_nid() counts the nodes as we sum the weights. We use that node count to limit the maximum skew a single node can host. interleave_nid() walks with next_node_in() rather than next_node(), so a mask that shrank mid-walk wraps to a node still in the policy. alloc_pages_bulk_weighted_interleave() derives per-node counts from a weight total summed over the mask, so a changing mask can make them exceed the request. Clamp each chunk to the space left in page_array. A cpuset cookie will not work here: two of these take VMA policies, which mpol_rebind_mm() rebinds under mmap_write_lock(), not mems_allowed_seq. Cost is distribution accuracy during a rebind - but the copy never corrected this anyway, it was just a safety mechanism to prevent div/0 and overrunning the alloc request buffer. Remove read_once_policy_nodemask(), now unused. -fstack-usage at MAX_NUMNODES=1024: weighted_interleave_nid 184 -> 56 interleave_nid 168 -> 32 alloc_pages_bulk_mempolicy_noprof 360 -> 136 Link: https://lore.kernel.org/20260918001203.3389165-3-gourry@gourry.net Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Rakie Kim Assisted-by: LLM Cc: Alistair Popple Cc: Byungchul Park Cc: "Huang, Ying" Cc: Joshua Hahn Cc: Matthew Brost Cc: Matthew Wilcox Cc: Zi Yan --- mm/mempolicy.c | 93 ++++++++++++++++++++++++++++++-------------------- 1 file changed, 56 insertions(+), 37 deletions(-) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 2643915dc96699..fd97fb0289bc98 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -2197,34 +2197,15 @@ unsigned int mempolicy_slab_node(void) } } -static unsigned int read_once_policy_nodemask(struct mempolicy *pol, - nodemask_t *mask) -{ - /* - * barrier stabilizes the nodemask locally so that it can be iterated - * over safely without concern for changes. Allocators validate node - * selection does not violate mems_allowed, so this is safe. - */ - barrier(); - memcpy(mask, &pol->nodes, sizeof(nodemask_t)); - barrier(); - return nodes_weight(*mask); -} - static unsigned int weighted_interleave_nid(struct mempolicy *pol, pgoff_t ilx) { struct weighted_interleave_state *state; - nodemask_t nodemask; - unsigned int target, nr_nodes; + unsigned int target, nnodes = 0; u8 *table = NULL; unsigned int weight_total = 0; u8 weight; int nid = 0; - nr_nodes = read_once_policy_nodemask(pol, &nodemask); - if (!nr_nodes) - return numa_node_id(); - rcu_read_lock(); state = rcu_dereference(wi_state); @@ -2232,22 +2213,45 @@ static unsigned int weighted_interleave_nid(struct mempolicy *pol, pgoff_t ilx) if (state) table = state->iw_table; - /* calculate the total weight */ - for_each_node_mask(nid, nodemask) + /* calculate the total weight and the node count */ + for_each_node_mask(nid, pol->nodes) { weight_total += table ? table[nid] : 1; + nnodes++; + } + + /* the mask is empty */ + if (!weight_total) { + rcu_read_unlock(); + return numa_node_id(); + } /* Calculate the node offset based on totals */ target = ilx % weight_total; - nid = first_node(nodemask); - while (target) { + nid = first_node(pol->nodes); + + /* + * The target was calculated in a separate loop, and a concurrent + * rebind can change the contents of pol->nodes as we calculate. + * Access is safe, in the worst case we suddenly perceive an empty + * nodemask and return numa_node_id() below - otherwise we may + * simply cause a skew in allocations. + * + * Clamp this loop to a single pass (nnodes) to keep the walk + * bounded by node count. + */ + while (target && nnodes-- && nid < MAX_NUMNODES) { /* detect system default usage */ weight = table ? table[nid] : 1; if (target < weight) break; target -= weight; - nid = next_node_in(nid, nodemask); + nid = next_node_in(nid, pol->nodes); } rcu_read_unlock(); + + /* the mask emptied under the walk */ + if (nid >= MAX_NUMNODES) + return numa_node_id(); return nid; } @@ -2258,18 +2262,23 @@ static unsigned int weighted_interleave_nid(struct mempolicy *pol, pgoff_t ilx) */ static unsigned int interleave_nid(struct mempolicy *pol, pgoff_t ilx) { - nodemask_t nodemask; unsigned int target, nnodes; int i; int nid; - nnodes = read_once_policy_nodemask(pol, &nodemask); + nnodes = nodes_weight(pol->nodes); if (!nnodes) return numa_node_id(); target = ilx % nnodes; - nid = first_node(nodemask); - for (i = 0; i < target; i++) - nid = next_node(nid, nodemask); + nid = first_node(pol->nodes); + + /* A concurrent cpuset rebind may cause us to see an empty nodemask */ + for (i = 0; i < target && nid < MAX_NUMNODES; i++) + nid = next_node_in(nid, pol->nodes); + + /* the mask emptied under the walk */ + if (nid >= MAX_NUMNODES) + return numa_node_id(); return nid; } @@ -2665,7 +2674,6 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, u8 *table, weight; unsigned int weight_total = 0; unsigned long rem_pages = nr_pages; - nodemask_t nodes; int nnodes, node; int resume_node = MAX_NUMNODES - 1; u8 resume_weight = 0; @@ -2675,10 +2683,10 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, if (!nr_pages) return 0; - /* read the nodes onto the stack, retry if done during rebind */ + /* count the nodes, retry if a rebind happened during the read */ do { cpuset_mems_cookie = read_mems_allowed_begin(); - nnodes = read_once_policy_nodemask(pol, &nodes); + nnodes = nodes_weight(pol->nodes); } while (read_mems_allowed_retry(cpuset_mems_cookie)); /* if the nodemask has become invalid, we cannot do anything */ @@ -2688,7 +2696,7 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, /* Continue allocating from most recent node and adjust the nr_pages */ node = me->il_prev; weight = me->il_weight; - if (weight && node_isset(node, nodes)) { + if (weight && node_isset(node, pol->nodes)) { node_pages = min(rem_pages, weight); nr_allocated = __alloc_pages_bulk(gfp, node, NULL, node_pages, page_array); @@ -2712,9 +2720,13 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, table = state ? state->iw_table : NULL; /* calculate total, detect system default usage */ - for_each_node_mask(node, nodes) + for_each_node_mask(node, pol->nodes) weight_total += table ? table[node] : 1; + /* the mask emptied since it was counted */ + if (!weight_total) + goto out; + /* * Calculate rounds/partial rounds to minimize __alloc_pages_bulk calls. * Track which node weighted interleave should resume from. @@ -2724,10 +2736,14 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, */ rounds = rem_pages / weight_total; delta = rem_pages % weight_total; - resume_node = next_node_in(prev_node, nodes); + resume_node = next_node_in(prev_node, pol->nodes); + if (resume_node >= MAX_NUMNODES) + goto out; resume_weight = table ? table[resume_node] : 1; for (i = 0; i < nnodes; i++) { - node = next_node_in(prev_node, nodes); + node = next_node_in(prev_node, pol->nodes); + if (node >= MAX_NUMNODES) + break; weight = table ? table[node] : 1; node_pages = weight * rounds; /* If a delta exists, add this node's portion of the delta */ @@ -2744,6 +2760,8 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, /* node_pages can be 0 if an allocation fails and rounds == 0 */ if (!node_pages) break; + /* a rebind can invalidate the counts: never overrun page_array */ + node_pages = min(node_pages, nr_pages - total_allocated); nr_allocated = __alloc_pages_bulk(gfp, node, NULL, node_pages, page_array); page_array += nr_allocated; @@ -2754,6 +2772,7 @@ static unsigned long alloc_pages_bulk_weighted_interleave(gfp_t gfp, } me->il_prev = resume_node; me->il_weight = resume_weight; +out: srcu_read_unlock_fast(&wi_srcu, scp); return total_allocated; } From 9505b3c6d7037fdc57f770ff8ac1eb2b6d741575 Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 19 Aug 2026 01:48:30 +0900 Subject: [PATCH 0299/1012] percpu: fix the comment about which sizes share a slot A percpu allocation is at least PCPU_MIN_ALLOC_SIZE bytes, and __pcpu_size_to_slot() returns 1 for sizes below 16 bytes and 2 for sizes from 16 to 31 bytes. So fix the wrong comment. Link: https://lore.kernel.org/20260818164831.3138490-1-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Cc: Dennis Zhou Cc: Tejun Heo Cc: Christoph Lameter --- mm/percpu.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/percpu.c b/mm/percpu.c index a802d72c116fbb..47a903fe3b5124 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -100,7 +100,7 @@ /* * The slots are sorted by the size of the biggest continuous free area. - * 1-31 bytes share the same slot. + * [PCPU_MIN_ALLOC_SIZE..15] bytes share the same slot. */ #define PCPU_SLOT_BASE_SHIFT 5 /* chunks in slots below this are subject to being sidelined on failed alloc */ From 26d01dedbadd8fe70548d7ce7f7bf063b3df2ca6 Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Tue, 18 Aug 2026 03:06:23 -0700 Subject: [PATCH 0300/1012] mm, swap: distinguish a malformed swap entry from a dying device Patch series "mm, swap: don't spin on a bad swap entry", v3. I've seen some machines at Meta fleet that show the following type of problem: 1) It gets some weird warning: BUG: Bad page map in process khugepaged pte:f000eef300000017 pmd:00000067 addr:00007f57c0a01000 vm_flags:20200073 anon_vma:ffff88829af7c340 mapping:0000000000000000 index:7f57c0a01 The corruption is most likely the collapse/PT_RECLAIM race fixed by commit 366a4532d96f ("mm: fix the race between collapse and PT_RECLAIM under per-vma lock"). But this series is not about this one. 2) Then the fault never makes progress. do_swap_page() returns 0 when get_swap_device() fails, so the fault is retried, reads the same entry and faults again. Nothing in the round trip changes the PTE, and the same line comes out on every pass: get_swap_device: Bad swap offset entry 3ffffffc043c5 Patch 1 makes get_swap_device() return ERR_PTR(-EIO) for a malformed entry, keeping NULL for a device swapoff is taking away, and converts the callers. No functional change expected. Patch 2 uses that to return VM_FAULT_SIGBUS instead of retrying. This patch (of 2): get_swap_device() returns NULL for two different things: an entry whose type names no swap device or whose offset is past the end of one, and a device that swapoff is taking away. The first never becomes valid, the second does, and callers cannot tell them apart. Return ERR_PTR(-EIO) for the two malformed cases and keep NULL for swapoff. copy_nonpresent_pte() already reports -EIO for an entry whose type names no device. Callers bail out on failure either way, so switch them to IS_ERR_OR_NULL(), and let the two paths that drop the reference skip an error pointer. No functional change. Link: https://lore.kernel.org/20260818-swap-v3-0-d3fa52598a59@debian.org Link: https://lore.kernel.org/20260818-swap-v3-1-d3fa52598a59@debian.org Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: Barry Song Acked-by: Kairui Song Reviewed-by: Nhat Pham Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Chengming Zhou Cc: Chris Li Cc: Hugh Dickins Cc: Jann Horn Cc: Johannes Weiner Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Pedro Falcato Cc: Peter Xu Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/memory.c | 6 +++--- mm/mincore.c | 2 +- mm/shmem.c | 2 +- mm/swap_state.c | 4 ++-- mm/swapfile.c | 15 ++++++++++----- mm/userfaultfd.c | 4 ++-- mm/zswap.c | 2 +- 7 files changed, 20 insertions(+), 15 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 27ffe1a99a08b0..e59d6c2a343205 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4956,9 +4956,9 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) goto out; } - /* Prevent swapoff from happening to us. */ + /* Prevent swapoff from happening to us, and reject a bad entry. */ si = get_swap_device(entry); - if (unlikely(!si)) + if (IS_ERR_OR_NULL(si)) goto out; folio = swap_cache_get_folio(entry); @@ -5268,7 +5268,7 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) if (vmf->pte) pte_unmap_unlock(vmf->pte, vmf->ptl); out: - if (si) + if (!IS_ERR_OR_NULL(si)) put_swap_device(si); return ret; out_nomap: diff --git a/mm/mincore.c b/mm/mincore.c index ff4ac828176837..c086836bc4bcc5 100644 --- a/mm/mincore.c +++ b/mm/mincore.c @@ -71,7 +71,7 @@ static unsigned char mincore_swap(swp_entry_t entry, bool shmem) */ if (shmem) { si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) return 0; } folio = swap_cache_get_folio(entry); diff --git a/mm/shmem.c b/mm/shmem.c index ae39966fc7304e..d3f24b5977bd5f 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -2481,7 +2481,7 @@ static int shmem_swapin_folio(struct inode *inode, pgoff_t index, si = get_swap_device(index_entry); order = shmem_confirm_swap(mapping, index, index_entry); - if (unlikely(!si)) { + if (IS_ERR_OR_NULL(si)) { if (order < 0) return -EEXIST; else diff --git a/mm/swap_state.c b/mm/swap_state.c index f3961fdd857dc6..305877e1f4d7bf 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -716,7 +716,7 @@ struct folio *read_swap_cache_async(struct swap_io_ctx *ctx, swp_entry_t entry, struct folio *folio; si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) return NULL; mpol = get_vma_policy(vma, addr, 0, &ilx); @@ -952,7 +952,7 @@ static struct folio *swap_vma_readahead(swp_entry_t targ_entry, gfp_t gfp_mask, */ if (swp_type(entry) != swp_type(targ_entry)) { si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) continue; } folio = swap_cache_read_folio(&ctx, entry, gfp_mask, mpol, ilx, diff --git a/mm/swapfile.c b/mm/swapfile.c index 3b2279a16d1cf8..408f6c72fb5a69 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1504,7 +1504,7 @@ int swap_retry_table_alloc(swp_entry_t entry, gfp_t gfp) unsigned long offset = swp_offset(entry); si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) return 0; ci = __swap_offset_to_cluster(si, offset); @@ -1859,7 +1859,10 @@ void folio_put_swap(struct folio *folio, struct page *page) * Check whether swap entry is valid in the swap device. If so, * return pointer to swap_info_struct, and keep the swap entry valid * via preventing the swap device from being swapoff, until - * put_swap_device() is called. Otherwise return NULL. + * put_swap_device() is called. Return NULL for an empty entry or a + * device that is going away, and ERR_PTR(-EIO) if the entry's type + * names no swap device or its offset is past the end of one. These EIOs + * are preceded by pr_err(). * * Notice that swapoff or swapoff+swapon can still happen before the * percpu_ref_tryget_live() in get_swap_device() or after the @@ -1900,12 +1903,14 @@ struct swap_info_struct *get_swap_device(swp_entry_t entry) return si; bad_nofile: pr_err_ratelimited("%s: %s%08lx\n", __func__, Bad_file, entry.val); + return ERR_PTR(-EIO); + out: return NULL; put_out: pr_err_ratelimited("%s: %s%08lx\n", __func__, Bad_offset, entry.val); percpu_ref_put(&si->users); - return NULL; + return ERR_PTR(-EIO); } /* @@ -2001,7 +2006,7 @@ int swp_swapcount(swp_entry_t entry) int count; si = get_swap_device(entry); - if (!si) + if (IS_ERR_OR_NULL(si)) return 0; ci = swap_cluster_lock(si, swp_offset(entry)); @@ -2127,7 +2132,7 @@ void swap_put_entries_direct(swp_entry_t entry, int nr) struct swap_info_struct *si; si = get_swap_device(entry); - if (WARN_ON_ONCE(!si)) + if (WARN_ON_ONCE(IS_ERR_OR_NULL(si))) return; if (WARN_ON_ONCE(end_offset > si->max)) goto out; diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index f39f109f17989e..b909ec8ef20bce 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -1701,7 +1701,7 @@ static long move_pages_ptes(struct mm_struct *mm, pmd_t *dst_pmd, pmd_t *src_pmd } si = get_swap_device(entry); - if (unlikely(!si)) { + if (IS_ERR_OR_NULL(si)) { ret = -EAGAIN; goto out; } @@ -1758,7 +1758,7 @@ static long move_pages_ptes(struct mm_struct *mm, pmd_t *dst_pmd, pmd_t *src_pmd if (dst_pte) pte_unmap(dst_pte); mmu_notifier_invalidate_range_end(&range); - if (si) + if (!IS_ERR_OR_NULL(si)) put_swap_device(si); return ret; diff --git a/mm/zswap.c b/mm/zswap.c index fc869d60ef1b48..f3ae3c81e48eac 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -984,7 +984,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry, /* try to allocate swap cache folio */ si = get_swap_device(swpentry); - if (!si) + if (IS_ERR_OR_NULL(si)) return -EEXIST; mpol = get_task_policy(current); From fc90c1f21a8c3a8e9ba8c6e79af284c8a9dc0b1c Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Tue, 18 Aug 2026 03:06:24 -0700 Subject: [PATCH 0301/1012] mm: fail the fault on a malformed swap entry instead of retrying it do_swap_page() returns 0 when get_swap_device() fails, which the fault handler reads as "handled". For an entry that can never become valid the retry takes the same fault again, so the thread spins forever, retrying on the same fault. Return VM_FAULT_SIGBUS (Bad access) for a malformed entry (pr_err() was called at get_swap_device()). Link: https://lore.kernel.org/20260818-swap-v3-2-d3fa52598a59@debian.org Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Acked-by: Kairui Song Reviewed-by: Barry Song Reviewed-by: Nhat Pham Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Chengming Zhou Cc: Chris Li Cc: Hugh Dickins Cc: Jann Horn Cc: Johannes Weiner Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Pedro Falcato Cc: Peter Xu Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/memory.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/mm/memory.c b/mm/memory.c index e59d6c2a343205..347db2acd0f83b 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -4958,8 +4958,11 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) /* Prevent swapoff from happening to us, and reject a bad entry. */ si = get_swap_device(entry); - if (IS_ERR_OR_NULL(si)) + if (IS_ERR_OR_NULL(si)) { + if (IS_ERR(si)) + ret = VM_FAULT_SIGBUS; goto out; + } folio = swap_cache_get_folio(entry); if (folio) From ad99c9ee405beb8c4bfad40bb5411f71c077483a Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Mon, 17 Aug 2026 18:08:08 -0400 Subject: [PATCH 0302/1012] mm/huge_memory: skip zone device folios in madvise_free_huge_pmd() Patch series "mm: reject zone device folios in more folio walkers", v2. Several LRU-oriented mm walkers resolve the folio backing a PMD entry (or a physical pfn) and then reclaim, age, migrate, or lazyfree it without ever checking for ZONE_DEVICE memory. This series adds missing folio_is_zone_device() rejections, matching the checks that comparable walkers already perform. - mm/huge_memory, mm/madvise: the !pmd_present branch above these sites only filters device-private entries (which are non-present). A present zone device PMD (e.g. device-coherent) would still reach the folio and be lazyfreed / aged / paged out. Add an explicit check. - mm/mempolicy: queue_folios_pmd() can see a present zone device PMD (e.g. device-coherent) and queue it for migration. No crash reproducer - this is a correctness/hardening cleanup found by inspection. All checks are placed after the folio is resolved and before it is acted upon, on paths that already hold the relevant page-table lock, so no locking or refcount changes are involved. This patch (of 3): madvise_free_huge_pmd() resolves the folio backing a PMD via pmd_folio() and marks it lazyfree without checking for zone device memory. The surrounding guards do not cover every zone device case: - MADV_FREE only operates on anonymous VMAs (DAX mappings are excluded) - !pmd_present() branch rejects device-private and migration entries - present zone device PMD (device coherent THP) is not filtered. Unlike vm_normal_page_pmd(), it performs no special/pfnmap check, and would be marked lazyfree here. Bail out when the folio is a zone device folio. Link: https://lore.kernel.org/20260817220810.1175596-1-gourry@gourry.net Link: https://lore.kernel.org/20260817220810.1175596-2-gourry@gourry.net Fixes: a30b48bf1b24 ("mm/migrate_device: implement THP migration of zone device pages") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Tested-by: Lance Yang Cc: Alistair Popple Cc: Balbir Singh Cc: Baolin Wang Cc: Barry Song Cc: Byungchul Park Cc: Dev Jain Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Liam R. Howlett Cc: Matthew Brost Cc: Rakie Kim Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Zi Yan Cc: --- mm/huge_memory.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 1e5d68acf62a52..54494c3fa9835e 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -2423,6 +2423,10 @@ bool madvise_free_huge_pmd(struct mmu_gather *tlb, struct vm_area_struct *vma, } folio = pmd_folio(orig_pmd); + + if (folio_is_zone_device(folio)) + goto out; + /* * If other processes are mapping this folio, we couldn't discard * the folio unless they all do MADV_FREE so let's skip the folio. From 46b1108bc85f24ae82ace154db9d7fb75dfe9bac Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Mon, 17 Aug 2026 18:08:09 -0400 Subject: [PATCH 0303/1012] mm/madvise: skip zone device folios in cold/pageout PMD range madvise_cold_or_pageout_pte_range() resolves the folio backing a PMD via pmd_folio() and ages or reclaims it without checking for zone device memory. The surrounding guards do not cover every zone device case: - can_madv_lru_vma() excludes VM_PFNMAP and VM_HUGETLB VMAs (so device DAX is filtered) - !pmd_present() branch above rejects device-private and migration entries, which are non-present. - A present zone device PMD - e.g. a device-coherent THP - is not filtered by any of these, nor by pmd_folio() (unlike vm_normal_page_pmd(), it performs no special/pfnmap check), and would be aged or paged out here. Skip ZONE_DEVICE folios explicitly during MADV_COLD/PAGEOUT. Link: https://lore.kernel.org/20260817220810.1175596-3-gourry@gourry.net Fixes: a30b48bf1b24 ("mm/migrate_device: implement THP migration of zone device pages") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Balbir Singh Tested-by: Lance Yang Cc: Alistair Popple Cc: Baolin Wang Cc: Barry Song Cc: Byungchul Park Cc: Dev Jain Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Liam R. Howlett Cc: Matthew Brost Cc: Rakie Kim Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Zi Yan Cc: --- mm/madvise.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/mm/madvise.c b/mm/madvise.c index eeee82cf2b3f4b..73c2901b9adbf0 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -404,6 +404,9 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd, folio = pmd_folio(orig_pmd); + if (folio_is_zone_device(folio)) + goto huge_unlock; + /* Do not interfere with other mappings of this folio */ if (folio_maybe_mapped_shared(folio)) goto huge_unlock; From 37b072d1d5cfdc9811585d2b384307ddf65ce1c5 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Mon, 17 Aug 2026 18:08:10 -0400 Subject: [PATCH 0304/1012] mm/mempolicy: skip zone device folios when queueing folios queue_folios_pte_range() already pairs vm_normal_folio() with an explicit folio_is_zone_device() check before adding folios to the migration pagelist. vm_normal_folio() alone does not reject zone device memory (a present device-coherent page in a normal VMA is returned as "normal"). Mirror the explicit check in queue_folios_pmd() as well. queue_folios_pmd() uses pmd_folio() directly and can encounter a present zone device PMD - e.g. a device-coherent THP. This is not filtered by existing checks: !pmd_present() - only rejects non-present device-private and migration entries vma_migratable() - excludes DAX and VM_PFNMAP. The early return also means such a folio is no longer counted in qp->nr_failed under MPOL_MF_STRICT. This is the same pattern used by queue_folios_pte_range() (skipping zone device without failing). Link: https://lore.kernel.org/20260817220810.1175596-4-gourry@gourry.net Fixes: a30b48bf1b24 ("mm/migrate_device: implement THP migration of zone device pages") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Reviewed-by: Balbir Singh Acked-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Tested-by: Lance Yang Cc: Alistair Popple Cc: Baolin Wang Cc: Barry Song Cc: Byungchul Park Cc: Dev Jain Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Liam R. Howlett Cc: Matthew Brost Cc: Rakie Kim Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Zi Yan Cc: --- mm/mempolicy.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index fd97fb0289bc98..95dba5d919e925 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -679,6 +679,8 @@ static void queue_folios_pmd(pmd_t *pmd, struct mm_walk *walk) return; } folio = pmd_folio(pmdval); + if (folio_is_zone_device(folio)) + return; if (is_huge_zero_folio(folio)) { walk->action = ACTION_CONTINUE; return; From 7f9bb7eef665654ef69b84f93905913e8e853d99 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Mon, 17 Aug 2026 14:19:55 +0800 Subject: [PATCH 0305/1012] selftests/mm: khugepaged: consolidate error exits via kselftest helpers Replace the perror()+exit(EXIT_FAILURE) pattern with ksft_exit_fail_perror() so failures are reported through the kselftest framework, consistent with the rest of the file. Link: https://lore.kernel.org/20260817061955.45454-1-hongfu.li@linux.dev Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Zi Yan Reviewed-by: Lance Yang Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/khugepaged.c | 20 +++++++------------- 1 file changed, 7 insertions(+), 13 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 83d27d069c4139..f82673f5f6b47e 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -337,21 +337,15 @@ static void *file_setup_area_common(int nr_hpages, enum file_setup_ops setup) ksft_exit_fail_perror("open()"); size = nr_hpages * hpage_pmd_size; - if (ftruncate(fd, size)) { - perror("ftruncate()"); - exit(EXIT_FAILURE); - } + if (ftruncate(fd, size)) + ksft_exit_fail_perror("ftruncate()"); p = mmap(BASE_ADDR, size, PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0); - if (p != BASE_ADDR) { - perror("mmap()"); - exit(EXIT_FAILURE); - } + if (p != BASE_ADDR) + ksft_exit_fail_perror("mmap()"); fill_memory(p, 0, size); - if (msync(p, size, MS_SYNC)) { - perror("msync()"); - exit(EXIT_FAILURE); - } + if (msync(p, size, MS_SYNC)) + ksft_exit_fail_perror("msync()"); close(fd); munmap(p, size); success("OK"); @@ -426,7 +420,7 @@ static bool file_check_huge(void *addr, size_t len, int nr_hpages, case VMA_SHMEM: return check_huge_shmem(addr, len, nr_hpages, hpage_size); default: - exit(EXIT_FAILURE); + ksft_exit_fail_msg("Unknown VMA type\n"); return false; } } From b9e42e81b4fb3825570d2bf85e575af98d263d64 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Sun, 30 Aug 2026 08:20:43 +0800 Subject: [PATCH 0306/1012] memcg: acquire peaks_lock when reading memory.peak MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "mm, memcg: fix memory.peak reset clobbering other fds' watermark", v4. The memory.peak / memory.swap.peak per-fd watermark tracking has two issues. Each open fd is a watcher and reads back max(its own value, the shared local_watermark); both bugs live in that scheme. Worst case for both is the same and is userspace-visible: a reader of memory.peak (or memory.swap.peak) gets a value lower than the true peak, so a tool that sizes or bills a cgroup by its peak usage under-reports it. Patch 1 (read side) fixes the race Sashiko pointed out in the v1 review [1]: peak_show() inspects local_watermark and the per-fd values without holding peaks_lock, so a reader that races an unrelated peak_write() reset briefly observes the lowered value. Transient. It takes peaks_lock in the show path. Patch 2 (write side) fixes peak_write(): on a reset it stores the current usage into the other watchers instead of the old watermark, so once usage has dropped from a peak a reset on one fd drags every other fd's peak down too, even fds that never reset. This patch (of 2): Sashiko reported that a reader can transiently observe a lower peak within a race window [1]. peak_show() returns max(local_watermark, ofp->value), but peak_write() updates those two under peaks_lock while the reader takes no lock. The interleaving is: writer (reset on fd A) reader (fd B) ---------------------- ------------- usage = page_counter_read(pc) WRITE_ONCE(local_watermark, usage) // watermark lowered to usage lw = READ_ONCE(local_watermark) // sees the lowered usage val = READ_ONCE(ofp->value) // B's value not updated yet return max(lw, val) // both low -> low peak WRITE_ONCE(peer_ctx->value, usage) // B updated, but too late Fix it by acquiring peaks_lock when reading the peak, so the reader sees a consistent snapshot of local_watermark and the per-fd values. The same race applies to memory.swap.peak, which shares peaks_lock and the peak_write() path, so take the lock there as well. Link: https://lore.kernel.org/20260830002044.1938621-1-ridong.chen@linux.dev Link: https://lore.kernel.org/20260830002044.1938621-2-ridong.chen@linux.dev Link: https://sashiko.dev/#/patchset/20260730115314.1069089-1-ridong.chen@linux.dev?part=1 [1] Fixes: c6f53ed8f213 ("mm, memcg: cg2 memory{.swap,}.peak write handlers") Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Acked-by: Shakeel Butt Reviewed-by: Muchun Song Assisted-by: Claude:claude-opus-4-8 Cc: David Finkel Cc: Michal Hocko Cc: Michal Koutný Cc: Roman Gushchin Cc: Tejun Heo Cc: Tao Cui --- mm/memcontrol.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 30636b9d96739e..1ee974cb6d3727 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4745,6 +4745,7 @@ static int memory_peak_show(struct seq_file *sf, void *v) { struct mem_cgroup *memcg = mem_cgroup_from_css(seq_css(sf)); + guard(spinlock)(&memcg->peaks_lock); return peak_show(sf, v, &memcg->memory); } @@ -5888,6 +5889,7 @@ static int swap_peak_show(struct seq_file *sf, void *v) { struct mem_cgroup *memcg = mem_cgroup_from_css(seq_css(sf)); + guard(spinlock)(&memcg->peaks_lock); return peak_show(sf, v, &memcg->swap); } From 00aed3d8c48ed0178652df8fdcb6bd86456b9c12 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Sun, 30 Aug 2026 08:20:44 +0800 Subject: [PATCH 0307/1012] mm, memcg: fix memory.peak reset clobbering other fds' watermark MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Writing to memory.peak resets the peak for that fd only. Each fd is a watcher and reads back max(its own value, the shared local_watermark). peak_write() resets by lowering local_watermark to the current usage. To keep the other watchers' peaks it then walks the watcher list, but it stores the current usage into them instead of the old watermark. So once usage has dropped from a peak, a reset on one fd wrongly drags every other fd's peak down too, even fds that never reset. Reproduced on 7.2.0-rc5-next under QEMU, two fds A and B on one cgroup: B sees the peak (410624 KB), usage drops, then A resets -- and B's peak collapses to 1060 KB although B never reset. With this patch B keeps reading 410624 KB. Fix: save the old watermark before lowering it and use that to floor the other watchers, so a reset only affects the fd that issued it. Link: https://lore.kernel.org/20260830002044.1938621-3-ridong.chen@linux.dev Fixes: c6f53ed8f213 ("mm, memcg: cg2 memory{.swap,}.peak write handlers") Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Closes: https://sashiko.dev/#/patchset/20260807090000.1532495-1-ridong.chen@linux.dev Acked-by: Tao Cui Acked-by: Johannes Weiner Acked-by: Shakeel Butt Assisted-by: Claude:claude-opus-4-8 Cc: David Finkel Cc: Michal Hocko Cc: Michal Koutný Cc: Muchun Song Cc: Roman Gushchin Cc: Tejun Heo --- mm/memcontrol.c | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 1ee974cb6d3727..d5ebe83eae3efc 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4775,7 +4775,7 @@ static ssize_t peak_write(struct kernfs_open_file *of, char *buf, size_t nbytes, loff_t off, struct page_counter *pc, struct list_head *watchers) { - unsigned long usage; + unsigned long usage, old_watermark; struct cgroup_of_peak *peer_ctx; struct mem_cgroup *memcg = mem_cgroup_from_css(of_css(of)); struct cgroup_of_peak *ofp = of_peak(of); @@ -4783,11 +4783,12 @@ static ssize_t peak_write(struct kernfs_open_file *of, char *buf, size_t nbytes, spin_lock(&memcg->peaks_lock); usage = page_counter_read(pc); + old_watermark = READ_ONCE(pc->local_watermark); WRITE_ONCE(pc->local_watermark, usage); list_for_each_entry(peer_ctx, watchers, list) - if (usage > peer_ctx->value) - WRITE_ONCE(peer_ctx->value, usage); + if (peer_ctx != ofp && old_watermark > peer_ctx->value) + WRITE_ONCE(peer_ctx->value, old_watermark); /* initial write, register watcher */ if (ofp->value == OFP_PEAK_UNSET) From e9b86094d90025e8228419439eb098a791da51a5 Mon Sep 17 00:00:00 2001 From: Eamon Sippy Date: Sat, 15 Aug 2026 10:32:45 +0000 Subject: [PATCH 0308/1012] mm: cma: make mm/cma.h self-contained and conditionalize includes mm/cma.h uses types from , , and without explicitly including them, violating the kernel header self-containment guidelines. and are also included unconditionally even though they are only needed under CONFIG_CMA_DEBUGFS and CONFIG_CMA_SYSFS respectively. Move the struct cma_kobject definition and inside the CONFIG_CMA_SYSFS block, and move inside CONFIG_CMA_DEBUGFS. Remove spurious trailing semicolons after the empty inline function bodies in the CONFIG_CMA_SYSFS #else branch. Add so that MAX_CMA_AREAS and CMA_MAX_NAME are always available when this header is included. Link: https://lore.kernel.org/20260815103246.5315-1-eamon112009@gmail.com Signed-off-by: Eamon Sippy Signed-off-by: Andrew Morton Reviewed-by: Barry Song --- mm/cma.h | 23 +++++++++++++++++------ 1 file changed, 17 insertions(+), 6 deletions(-) diff --git a/mm/cma.h b/mm/cma.h index ab6d39898ea52e..68e574b0f95187 100644 --- a/mm/cma.h +++ b/mm/cma.h @@ -2,14 +2,24 @@ #ifndef __MM_CMA_H__ #define __MM_CMA_H__ +#include #include +#include +#include +#include + +#ifdef CONFIG_CMA_DEBUGFS #include +#endif + +#ifdef CONFIG_CMA_SYSFS #include struct cma_kobject { struct kobject kobj; struct cma *cma; }; +#endif /* * Multi-range support. This can be useful if the size of the allocation @@ -38,10 +48,10 @@ struct cma_memrange { #define CMA_MAX_RANGES 8 struct cma { - unsigned long count; - unsigned long available_count; + unsigned long count; + unsigned long available_count; unsigned int order_per_bit; /* Order of pages represented by one bit */ - spinlock_t lock; + spinlock_t lock; struct mutex alloc_mutex; #ifdef CONFIG_CMA_DEBUGFS struct hlist_head mem_head; @@ -87,10 +97,11 @@ void cma_sysfs_account_fail_pages(struct cma *cma, unsigned long nr_pages); void cma_sysfs_account_release_pages(struct cma *cma, unsigned long nr_pages); #else static inline void cma_sysfs_account_success_pages(struct cma *cma, - unsigned long nr_pages) {}; + unsigned long nr_pages) {} static inline void cma_sysfs_account_fail_pages(struct cma *cma, - unsigned long nr_pages) {}; + unsigned long nr_pages) {} static inline void cma_sysfs_account_release_pages(struct cma *cma, - unsigned long nr_pages) {}; + unsigned long nr_pages) {} #endif + #endif From 3f6ceb108d29cba635953577c7ea491f0f9ded03 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Thu, 13 Aug 2026 17:38:09 +0800 Subject: [PATCH 0309/1012] mm/oom_kill: remove unreachable __GFP_THISNODE check in constrained_alloc() The __GFP_THISNODE check in constrained_alloc() is dead code: global OOM is never triggered with __GFP_THISNODE (blocked in __alloc_pages_may_oom before out_of_memory() is called), and memcg OOM returns CONSTRAINT_MEMCG at the top of the function before reaching this point. Remove the check, its stale comment, and update the following comment that referenced __GFP_THISNODE. Link: https://lore.kernel.org/20260813093810.573302-1-ye.liu@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Acked-by: Michal Hocko Acked-by: Shakeel Butt Cc: David Rientjes --- mm/oom_kill.c | 13 +++---------- 1 file changed, 3 insertions(+), 10 deletions(-) diff --git a/mm/oom_kill.c b/mm/oom_kill.c index 5f372f6e26fa32..fd3c476846a302 100644 --- a/mm/oom_kill.c +++ b/mm/oom_kill.c @@ -267,18 +267,11 @@ static enum oom_constraint constrained_alloc(struct oom_control *oc) if (!oc->zonelist) return CONSTRAINT_NONE; - /* - * Reach here only when __GFP_NOFAIL is used. So, we should avoid - * to kill current.We have to random task kill in this case. - * Hopefully, CONSTRAINT_THISNODE...but no way to handle it, now. - */ - if (oc->gfp_mask & __GFP_THISNODE) - return CONSTRAINT_NONE; /* - * This is not a __GFP_THISNODE allocation, so a truncated nodemask in - * the page allocator means a mempolicy is in effect. Cpuset policy - * is enforced in get_page_from_freelist(). + * A truncated nodemask in the page allocator means a mempolicy + * is in effect. Cpuset policy is enforced in + * get_page_from_freelist(). */ if (oc->nodemask && !nodes_subset(node_states[N_MEMORY], *oc->nodemask)) { From 02f7e338c4975d0b484180d3da7b1aa965ea91df Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Tue, 11 Aug 2026 11:36:08 +0800 Subject: [PATCH 0310/1012] mm/oom_kill, proc: replace magic number 1000 with OOM_SCORE_ADJ_MAX In oom_badness() and proc_oom_score(), the oom_score_adj normalization uses a hardcoded 1000, which is the value of OOM_SCORE_ADJ_MAX defined in include/uapi/linux/oom.h. Other code in the kernel (e.g. fs/proc/base.c oom_adj handling) already uses OOM_SCORE_ADJ_MAX for the same purpose. Replace the magic number with the macro for consistency and readability. No functional change. Link: https://lore.kernel.org/20260811033609.3992348-1-ye.liu@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Acked-by: Michal Hocko Cc: David Rientjes Cc: Shakeel Butt Cc: Song Hu --- fs/proc/base.c | 3 ++- mm/oom_kill.c | 2 +- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/fs/proc/base.c b/fs/proc/base.c index 6a39de424f62a1..58be3894246054 100644 --- a/fs/proc/base.c +++ b/fs/proc/base.c @@ -594,7 +594,8 @@ static int proc_oom_score(struct seq_file *m, struct pid_namespace *ns, * exporting for a long time so userspace might depend on it. */ if (badness != LONG_MIN) - points = (1000 + badness * 1000 / (long)totalpages) * 2 / 3; + points = (OOM_SCORE_ADJ_MAX + + badness * OOM_SCORE_ADJ_MAX / (long)totalpages) * 2 / 3; seq_printf(m, "%lu\n", points); diff --git a/mm/oom_kill.c b/mm/oom_kill.c index fd3c476846a302..5d48bd862c27b2 100644 --- a/mm/oom_kill.c +++ b/mm/oom_kill.c @@ -230,7 +230,7 @@ long oom_badness(struct task_struct *p, unsigned long totalpages) task_unlock(p); /* Normalize to oom_score_adj units */ - adj *= totalpages / 1000; + adj *= totalpages / OOM_SCORE_ADJ_MAX; points += adj; return points; From 676a4b82da1b93714a6d13d15e212d9c57fa171b Mon Sep 17 00:00:00 2001 From: Pedro Falcato Date: Tue, 11 Aug 2026 18:21:55 +0100 Subject: [PATCH 0311/1012] mm: replace custom bad page map ratelimiting logic Patch series "mm: replace custom ratelimiting logic". The kernel has a perfectly cromulent and mostly-equivalent variant in lib/ratelimit.c that can be used. This patch (of 2): The current logic (allow up to $BURST prints per minute) can be entirely replaced by the generic version in lib/ratelimit.c, used around the kernel. Do so. The only functional difference should be that the new logs will read something like: KERN_WARNING "print_bad_page_map: %d callbacks suppressed\n", ... But that should be fine enough. Link: https://lore.kernel.org/20260811172156.356053-1-pfalcato@suse.de Link: https://lore.kernel.org/20260811172156.356053-2-pfalcato@suse.de Signed-off-by: Pedro Falcato Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Acked-by: Zi Yan Reviewed-by: SJ Park Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Acked-by: Mike Rapoport (Microsoft) Cc: Brendan Jackman Cc: Liam R. Howlett Cc: Michal Hocko Cc: Suren Baghdasaryan --- mm/memory.c | 30 +++--------------------------- 1 file changed, 3 insertions(+), 27 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index 347db2acd0f83b..09ac784f8b7b39 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -492,32 +492,8 @@ static inline void add_mm_rss_vec(struct mm_struct *mm, int *rss) add_mm_counter(mm, i, rss[i]); } -static bool is_bad_page_map_ratelimited(void) -{ - static unsigned long resume; - static unsigned long nr_shown; - static unsigned long nr_unshown; - - /* - * Allow a burst of 60 reports, then keep quiet for that minute; - * or allow a steady drip of one report per second. - */ - if (nr_shown == 60) { - if (time_before(jiffies, resume)) { - nr_unshown++; - return true; - } - if (nr_unshown) { - pr_alert("BUG: Bad page map: %lu messages suppressed\n", - nr_unshown); - nr_unshown = 0; - } - nr_shown = 0; - } - if (nr_shown++ == 0) - resume = jiffies + 60 * HZ; - return false; -} +/* Allow a burst of 60 bad page map reports per minute. */ +static DEFINE_RATELIMIT_STATE(bad_page_map_ratelimit, 60 * HZ, 60); static void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size) { @@ -633,7 +609,7 @@ static void print_bad_page_map(struct vm_area_struct *vma, char entry_str[PTVAL_STR_MAX]; pgoff_t index, anon_index; - if (is_bad_page_map_ratelimited()) + if (!__ratelimit(&bad_page_map_ratelimit)) return; mapping = vma->vm_file ? vma->vm_file->f_mapping : NULL; From 29c5c87ed8101567656b8667aea37dfe41163929 Mon Sep 17 00:00:00 2001 From: Pedro Falcato Date: Tue, 11 Aug 2026 18:21:56 +0100 Subject: [PATCH 0312/1012] mm/page_alloc: replace custom bad page ratelimiting logic The current logic (allow up to $BURST prints per minute) can be entirely replaced by the generic version in lib/ratelimit.c, used around the kernel. Do so. The only functional difference should be that the new logs will read something like: KERN_WARNING "bad_page: %d callbacks suppressed\n", ... But that should be fine enough. Link: https://lore.kernel.org/20260811172156.356053-3-pfalcato@suse.de Signed-off-by: Pedro Falcato Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Acked-by: Zi Yan Reviewed-by: SJ Park Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Acked-by: Mike Rapoport (Microsoft) Cc: Brendan Jackman Cc: Liam R. Howlett Cc: Michal Hocko Cc: Suren Baghdasaryan --- mm/page_alloc.c | 28 +++++----------------------- 1 file changed, 5 insertions(+), 23 deletions(-) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index aaeb30a7524b1b..be1b6672368296 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -613,31 +613,13 @@ static inline bool __maybe_unused bad_range(struct zone *zone, struct page *page } #endif +/* Allow a burst of 60 reports per minute */ +static DEFINE_RATELIMIT_STATE(bad_page_ratelimit, 60 * HZ, 60); + static void bad_page(struct page *page, const char *reason) { - static unsigned long resume; - static unsigned long nr_shown; - static unsigned long nr_unshown; - - /* - * Allow a burst of 60 reports, then keep quiet for that minute; - * or allow a steady drip of one report per second. - */ - if (nr_shown == 60) { - if (time_before(jiffies, resume)) { - nr_unshown++; - goto out; - } - if (nr_unshown) { - pr_alert( - "BUG: Bad page state: %lu messages suppressed\n", - nr_unshown); - nr_unshown = 0; - } - nr_shown = 0; - } - if (nr_shown++ == 0) - resume = jiffies + 60 * HZ; + if (!__ratelimit(&bad_page_ratelimit)) + goto out; pr_alert("BUG: Bad page state in process %s pfn:%05lx\n", current->comm, page_to_pfn(page)); From b47870e46faf23308ac146136dd1f7f7708e59ca Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Mon, 31 Aug 2026 14:42:11 +0100 Subject: [PATCH 0313/1012] mm/vmpressure: remove window size TODO There has been a steady stream of patches that have been submitted by newcomers to core mm 'fixing' this TODO, with all but the original having very likely been generated by LLMs. It appears that TODOs to LLMs are like red rags to a bull. In addition, TODOs in code often bitrot and are distracting - those who understand the code know what could be improved in future. Therefore remove the TODO. The work required to actually fix this TODO requires somebody who both has understanding of the code and significant real-world data to back their changes. Such a person doesn't require a TODO prompt to implement this change, so nothing of value is being lost here. Link: https://lore.kernel.org/all/20260831130316.448-1-tahasezer.is@gmail.com/ Link: https://lore.kernel.org/linux-mm/20260724054305.516126-1-cui.tao@linux.dev/ Link: https://lore.kernel.org/linux-mm/20260715143646.15828-1-gaikwad.dcg@gmail.com/ Link: https://lore.kernel.org/all/20260227221555.29969-1-mcq@disroot.org/ Link: https://lore.kernel.org/20260831-remove-vmpressure-todo-v1-1-498515e59cdf@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Vlastimil Babka (SUSE) Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan --- mm/vmpressure.c | 3 --- 1 file changed, 3 deletions(-) diff --git a/mm/vmpressure.c b/mm/vmpressure.c index 9629240d77adc7..3de99fef392894 100644 --- a/mm/vmpressure.c +++ b/mm/vmpressure.c @@ -30,9 +30,6 @@ * * As the vmscan reclaimer logic works with chunks which are multiple of * SWAP_CLUSTER_MAX, it makes sense to use it for the window size as well. - * - * TODO: Make the window size depend on machine size, as we do for vmstat - * thresholds. Currently we set it to 512 pages (2MB for 4KB pages). */ const unsigned long vmpressure_win = SWAP_CLUSTER_MAX * 16; From 4433472d6c16227ebd69ffc7f450a64f9416c29f Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Mon, 31 Aug 2026 15:18:23 +0100 Subject: [PATCH 0314/1012] tools/testing/selftests/mm: add missing .gitignore entries Commit 2bee308f3adb ("selftests/mm: use pattern matching in .gitignore") switched to a pattern-matching mechanism to reduce churn in .gitignore. It however accidentally excluded the page_frag test's-generated module intermediate C file with .mod.c extension, and also the local_config.h header generated if liburing is available locally. Explicitly fix both the issues, fixing the module-generated C file as a general pattern as these are always intermediate files that should be ignored. Since this is a trivial .gitignore change it doesn't seem necessary to treat it as a hotfix. Link: https://lore.kernel.org/20260831-fix-mm-selftests-gitignore-v1-1-c984bbd4c5e4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Gregory Price (Meta) Reviewed-by: Sarthak Sharma Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/.gitignore | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tools/testing/selftests/mm/.gitignore b/tools/testing/selftests/mm/.gitignore index fcd892ed21e32c..a306d775478690 100644 --- a/tools/testing/selftests/mm/.gitignore +++ b/tools/testing/selftests/mm/.gitignore @@ -2,7 +2,9 @@ * !/**/ !*.c +*.mod.c !*.h +local_config.h !*.sh !.gitignore !Makefile From 44bb6ac73a5c30d1edfa8e2d2b2c8c7b2a50cbc0 Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:52 -0700 Subject: [PATCH 0315/1012] mm: make per-VMA locks available universally MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "mm: Unconditional per-VMA locks and cleanups", v7. tl;dr: Make per-VMA locks available in all configs. Simplify some of the per-VMA lock users now that they can rely on them being always available. Binder and networking folks: Your code is the target of the cleanups. I'm cc'ing you now on v2 because there's emerging consensus on the mm side that the approach here is sane. I'm not quite sure how this pile would get merged, but ack/review tags would be appreciated if this looks good to you. Longer version: When working on some x86 shadow stack code, it was a real pain to avoid causing recursive locking problems with mmap_lock. One way to avoid those was to avoid mmap_lock and use per-VMA locks instead. They are great, but they are not available in all configs which makes them unusable in generic code, or if you want to completely avoid mmap_lock. Make per-VMA locks available in all configs. Right now, they are only available on select architectures when SMP and MMU are enabled. But all of the primitives that per-VMA locks are built on (RCU, maple trees, refcounts) work just fine without SMP or MMU. The only real downside is that making VMAs a wee bit bigger on !MMU and !SMP builds. The upside is much cleaner code, lower complexity and less #ifdeffery. Clean up a binder VMA locking site now that it can rely on per-VMA locks. Building on top of universally-available per-VMA locks, introduce a new helper. Since the new API does not require callers to have a fallback to mmap_lock, it's much easier to use. Callers can potentially replace this very common kernel idiom: mmap_read_lock(mm); vma = vma_lookup() // fiddle with vma mmap_read_unlock(mm); with: vma = vma_start_read_unlocked(mm, address); // fiddle with vma vma_end_read(vma); Which avoids mmap_lock entirely in the fast path. Use that new API for another binder site and one in the TCP code. This patch (of 7): The per-VMA locks have been around for several years. They've had some bugs worked out of them and have seen quite wide use. However, they are still only available when architectures explicitly enable them. Remove the conditional compilation around the per-VMA locks, making them available on all architectures and configs. The approach up to now seemed to be to add ARCH_SUPPORTS_PER_VMA_LOCK when the architecture started using per-VMA locks in the fault handler. But, contrary to the naming, the Kconfig option does not really indicate whether the architecture supports per-VMA locks or not. It is more of a marker for whether the architecture is likely to benefit from per-VMA locks. To me, the most important thing side-effect of universal availability is letting per-VMA locks be used in SMP=n configs. This lets us use per-VMA locking in all x86 code without fallbacks. Overall, this just generally makes the kernel simpler. Just look at the diffstat. It also opens the door to users that want to use the per-VMA locks in common code. Doing *that* brings additional simplifications. The downside of this is adding some fields to vm_area_struct and mm_struct. There are likely ways to optimize this, especially for things like SMP=n configs. For now, do the simplest thing: use the same implementation everywhere. == Considerations for NOMMU config == NOMMU systems do not write-lock VMAs, therefore read-locking a VMA would always succeed unless VMA is detached. Therefore for NOMMU config we make vma_mark_attached() a NOOP, which keeps VMAs always in detached state. This causes VMA read-locking to always fail and the caller falls back to locking mmap_lock. The following functions will have a different implementation in NOMMU config: - vma_mark_attached(), vma_mark_detached() are made NOOPs, keeping VMAs always in a detached state and preventing assertions and refcount underflows; - vma_start_write(), vma_start_write_killable() are made NOOPs to avoid warnings in __vma_start_write() due to VMAs being detached. These functions are not used in NOMMU code but __vma_start_write() is an exported function, therefore might be used by drivers. - vma_assert_attached() is made NOOP because it's reachable from NOMMU code via split_vma()->vma_iter_store_new()->vma_iter_store_overwrite(); - vma_assert_write_locked() is asserting vma->vm_mm is write-locked, as was done before this change; - vma_assert_locked() is asserting vma->vm_mm is locked, as was done before this change; The following functions work for both MMU and NOMMU configs: - vma_lock_init() performs the same initialization as for MMU config; - mm_lock_seqcount_init(), mm_lock_seqcount_begin(), mm_lock_seqcount_end() are called from mmap_write_{lock|unlock} and update mm_lock_seq correctly. - mmap_lock_speculate_try_begin(), mmap_lock_speculate_retry() work as is because mm_lock_seq is updated correctly; - vma_start_read(), vma_start_read_locked() will always fail because VMAs are always detached; - vma_end_read() will never be called because vma_start_read() never succeeds; - vma_is_attached() always return false because VMAs are always detached; - vma_assert_detached() will never trigger because VMAs are never attached; - vma_start_read_locked() always return false because VMAs are always detached; - lock_vma_under_rcu() will be safe as the attempted read lock will bail; Changes in the following files are not affecting NOMMU config: task_mmu.c - not compiled when CONFIG_MMU=n; pagewalk.c - not compiled when CONFIG_MMU=n; userfaultfd.c - not compiled when CONFIG_MMU=n (CONFIG_USERFAULTFD depends on CONFIG_MMU); The following changes in the BPF code are made to keep NOMMU config working like before: stack_map_lock_vma() - keeps mmap_lock in NOMMU config; bpf_iter_task_vma_new() - bails out in NOMMU config; Link: https://lore.kernel.org/20260831203056.838265-1-surenb@google.com Link: https://lore.kernel.org/20260831203056.838265-2-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Todd Kjos Cc: Christian Brauner Cc: Carlos Llamas Cc: Alice Ryhl Cc: David S. Miller Cc: David Ahern Cc: Arve Hjønnevåg --- arch/arm/Kconfig | 1 - arch/arm64/Kconfig | 1 - arch/loongarch/Kconfig | 1 - arch/powerpc/platforms/powernv/Kconfig | 1 - arch/powerpc/platforms/pseries/Kconfig | 1 - arch/riscv/Kconfig | 1 - arch/s390/Kconfig | 1 - arch/x86/Kconfig | 2 - fs/proc/internal.h | 2 - fs/proc/task_mmu.c | 93 -------------------------- include/linux/mm.h | 12 ---- include/linux/mm_types.h | 8 +-- include/linux/mmap_lock.h | 75 +++++++-------------- kernel/bpf/stackmap.c | 17 ++--- kernel/bpf/task_iter.c | 2 +- kernel/fork.c | 2 - mm/Kconfig | 12 ---- mm/Kconfig.debug | 1 - mm/debug.c | 4 -- mm/init-mm.c | 2 - mm/memory.c | 2 - mm/mmap_lock.c | 26 +------ mm/pagewalk.c | 2 - mm/rmap.c | 2 - mm/userfaultfd.c | 55 --------------- rust/kernel/mm.rs | 32 +++------ tools/testing/vma/include/dup.h | 5 +- tools/testing/vma/vma_internal.h | 1 - 28 files changed, 48 insertions(+), 316 deletions(-) diff --git a/arch/arm/Kconfig b/arch/arm/Kconfig index ffbc7f38613151..408aa58a2a5bbc 100644 --- a/arch/arm/Kconfig +++ b/arch/arm/Kconfig @@ -42,7 +42,6 @@ config ARM select ARCH_SUPPORTS_ATOMIC_RMW select ARCH_SUPPORTS_CFI select ARCH_SUPPORTS_HUGETLBFS if ARM_LPAE - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_SUPPORTS_RT select ARCH_USE_BUILTIN_BSWAP select ARCH_USE_CMPXCHG_LOCKREF diff --git a/arch/arm64/Kconfig b/arch/arm64/Kconfig index b5a51b0ef9440a..2bbeded33da0da 100644 --- a/arch/arm64/Kconfig +++ b/arch/arm64/Kconfig @@ -81,7 +81,6 @@ config ARM64 select ARCH_HAS_PTE_PROTNONE select ARCH_SUPPORTS_NUMA_BALANCING select ARCH_SUPPORTS_PAGE_TABLE_CHECK - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_SUPPORTS_HUGE_PFNMAP if TRANSPARENT_HUGEPAGE select ARCH_SUPPORTS_RT select ARCH_SUPPORTS_SCHED_SMT diff --git a/arch/loongarch/Kconfig b/arch/loongarch/Kconfig index 2067d1f2ad7acb..1d8fb1e456d6d8 100644 --- a/arch/loongarch/Kconfig +++ b/arch/loongarch/Kconfig @@ -69,7 +69,6 @@ config LOONGARCH select ARCH_SUPPORTS_MSEAL_SYSTEM_MAPPINGS select ARCH_HAS_PTE_PROTNONE if 64BIT select ARCH_SUPPORTS_NUMA_BALANCING if NUMA - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_SUPPORTS_RT select ARCH_SUPPORTS_SCHED_SMT if SMP select ARCH_SUPPORTS_SCHED_MC if SMP diff --git a/arch/powerpc/platforms/powernv/Kconfig b/arch/powerpc/platforms/powernv/Kconfig index b5ad7c173ef0c1..dd8f6060fb7a2e 100644 --- a/arch/powerpc/platforms/powernv/Kconfig +++ b/arch/powerpc/platforms/powernv/Kconfig @@ -17,7 +17,6 @@ config PPC_POWERNV select PPC_DOORBELL select MMU_NOTIFIER select FORCE_SMP - select ARCH_SUPPORTS_PER_VMA_LOCK select PPC_RADIX_BROADCAST_TLBIE if PPC_RADIX_MMU default y diff --git a/arch/powerpc/platforms/pseries/Kconfig b/arch/powerpc/platforms/pseries/Kconfig index 74910ce3a541c3..7d125e288f6ef7 100644 --- a/arch/powerpc/platforms/pseries/Kconfig +++ b/arch/powerpc/platforms/pseries/Kconfig @@ -23,7 +23,6 @@ config PPC_PSERIES select HOTPLUG_CPU select FORCE_SMP select SWIOTLB - select ARCH_SUPPORTS_PER_VMA_LOCK select PPC_RADIX_BROADCAST_TLBIE if PPC_RADIX_MMU default y diff --git a/arch/riscv/Kconfig b/arch/riscv/Kconfig index d6c2dbf8455ced..5965666194b0f0 100644 --- a/arch/riscv/Kconfig +++ b/arch/riscv/Kconfig @@ -72,7 +72,6 @@ config RISCV select ARCH_SUPPORTS_LTO_CLANG_THIN select ARCH_SUPPORTS_MSEAL_SYSTEM_MAPPINGS if 64BIT && MMU select ARCH_SUPPORTS_PAGE_TABLE_CHECK if MMU - select ARCH_SUPPORTS_PER_VMA_LOCK if MMU select ARCH_HAS_PTE_PROTNONE if MMU select ARCH_SUPPORTS_RT select ARCH_SUPPORTS_SHADOW_CALL_STACK if HAVE_SHADOW_CALL_STACK diff --git a/arch/s390/Kconfig b/arch/s390/Kconfig index 4b51bc6e8948d7..b88b8504213692 100644 --- a/arch/s390/Kconfig +++ b/arch/s390/Kconfig @@ -156,7 +156,6 @@ config S390 select ARCH_HAS_PTE_PROTNONE select ARCH_SUPPORTS_NUMA_BALANCING select ARCH_SUPPORTS_PAGE_TABLE_CHECK - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_USES_CFI_GENERIC_LLVM_PASS if CC_IS_CLANG select ARCH_USE_BUILTIN_BSWAP select ARCH_USE_CMPXCHG_LOCKREF diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig index 7aa74bcc72f9db..a8c3b3d31a2761 100644 --- a/arch/x86/Kconfig +++ b/arch/x86/Kconfig @@ -27,7 +27,6 @@ config X86_64 select ARCH_HAS_GIGANTIC_PAGE select ARCH_SUPPORTS_MSEAL_SYSTEM_MAPPINGS select ARCH_SUPPORTS_INT128 if CC_HAS_INT128 - select ARCH_SUPPORTS_PER_VMA_LOCK select ARCH_SUPPORTS_HUGE_PFNMAP if TRANSPARENT_HUGEPAGE select HAVE_ARCH_SOFT_DIRTY select MODULES_USE_ELF_RELA @@ -1848,7 +1847,6 @@ config X86_USER_SHADOW_STACK bool "X86 userspace shadow stack" depends on AS_WRUSS depends on X86_64 - depends on PER_VMA_LOCK select ARCH_USES_HIGH_VMA_FLAGS select ARCH_HAS_USER_SHADOW_STACK select X86_CET diff --git a/fs/proc/internal.h b/fs/proc/internal.h index 04bd6c9e65a722..623bb43ede5509 100644 --- a/fs/proc/internal.h +++ b/fs/proc/internal.h @@ -385,10 +385,8 @@ struct mem_size_stats; struct proc_maps_locking_ctx { struct mm_struct *mm; -#ifdef CONFIG_PER_VMA_LOCK bool mmap_locked; struct vm_area_struct *locked_vma; -#endif }; struct proc_maps_private { diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 5c54aebe211824..e671b4fd8dedd9 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -130,8 +130,6 @@ static void release_task_mempolicy(struct proc_maps_private *priv) } #endif -#ifdef CONFIG_PER_VMA_LOCK - static inline int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) { int ret = mmap_read_lock_killable(lock_ctx->mm); @@ -233,46 +231,6 @@ static inline void reacquire_rcu(struct proc_maps_private *priv) vma_iter_set(&priv->iter, priv->lock_ctx.locked_vma->vm_end); } -#else /* CONFIG_PER_VMA_LOCK */ - -static inline int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) -{ - return mmap_read_lock_killable(lock_ctx->mm); -} - -static inline void unlock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) -{ - mmap_read_unlock(lock_ctx->mm); -} - -static inline bool lock_vma_range(struct seq_file *m, - struct proc_maps_locking_ctx *lock_ctx) -{ - return lock_ctx_mm(lock_ctx) == 0; -} - -static inline void unlock_vma_range(struct proc_maps_locking_ctx *lock_ctx) -{ - unlock_ctx_mm(lock_ctx); -} - -static struct vm_area_struct *get_next_vma(struct proc_maps_private *priv, - loff_t last_pos) -{ - return vma_next(&priv->iter); -} - -static inline bool fallback_to_mmap_lock(struct proc_maps_private *priv, - loff_t pos) -{ - return false; -} - -static inline void drop_rcu(struct proc_maps_private *priv) {} -static inline void reacquire_rcu(struct proc_maps_private *priv) {} - -#endif /* CONFIG_PER_VMA_LOCK */ - static struct vm_area_struct *proc_get_vma(struct seq_file *m, loff_t *ppos) { struct proc_maps_private *priv = m->private; @@ -560,8 +518,6 @@ static int pid_maps_open(struct inode *inode, struct file *file) PROCMAP_QUERY_VMA_FLAGS \ ) -#ifdef CONFIG_PER_VMA_LOCK - static int query_vma_setup(struct proc_maps_locking_ctx *lock_ctx) { reset_lock_ctx(lock_ctx); @@ -612,26 +568,6 @@ static struct vm_area_struct *query_vma_find_by_addr(struct proc_maps_locking_ct return vma; } -#else /* CONFIG_PER_VMA_LOCK */ - -static int query_vma_setup(struct proc_maps_locking_ctx *lock_ctx) -{ - return mmap_read_lock_killable(lock_ctx->mm); -} - -static void query_vma_teardown(struct proc_maps_locking_ctx *lock_ctx) -{ - mmap_read_unlock(lock_ctx->mm); -} - -static struct vm_area_struct *query_vma_find_by_addr(struct proc_maps_locking_ctx *lock_ctx, - unsigned long addr) -{ - return find_vma(lock_ctx->mm, addr); -} - -#endif /* CONFIG_PER_VMA_LOCK */ - static struct vm_area_struct *query_matching_vma(struct proc_maps_locking_ctx *lock_ctx, unsigned long addr, u32 flags) { @@ -1314,8 +1250,6 @@ static const struct mm_walk_ops smaps_shmem_walk_ops = { .walk_lock = PGWALK_RDLOCK, }; -#ifdef CONFIG_PER_VMA_LOCK - static const struct mm_walk_ops smaps_walk_vma_lock_ops = { .pmd_entry = smaps_pte_range, .hugetlb_entry = smaps_hugetlb_range, @@ -1345,22 +1279,6 @@ get_smaps_shmem_walk_ops(struct proc_maps_private *priv) return &smaps_shmem_walk_vma_lock_ops; } -#else /* CONFIG_PER_VMA_LOCK */ - -static inline const struct mm_walk_ops * -get_smaps_walk_ops(struct proc_maps_private *priv) -{ - return &smaps_walk_ops; -} - -static inline const struct mm_walk_ops * -get_smaps_shmem_walk_ops(struct proc_maps_private *priv) -{ - return &smaps_shmem_walk_ops; -} - -#endif /* CONFIG_PER_VMA_LOCK */ - /* * Gather mem stats from @vma with the indicated beginning * address @start, and keep them in @mss. @@ -3497,7 +3415,6 @@ static const struct mm_walk_ops show_numa_ops = { .walk_lock = PGWALK_RDLOCK, }; -#ifdef CONFIG_PER_VMA_LOCK static const struct mm_walk_ops show_numa_vma_lock_ops = { .hugetlb_entry = gather_hugetlb_stats, .pmd_entry = gather_pte_stats, @@ -3512,16 +3429,6 @@ get_show_numa_ops(struct proc_maps_private *priv) return &show_numa_vma_lock_ops; } -#else /* CONFIG_PER_VMA_LOCK */ - -static inline const struct mm_walk_ops * -get_show_numa_ops(struct proc_maps_private *priv) -{ - return &show_numa_ops; -} - -#endif /* CONFIG_PER_VMA_LOCK */ - /* * Display pages allocated per node and memory policy via /proc. */ diff --git a/include/linux/mm.h b/include/linux/mm.h index b19711b6dbc69a..a9fbe26536f450 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -928,7 +928,6 @@ static inline void vma_numab_state_free(struct vm_area_struct *vma) {} * These must be here rather than mmap_lock.h as dependent on vm_fault type, * declared in this header. */ -#ifdef CONFIG_PER_VMA_LOCK static inline void release_fault_lock(struct vm_fault *vmf) { if (vmf->flags & FAULT_FLAG_VMA_LOCK) @@ -944,17 +943,6 @@ static inline void assert_fault_locked(const struct vm_fault *vmf) else mmap_assert_locked(vmf->vma->vm_mm); } -#else -static inline void release_fault_lock(struct vm_fault *vmf) -{ - mmap_read_unlock(vmf->vma->vm_mm); -} - -static inline void assert_fault_locked(const struct vm_fault *vmf) -{ - mmap_assert_locked(vmf->vma->vm_mm); -} -#endif /* CONFIG_PER_VMA_LOCK */ static inline bool mm_flags_test(int flag, const struct mm_struct *mm) { diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index f3e5a2fadbe5b7..2a3988178adfdc 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -950,7 +950,6 @@ struct vm_area_struct { vma_flags_t flags; }; -#ifdef CONFIG_PER_VMA_LOCK /* * Can only be written (using WRITE_ONCE()) while holding both: * - mmap_lock (in write mode) @@ -966,7 +965,7 @@ struct vm_area_struct { * slowpath. */ unsigned int vm_lock_seq; -#endif + /* * Low 32-bits of anonymous page offset. * See vma_start_anon_pgoff() comment for details. @@ -1003,7 +1002,6 @@ struct vm_area_struct { #ifdef CONFIG_NUMA_BALANCING struct vma_numab_state *numab_state; /* NUMA Balancing state */ #endif -#ifdef CONFIG_PER_VMA_LOCK /* * Used to keep track of firstly, whether the VMA is attached, secondly, * if attached, how many read locks are taken, and thirdly, if the @@ -1046,7 +1044,6 @@ struct vm_area_struct { #ifdef CONFIG_DEBUG_LOCK_ALLOC struct lockdep_map vmlock_dep_map; #endif -#endif #ifdef CONFIG_64BIT /* * High 32-bits of anonymous page offset. @@ -1254,7 +1251,6 @@ struct mm_struct { * init_mm.mmlist, and are protected * by mmlist_lock */ -#ifdef CONFIG_PER_VMA_LOCK struct rcuwait vma_writer_wait; /* * This field has lock-like semantics, meaning it is sometimes @@ -1274,7 +1270,7 @@ struct mm_struct { * mmap_lock. */ seqcount_t mm_lock_seq; -#endif + struct futex_mm_data futex; unsigned long hiwater_rss; /* High-watermark of RSS usage */ diff --git a/include/linux/mmap_lock.h b/include/linux/mmap_lock.h index b8a13b8d36a45b..6a0a8cf501bdea 100644 --- a/include/linux/mmap_lock.h +++ b/include/linux/mmap_lock.h @@ -76,8 +76,6 @@ static inline void mmap_assert_write_locked(const struct mm_struct *mm) rwsem_assert_held_write(&mm->mmap_lock); } -#ifdef CONFIG_PER_VMA_LOCK - #ifdef CONFIG_LOCKDEP #define __vma_lockdep_map(vma) (&vma->vmlock_dep_map) #else @@ -297,6 +295,9 @@ int __vma_start_write(struct vm_area_struct *vma, int state); */ static inline void vma_start_write(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + if (__is_vma_write_locked(vma)) return; @@ -319,6 +320,9 @@ static inline void vma_start_write(struct vm_area_struct *vma) static inline __must_check int vma_start_write_killable(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return 0; + if (__is_vma_write_locked(vma)) return 0; @@ -331,6 +335,11 @@ int vma_start_write_killable(struct vm_area_struct *vma) */ static inline void vma_assert_write_locked(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) { + mmap_assert_write_locked(vma->vm_mm); + return; + } + VM_WARN_ON_ONCE_VMA(!__is_vma_write_locked(vma), vma); } @@ -343,6 +352,11 @@ static inline void vma_assert_locked(struct vm_area_struct *vma) { unsigned int refcnt; + if (!IS_ENABLED(CONFIG_MMU)) { + mmap_assert_locked(vma->vm_mm); + return; + } + if (IS_ENABLED(CONFIG_LOCKDEP)) { if (!lock_is_held(__vma_lockdep_map(vma))) vma_assert_write_locked(vma); @@ -432,6 +446,9 @@ static inline bool vma_is_attached(struct vm_area_struct *vma) */ static inline void vma_assert_attached(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + WARN_ON_ONCE(!vma_is_attached(vma)); } @@ -442,6 +459,9 @@ static inline void vma_assert_detached(struct vm_area_struct *vma) static inline void vma_mark_attached(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + vma_assert_write_locked(vma); vma_assert_detached(vma); refcount_set_release(&vma->vm_refcnt, 1); @@ -451,6 +471,9 @@ void __vma_exclude_readers_for_detach(struct vm_area_struct *vma); static inline void vma_mark_detached(struct vm_area_struct *vma) { + if (!IS_ENABLED(CONFIG_MMU)) + return; + vma_assert_write_locked(vma); vma_assert_attached(vma); @@ -484,54 +507,6 @@ struct vm_area_struct *lock_next_vma(struct mm_struct *mm, struct vma_iterator *iter, unsigned long address); -#else /* CONFIG_PER_VMA_LOCK */ - -static inline void mm_lock_seqcount_init(struct mm_struct *mm) {} -static inline void mm_lock_seqcount_begin(struct mm_struct *mm) {} -static inline void mm_lock_seqcount_end(struct mm_struct *mm) {} - -static inline bool mmap_lock_speculate_try_begin(struct mm_struct *mm, unsigned int *seq) -{ - return false; -} - -static inline bool mmap_lock_speculate_retry(struct mm_struct *mm, unsigned int seq) -{ - return true; -} -static inline void vma_lock_init(struct vm_area_struct *vma, bool reset_refcnt) {} -static inline void vma_end_read(struct vm_area_struct *vma) {} -static inline void vma_start_write(struct vm_area_struct *vma) {} -static inline __must_check -int vma_start_write_killable(struct vm_area_struct *vma) { return 0; } -static inline void vma_assert_write_locked(struct vm_area_struct *vma) - { mmap_assert_write_locked(vma->vm_mm); } -static inline bool vma_is_attached(struct vm_area_struct *vma) - { return true; } -static inline void vma_assert_attached(struct vm_area_struct *vma) {} -static inline void vma_assert_detached(struct vm_area_struct *vma) {} -static inline void vma_mark_attached(struct vm_area_struct *vma) {} -static inline void vma_mark_detached(struct vm_area_struct *vma) {} - -static inline struct vm_area_struct *lock_vma_under_rcu(struct mm_struct *mm, - unsigned long address) -{ - return NULL; -} - -static inline void vma_assert_locked(struct vm_area_struct *vma) -{ - mmap_assert_locked(vma->vm_mm); -} - -static inline void vma_assert_stabilised(struct vm_area_struct *vma) -{ - /* If no VMA locks, then either mmap lock suffices to stabilise. */ - mmap_assert_locked(vma->vm_mm); -} - -#endif /* CONFIG_PER_VMA_LOCK */ - static inline void vma_assert_can_modify(struct vm_area_struct *vma) { if (vma_is_attached(vma)) diff --git a/kernel/bpf/stackmap.c b/kernel/bpf/stackmap.c index d09d4c3fe547c6..f7e8d766d12834 100644 --- a/kernel/bpf/stackmap.c +++ b/kernel/bpf/stackmap.c @@ -272,13 +272,10 @@ struct stack_map_vma_lock { /* * Acquire a stable read-side reference on the VMA covering @ip. * - * With CONFIG_PER_VMA_LOCK=y this returns a VMA with its per-VMA read - * lock held and mmap_lock dropped, so the caller may sleep. - * - * With CONFIG_PER_VMA_LOCK=n it returns a VMA with mmap_lock still - * held; the caller must snapshot any fields it needs and pin vm_file - * with get_file() before stack_map_unlock_vma() drops mmap_lock, as - * the VMA may be split, merged, or freed after that. + * On NOMMU configurations, returns with the mmap_lock held. If the MMU + * is enabled, the per-VMA lock will be held instead. The lock + * should be released with stack_map_unlock_vma() which will release the + * appropriate lock. Once the lock is released, the VMA may be freed. * * Returns NULL on failure, in which case no lock is held. */ @@ -288,7 +285,6 @@ stack_map_lock_vma(struct stack_map_vma_lock *lock, unsigned long ip) struct mm_struct *mm = lock->mm; struct vm_area_struct *vma; - /* noop under !CONFIG_PER_VMA_LOCK */ vma = lock_vma_under_rcu(mm, ip); if (vma) { lock->vma = vma; @@ -308,21 +304,20 @@ stack_map_lock_vma(struct stack_map_vma_lock *lock, unsigned long ip) return NULL; } -#ifdef CONFIG_PER_VMA_LOCK +#ifdef CONFIG_MMU if (!vma_start_read_locked(vma)) { mmap_read_unlock(mm); return NULL; } mmap_read_unlock(mm); #endif - lock->vma = vma; return vma; } static void stack_map_unlock_vma(struct stack_map_vma_lock *lock) { -#ifdef CONFIG_PER_VMA_LOCK +#ifdef CONFIG_MMU vma_end_read(lock->vma); #else mmap_read_unlock(lock->mm); diff --git a/kernel/bpf/task_iter.c b/kernel/bpf/task_iter.c index 13e1aabe6f8868..c65ba1dcd86672 100644 --- a/kernel/bpf/task_iter.c +++ b/kernel/bpf/task_iter.c @@ -869,7 +869,7 @@ __bpf_kfunc int bpf_iter_task_vma_new(struct bpf_iter_task_vma *it, BUILD_BUG_ON(sizeof(struct bpf_iter_task_vma_kern) != sizeof(struct bpf_iter_task_vma)); BUILD_BUG_ON(__alignof__(struct bpf_iter_task_vma_kern) != __alignof__(struct bpf_iter_task_vma)); - if (!IS_ENABLED(CONFIG_PER_VMA_LOCK)) { + if (!IS_ENABLED(CONFIG_MMU)) { kit->data = NULL; return -EOPNOTSUPP; } diff --git a/kernel/fork.c b/kernel/fork.c index 10f2d05d816a5f..22eaf5fb844d0a 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -1083,9 +1083,7 @@ static void mmap_init_lock(struct mm_struct *mm) { init_rwsem(&mm->mmap_lock); mm_lock_seqcount_init(mm); -#ifdef CONFIG_PER_VMA_LOCK rcuwait_init(&mm->vma_writer_wait); -#endif } static struct mm_struct *mm_init(struct mm_struct *mm, struct task_struct *p) diff --git a/mm/Kconfig b/mm/Kconfig index 2c385f8b29445e..c1ddf59c0d71a8 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -1425,18 +1425,6 @@ config LRU_GEN_WALKS_MMU depends on LRU_GEN && ARCH_HAS_HW_PTE_YOUNG # } -config ARCH_SUPPORTS_PER_VMA_LOCK - def_bool n - -config PER_VMA_LOCK - def_bool y - depends on ARCH_SUPPORTS_PER_VMA_LOCK && MMU && SMP - help - Allow per-vma locking during page fault handling. - - This feature allows locking each virtual memory area separately when - handling page faults instead of taking mmap_lock. - config LOCK_MM_AND_FIND_VMA bool depends on !STACK_GROWSUP diff --git a/mm/Kconfig.debug b/mm/Kconfig.debug index 15dca19dd07da9..9eaa25d1cf2340 100644 --- a/mm/Kconfig.debug +++ b/mm/Kconfig.debug @@ -310,7 +310,6 @@ config DEBUG_KMEMLEAK_VERBOSE config PER_VMA_LOCK_STATS bool "Statistics for per-vma locks" - depends on PER_VMA_LOCK help Say Y here to enable success, retry and failure counters of page faults handled under protection of per-vma locks. When enabled, the diff --git a/mm/debug.c b/mm/debug.c index 9a0297b3988d89..655e6bcc0e8d91 100644 --- a/mm/debug.c +++ b/mm/debug.c @@ -157,17 +157,13 @@ void dump_vma(const struct vm_area_struct *vma) pr_emerg("vma %px start %px end %px mm %px\n" "prot %lx anon_vma %px vm_ops %px\n" "pgoff %lx file %px private_data %px\n" -#ifdef CONFIG_PER_VMA_LOCK "refcnt %x\n" -#endif "flags: %#lx(%pGv)\n", vma, (void *)vma->vm_start, (void *)vma->vm_end, vma->vm_mm, (unsigned long)pgprot_val(vma->vm_page_prot), vma->anon_vma, vma->vm_ops, vma_start_pgoff(vma), vma->vm_file, vma->vm_private_data, -#ifdef CONFIG_PER_VMA_LOCK refcount_read(&vma->vm_refcnt), -#endif vma->vm_flags, &vma->vm_flags); } EXPORT_SYMBOL(dump_vma); diff --git a/mm/init-mm.c b/mm/init-mm.c index 3e792aad762616..a1bb2c2d0284a1 100644 --- a/mm/init-mm.c +++ b/mm/init-mm.c @@ -39,10 +39,8 @@ struct mm_struct init_mm = { .page_table_lock = __SPIN_LOCK_UNLOCKED(init_mm.page_table_lock), .arg_lock = __SPIN_LOCK_UNLOCKED(init_mm.arg_lock), .mmlist = LIST_HEAD_INIT(init_mm.mmlist), -#ifdef CONFIG_PER_VMA_LOCK .vma_writer_wait = __RCUWAIT_INITIALIZER(init_mm.vma_writer_wait), .mm_lock_seq = SEQCNT_ZERO(init_mm.mm_lock_seq), -#endif #ifdef CONFIG_SCHED_MM_CID .mm_cid.lock = __RAW_SPIN_LOCK_UNLOCKED(init_mm.mm_cid.lock), #endif diff --git a/mm/memory.c b/mm/memory.c index 09ac784f8b7b39..bc14cae3c49d72 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -6799,7 +6799,6 @@ static vm_fault_t sanitize_fault_flags(struct vm_area_struct *vma, !vma_is_cow_mapping(vma))) return VM_FAULT_SIGSEGV; } -#ifdef CONFIG_PER_VMA_LOCK /* * Per-VMA locks can't be used with FAULT_FLAG_RETRY_NOWAIT because of * the assumption that lock is dropped on VM_FAULT_RETRY. @@ -6808,7 +6807,6 @@ static vm_fault_t sanitize_fault_flags(struct vm_area_struct *vma, (FAULT_FLAG_VMA_LOCK | FAULT_FLAG_RETRY_NOWAIT)) == (FAULT_FLAG_VMA_LOCK | FAULT_FLAG_RETRY_NOWAIT))) return VM_FAULT_SIGSEGV; -#endif return 0; } diff --git a/mm/mmap_lock.c b/mm/mmap_lock.c index 898c2ef1e95803..272f9ac762b91e 100644 --- a/mm/mmap_lock.c +++ b/mm/mmap_lock.c @@ -43,9 +43,6 @@ void __mmap_lock_do_trace_released(struct mm_struct *mm, bool write) EXPORT_SYMBOL(__mmap_lock_do_trace_released); #endif /* CONFIG_TRACING */ -#ifdef CONFIG_MMU -#ifdef CONFIG_PER_VMA_LOCK - /* State shared across __vma_[start, end]_exclude_readers. */ struct vma_exclude_readers_state { /* Input parameters. */ @@ -299,6 +296,8 @@ struct vm_area_struct *lock_vma_under_rcu(struct mm_struct *mm, MA_STATE(mas, &mm->mm_mt, address, address); struct vm_area_struct *vma; + if (!IS_ENABLED(CONFIG_MMU)) + return NULL; retry: rcu_read_lock(); vma = mas_walk(&mas); @@ -431,7 +430,6 @@ struct vm_area_struct *lock_next_vma(struct mm_struct *mm, return vma; } -#endif /* CONFIG_PER_VMA_LOCK */ #ifdef CONFIG_LOCK_MM_AND_FIND_VMA #include @@ -548,23 +546,3 @@ struct vm_area_struct *lock_mm_and_find_vma(struct mm_struct *mm, return NULL; } #endif /* CONFIG_LOCK_MM_AND_FIND_VMA */ - -#else /* CONFIG_MMU */ - -/* - * At least xtensa ends up having protection faults even with no - * MMU.. No stack expansion, at least. - */ -struct vm_area_struct *lock_mm_and_find_vma(struct mm_struct *mm, - unsigned long addr, struct pt_regs *regs) -{ - struct vm_area_struct *vma; - - mmap_read_lock(mm); - vma = vma_lookup(mm, addr); - if (!vma) - mmap_read_unlock(mm); - return vma; -} - -#endif /* CONFIG_MMU */ diff --git a/mm/pagewalk.c b/mm/pagewalk.c index cc07fcf50e87b3..7411702a37f58d 100644 --- a/mm/pagewalk.c +++ b/mm/pagewalk.c @@ -444,7 +444,6 @@ static inline void process_mm_walk_lock(struct mm_struct *mm, static inline void process_vma_walk_lock(struct vm_area_struct *vma, enum page_walk_lock walk_lock) { -#ifdef CONFIG_PER_VMA_LOCK switch (walk_lock) { case PGWALK_WRLOCK: vma_start_write(vma); @@ -459,7 +458,6 @@ static inline void process_vma_walk_lock(struct vm_area_struct *vma, /* PGWALK_RDLOCK is handled by process_mm_walk_lock */ break; } -#endif } /* diff --git a/mm/rmap.c b/mm/rmap.c index f3b21aaa34ee98..3c67ad0e95620f 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -264,11 +264,9 @@ static void check_anon_vma_clone(struct vm_area_struct *dst, /* For the anon_vma to be compatible, it can only be singular. */ VM_WARN_ON_ONCE(operation == VMA_OP_MERGE_UNFAULTED && !list_is_singular(&src->anon_vma_chain)); -#ifdef CONFIG_PER_VMA_LOCK /* Only merging an unfaulted VMA leaves the destination attached. */ VM_WARN_ON_ONCE(operation != VMA_OP_MERGE_UNFAULTED && vma_is_attached(dst)); -#endif } static void maybe_reuse_anon_vma(struct vm_area_struct *dst, diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index b909ec8ef20bce..cf9c6ad3b3ad2c 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -122,7 +122,6 @@ struct vm_area_struct *find_vma_and_prepare_anon(struct mm_struct *mm, return vma; } -#ifdef CONFIG_PER_VMA_LOCK /* * uffd_lock_vma() - Lookup and lock vma corresponding to @address. * @mm: mm to search vma in. @@ -182,34 +181,6 @@ static void uffd_mfill_unlock(struct vm_area_struct *vma) vma_end_read(vma); } -#else - -static struct vm_area_struct *uffd_mfill_lock(struct mm_struct *dst_mm, - unsigned long dst_start, - unsigned long len) -{ - struct vm_area_struct *dst_vma; - - mmap_read_lock(dst_mm); - dst_vma = find_vma_and_prepare_anon(dst_mm, dst_start); - if (IS_ERR(dst_vma)) - goto out_unlock; - - if (validate_dst_vma(dst_vma, dst_start + len)) - return dst_vma; - - dst_vma = ERR_PTR(-ENOENT); -out_unlock: - mmap_read_unlock(dst_mm); - return dst_vma; -} - -static void uffd_mfill_unlock(struct vm_area_struct *vma) -{ - mmap_read_unlock(vma->vm_mm); -} -#endif - static void mfill_put_vma(struct mfill_state *state) { if (!state->vma) @@ -1851,7 +1822,6 @@ int find_vmas_mm_locked(struct mm_struct *mm, return 0; } -#ifdef CONFIG_PER_VMA_LOCK static int uffd_move_lock(struct mm_struct *mm, unsigned long dst_start, unsigned long src_start, @@ -1926,31 +1896,6 @@ static void uffd_move_unlock(struct vm_area_struct *dst_vma, vma_end_read(dst_vma); } -#else - -static int uffd_move_lock(struct mm_struct *mm, - unsigned long dst_start, - unsigned long src_start, - struct vm_area_struct **dst_vmap, - struct vm_area_struct **src_vmap) -{ - int err; - - mmap_read_lock(mm); - err = find_vmas_mm_locked(mm, dst_start, src_start, dst_vmap, src_vmap); - if (err) - mmap_read_unlock(mm); - return err; -} - -static void uffd_move_unlock(struct vm_area_struct *dst_vma, - struct vm_area_struct *src_vma) -{ - mmap_assert_locked(src_vma->vm_mm); - mmap_read_unlock(dst_vma->vm_mm); -} -#endif - /** * move_pages - move arbitrary anonymous pages of an existing vma * @ctx: pointer to the userfaultfd context diff --git a/rust/kernel/mm.rs b/rust/kernel/mm.rs index 4764d7b68f2a7f..f4fa54616085f2 100644 --- a/rust/kernel/mm.rs +++ b/rust/kernel/mm.rs @@ -170,30 +170,20 @@ impl MmWithUser { /// /// This is an optimistic trylock operation, so it may fail if there is contention. In that /// case, you should fall back to taking the mmap read lock. - /// - /// When per-vma locks are disabled, this always returns `None`. #[inline] pub fn lock_vma_under_rcu(&self, vma_addr: usize) -> Option> { - #[cfg(CONFIG_PER_VMA_LOCK)] - { - // SAFETY: Calling `bindings::lock_vma_under_rcu` is always okay given an mm where - // `mm_users` is non-zero. - let vma = unsafe { bindings::lock_vma_under_rcu(self.as_raw(), vma_addr) }; - if !vma.is_null() { - return Some(VmaReadGuard { - // SAFETY: If `lock_vma_under_rcu` returns a non-null ptr, then it points at a - // valid vma. The vma is stable for as long as the vma read lock is held. - vma: unsafe { VmaRef::from_raw(vma) }, - _nts: NotThreadSafe, - }); - } + // SAFETY: Calling `bindings::lock_vma_under_rcu` is always okay given an mm where + // `mm_users` is non-zero. + let vma = unsafe { bindings::lock_vma_under_rcu(self.as_raw(), vma_addr) }; + if vma.is_null() { + return None; } - - // Silence warnings about unused variables. - #[cfg(not(CONFIG_PER_VMA_LOCK))] - let _ = vma_addr; - - None + Some(VmaReadGuard { + // SAFETY: If `lock_vma_under_rcu` returns a non-null ptr, then it points at a + // valid vma. The vma is stable for as long as the vma read lock is held. + vma: unsafe { VmaRef::from_raw(vma) }, + _nts: NotThreadSafe, + }) } /// Lock the mmap read lock. diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 4c58487b764e9d..57046d8ac81d80 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -560,7 +560,6 @@ struct vm_area_struct { vma_flags_t flags; }; -#ifdef CONFIG_PER_VMA_LOCK /* * Can only be written (using WRITE_ONCE()) while holding both: * - mmap_lock (in write mode) @@ -576,7 +575,7 @@ struct vm_area_struct { * slowpath. */ unsigned int vm_lock_seq; -#endif + unsigned int __vm_anon_pgoff_lo; /* @@ -610,10 +609,8 @@ struct vm_area_struct { #ifdef CONFIG_NUMA_BALANCING struct vma_numab_state *numab_state; /* NUMA Balancing state */ #endif -#ifdef CONFIG_PER_VMA_LOCK /* Unstable RCU readers are allowed to read this. */ refcount_t vm_refcnt; -#endif #ifdef CONFIG_64BIT unsigned int __vm_anon_pgoff_hi; #endif diff --git a/tools/testing/vma/vma_internal.h b/tools/testing/vma/vma_internal.h index 8a48b231aa7abf..54d5c3360aa26b 100644 --- a/tools/testing/vma/vma_internal.h +++ b/tools/testing/vma/vma_internal.h @@ -15,7 +15,6 @@ #include #define CONFIG_MMU 1 -#define CONFIG_PER_VMA_LOCK 1 #ifdef __CONCAT #undef __CONCAT From 078a358bedeb9aa8ba1400f7068e51cb0c33df4f Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:53 -0700 Subject: [PATCH 0316/1012] binder: make shrinker rely solely on per-VMA lock MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit tl;dr: lock_vma_under_rcu() is already a trylock. No need to do both it and mmap_read_trylock(). Long Version: == Background == Historically, binder used an mmap_read_trylock() in its shrinker code. This ensures that reclaim is not blocked on an mmap_lock. Commit 95bc2d4a9020 ("binder: use per-vma lock in page reclaiming") added support for the per-VMA lock, but left mmap_read_trylock() as a fallback. This was presumably because the per-VMA locking can fail for several reasons and most (all?) lock_vma_under_rcu() callers have a fallback to mmap_read_trylock(). == Problem == The fallback is not worth the complexity here. lock_vma_under_rcu() is essentially already a non-blocking trylock. The main reason it fails is also the reason mmap_read_trylock() fails: something is holding mmap_write_lock(). The only remedy for a collision with mmap_write_lock() is to wait, which this code can not do. So the "fallback" after lock_vma_under_rcu() failure is not really a fallback: it is really likely to just be retrying in vain. That retry in an of itself isn't horrible. But it adds complexity. == Solution == Now that per-VMA locks are universally available, lock_vma_under_rcu() will not persistently fail. Rely on it alone and simplify the code. The removal of the fallback does not affect NOMMU case because binder driver depends on CONFIG_MMU. While at it we also make the handling of the cases where the original binder VMA is gone consistent. There are two cases to consider when Binder VMA is gone: 1. there is no VMA at that location anymore. 2. there is now another unrelated VMA at that location. Before this change we handle case 1 by having the shrinker proceed to free the page, and just skip the zap_vma_range() call. And we handle case 2 by having the shrinker return LRU_SKIP. While either behavior is acceptable, we need to handle them in a consistent way. Handle both cases by freeing the page without touching the VMA (skipping the zap_vma_range()). Full disclosure: I originally tried to do this with lock_vma_under_rcu_wait(), but it did not fit well with the mmap_lock trylock semantics. Claude caught this in a review and suggested the approach in this path. It seemed sane to me. So, Suggesed-by: Claude, I guess. Link: https://lore.kernel.org/20260831203056.838265-3-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Alice Ryhl Acked-by: Lorenzo Stoakes (ARM) Acked-by: Carlos Llamas Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Todd Kjos Cc: Christian Brauner Cc: David S. Miller Cc: David Ahern Cc: Arve Hjønnevåg Cc: David Hildenbrand (Arm) --- drivers/android/binder_alloc.c | 46 ++++++++++++++++------------------ 1 file changed, 21 insertions(+), 25 deletions(-) diff --git a/drivers/android/binder_alloc.c b/drivers/android/binder_alloc.c index e4488ad86a6557..fcb744088e77c6 100644 --- a/drivers/android/binder_alloc.c +++ b/drivers/android/binder_alloc.c @@ -1142,7 +1142,6 @@ enum lru_status binder_alloc_free_page(struct list_head *item, struct vm_area_struct *vma; struct page *page_to_free; unsigned long page_addr; - int mm_locked = 0; size_t index; if (!mmget_not_zero(mm)) @@ -1151,27 +1150,25 @@ enum lru_status binder_alloc_free_page(struct list_head *item, index = mdata->page_index; page_addr = alloc->vm_start + index * PAGE_SIZE; - /* attempt per-vma lock first */ + /* + * Attempt per-vma lock. This is essentially a + * "trylock". It can fail even if the VMA exists + * for 'page_addr'. + */ vma = lock_vma_under_rcu(mm, page_addr); if (!vma) { - /* fall back to mmap_lock */ - if (!mmap_read_trylock(mm)) - goto err_mmap_read_lock_failed; - mm_locked = 1; - vma = vma_lookup(mm, page_addr); + /* + * If the vma exists, we can't continue because we cannot + * remove the page from the vma. However, if the vma was + * unmapped, it's okay to continue. + */ + if (binder_alloc_is_mapped(alloc)) + goto err_vma_lock_failed; } if (!mutex_trylock(&alloc->mutex)) goto err_get_alloc_mutex_failed; - /* - * Since a binder_alloc can only be mapped once, we ensure - * the vma corresponds to this mapping by checking whether - * the binder_alloc is still mapped. - */ - if (vma && !binder_alloc_is_mapped(alloc)) - goto err_invalid_vma; - trace_binder_unmap_kernel_start(alloc, index); page_to_free = alloc->pages[index]; @@ -1182,7 +1179,12 @@ enum lru_status binder_alloc_free_page(struct list_head *item, list_lru_isolate(lru, item); spin_unlock(&lru->lock); - if (vma) { + /* + * Since a binder_alloc can only be mapped once, we ensure + * the vma corresponds to this mapping by checking whether + * the binder_alloc is still mapped. + */ + if (vma && binder_alloc_is_mapped(alloc)) { trace_binder_unmap_user_start(alloc, index); zap_vma_range(vma, page_addr, PAGE_SIZE); @@ -1191,23 +1193,17 @@ enum lru_status binder_alloc_free_page(struct list_head *item, } mutex_unlock(&alloc->mutex); - if (mm_locked) - mmap_read_unlock(mm); - else + if (vma) vma_end_read(vma); mmput_async(mm); binder_free_page(page_to_free); return LRU_REMOVED_RETRY; -err_invalid_vma: - mutex_unlock(&alloc->mutex); err_get_alloc_mutex_failed: - if (mm_locked) - mmap_read_unlock(mm); - else + if (vma) vma_end_read(vma); -err_mmap_read_lock_failed: +err_vma_lock_failed: mmput_async(mm); err_mmget: return LRU_SKIP; From 4da94bd7a8d6ebfc0e1d79a7c02181d9cf69237e Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:54 -0700 Subject: [PATCH 0317/1012] mm: add RCU-based VMA lookup helper that waits for writers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit There are basically two parallel ways to look up a VMA: the traditional way, which is protected by mmap_read_lock, and the RCU-based per-VMA lock way which is based on RCU and refcounts. However, per-VMA locks will fail if the lock is help by a writer and therefore never waits. In a number of places we need to wait for the lock and it's done by falling back to mmap_read_lock, locking the VMA and releasing the mmap_lock once VMA is locked. Add vma_start_read_unlocked() - a variant of the RCU-based lookup that waits for writers. This is basically the same as the existing RCU-based lookup, but on a failure to lock it temporarily takes mmap_lock for read and waits for writers to finish before locking the VMA, dropping the mmap_read_lock and returning the locked VMA. This has some advantages: 1. Callers do not need to have a fallback path for when they collide with writers. 2. Its fast path does not require taking mmap_lock for read. Basically, when applied correctly, this approach results in faster *and* simpler code. While at it, fix the comments for vma_start_read_locked(), vma_start_read_locked_nested(), and uffd_lock_vma(). Link: https://lore.kernel.org/20260831203056.838265-4-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Suggested-by: Lorenzo Stoakes (ARM) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Vlastimil Babka (SUSE) Cc: Liam R. Howlett Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Todd Kjos Cc: Christian Brauner Cc: Carlos Llamas Cc: Alice Ryhl Cc: David S. Miller Cc: David Ahern Cc: Arve Hjønnevåg Cc: David Hildenbrand (Arm) --- include/linux/mmap_lock.h | 19 +++++++++++++++---- mm/mmap_lock.c | 35 +++++++++++++++++++++++++++++++++++ mm/userfaultfd.c | 6 ++++-- 3 files changed, 54 insertions(+), 6 deletions(-) diff --git a/include/linux/mmap_lock.h b/include/linux/mmap_lock.h index 6a0a8cf501bdea..28e3696ce9ff37 100644 --- a/include/linux/mmap_lock.h +++ b/include/linux/mmap_lock.h @@ -228,10 +228,14 @@ static inline void vma_refcount_put(struct vm_area_struct *vma) } /* - * Use only while holding mmap read lock which guarantees that locking will not - * fail (nobody can concurrently write-lock the vma). vma_start_read() should + * Use only while holding mmap read lock which guarantees that vma lock is not + * contended (nobody can concurrently write-lock the vma). vma_start_read() should * not be used in such cases because it might fail due to mm_lock_seq overflow. * This functionality is used to obtain vma read lock and drop the mmap read lock. + * + * VMA can't be detached while we are holding mmap lock, therefore in practice this + * function can fail only when there are so many readers that vm_refcnt overflows. + * The failure case is very unlikely and is already annotated as such internally. */ static inline bool vma_start_read_locked_nested(struct vm_area_struct *vma, int subclass) { @@ -247,16 +251,23 @@ static inline bool vma_start_read_locked_nested(struct vm_area_struct *vma, int } /* - * Use only while holding mmap read lock which guarantees that locking will not - * fail (nobody can concurrently write-lock the vma). vma_start_read() should + * Use only while holding mmap read lock which guarantees that vma lock is not + * contended (nobody can concurrently write-lock the vma). vma_start_read() should * not be used in such cases because it might fail due to mm_lock_seq overflow. * This functionality is used to obtain vma read lock and drop the mmap read lock. + * + * VMA can't be detached while we are holding mmap lock, therefore in practice this + * function can fail only when there are so many readers that vm_refcnt overflows. + * The failure case is very unlikely and is already annotated as such internally. */ static inline bool vma_start_read_locked(struct vm_area_struct *vma) { return vma_start_read_locked_nested(vma, 0); } +struct vm_area_struct *vma_start_read_unlocked(struct mm_struct *mm, + unsigned long address); + static inline void vma_end_read(struct vm_area_struct *vma) { vma_refcount_put(vma); diff --git a/mm/mmap_lock.c b/mm/mmap_lock.c index 272f9ac762b91e..2f94ee0fdee2a2 100644 --- a/mm/mmap_lock.c +++ b/mm/mmap_lock.c @@ -340,6 +340,41 @@ struct vm_area_struct *lock_vma_under_rcu(struct mm_struct *mm, return NULL; } +/** + * vma_start_read_unlocked() - Find the VMA covering 'address' and read-lock it. + * @mm: the mm_struct of the address space to search + * @address: address that the vma should contain + * + * The fast path does not take mmap_lock. Waits for writers to finish if the + * VMA is being modified by taking mmap_lock. + * Use when mmap_lock is not held, otherwise use vma_start_read_locked(). + * Nothing prevents VMAs being unmapped/mapped before or after the VMA is + * looked up, if a stronger guarantee is required, take an mmap_lock. + * + * Return: If a VMA exists which spans @address, return that VMA, read-locked. + * If no VMA is mapped there or, very unlikely, a reference count overflow + * occurred, return NULL. + */ +struct vm_area_struct *vma_start_read_unlocked(struct mm_struct *mm, + unsigned long address) +{ + struct vm_area_struct *vma; + + /* Fast path: return stable VMA covering 'address': */ + vma = lock_vma_under_rcu(mm, address); + if (vma) + return vma; + + /* Slow path: preclude VMA writers by temporarily getting mmap read lock. */ + mmap_read_lock(mm); + vma = vma_lookup(mm, address); + if (vma && !vma_start_read_locked(vma)) + vma = NULL; + mmap_read_unlock(mm); + + return vma; +} + static struct vm_area_struct *lock_next_vma_under_mmap_lock(struct mm_struct *mm, struct vma_iterator *vmi, unsigned long from_addr) diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index cf9c6ad3b3ad2c..bf50bff3838aa4 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -129,8 +129,10 @@ struct vm_area_struct *find_vma_and_prepare_anon(struct mm_struct *mm, * * Should be called without holding mmap_lock. * - * Return: A locked vma containing @address, -ENOENT if no vma is found, or - * -ENOMEM if anon_vma couldn't be allocated. + * Return: A locked vma containing @address, -ENOENT if no vma is found, + * -ENOMEM if anon_vma couldn't be allocated, or -EAGAIN if vma refcount + * overflow happened due to high number of readers and the caller should + * retry later. */ static struct vm_area_struct *uffd_lock_vma(struct mm_struct *mm, unsigned long address) From 23fa4d201860b9a30000d55ce99d866eea87ff5a Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:55 -0700 Subject: [PATCH 0318/1012] binder: remove mmap_lock fallback MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Previously, the per-VMA locking could fail in the face of writers which necessitate a fallback to mmap_lock. The new vma_start_read_unlocked() will wait for writers instead of failing. Use the new helper. Wait for writers. Remove the fallback to mmap_lock. Link: https://lore.kernel.org/20260831203056.838265-5-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Alice Ryhl Acked-by: Lorenzo Stoakes (ARM) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Todd Kjos Cc: Christian Brauner Cc: Carlos Llamas Cc: David S. Miller Cc: David Ahern Cc: Arve Hjønnevåg Cc: David Hildenbrand (Arm) --- drivers/android/binder/page_range.rs | 19 +++---------------- drivers/android/binder_alloc.c | 17 +++++------------ rust/kernel/mm.rs | 28 ++++++++++++++++++++++++++++ 3 files changed, 36 insertions(+), 28 deletions(-) diff --git a/drivers/android/binder/page_range.rs b/drivers/android/binder/page_range.rs index 52ffbf3504e7f7..71febd3d5b0730 100644 --- a/drivers/android/binder/page_range.rs +++ b/drivers/android/binder/page_range.rs @@ -439,22 +439,9 @@ impl ShrinkablePageRange { // workqueue. let mm = MmWithUser::into_mmput_async(self.mm.mmget_not_zero().ok_or(ESRCH)?); { - let vma_read; - let mmap_read; - let vma = if let Some(ret) = mm.lock_vma_under_rcu(vma_addr) { - vma_read = ret; - check_vma(&vma_read, self) - } else { - mmap_read = mm.mmap_read_lock(); - mmap_read - .vma_lookup(vma_addr) - .and_then(|vma| check_vma(vma, self)) - }; - - match vma { - Some(vma) => vma.vm_insert_page(user_page_addr, &new_page)?, - None => return Err(ESRCH), - } + let vma_read_guard = mm.vma_start_read_unlocked(vma_addr).ok_or(ESRCH)?; + let vma = check_vma(&vma_read_guard, self).ok_or(ESRCH)?; + vma.vm_insert_page(user_page_addr, &new_page)?; } let inner = self.lock.lock(); diff --git a/drivers/android/binder_alloc.c b/drivers/android/binder_alloc.c index fcb744088e77c6..d6eae0aa708541 100644 --- a/drivers/android/binder_alloc.c +++ b/drivers/android/binder_alloc.c @@ -259,21 +259,14 @@ static int binder_page_insert(struct binder_alloc *alloc, struct vm_area_struct *vma; int ret = -ESRCH; - /* attempt per-vma lock first */ - vma = lock_vma_under_rcu(mm, addr); - if (vma) { - if (binder_alloc_is_mapped(alloc)) - ret = vm_insert_page(vma, addr, page); - vma_end_read(vma); + vma = vma_start_read_unlocked(mm, addr); + if (!vma) return ret; - } - /* fall back to mmap_lock */ - mmap_read_lock(mm); - vma = vma_lookup(mm, addr); - if (vma && binder_alloc_is_mapped(alloc)) + if (binder_alloc_is_mapped(alloc)) ret = vm_insert_page(vma, addr, page); - mmap_read_unlock(mm); + + vma_end_read(vma); return ret; } diff --git a/rust/kernel/mm.rs b/rust/kernel/mm.rs index f4fa54616085f2..58bc1793fdaf5e 100644 --- a/rust/kernel/mm.rs +++ b/rust/kernel/mm.rs @@ -186,6 +186,34 @@ impl MmWithUser { }) } + /// Find the VMA covering 'address' and read-lock it. + /// + /// The fast path does not take mmap_lock. Waits for writers to finish if the + /// VMA is being modified by taking mmap_lock. + /// Use when mmap_lock is not held, otherwise use vma_start_read_locked(). + /// Nothing prevents VMAs being unmapped/mapped before or after the VMA is + /// looked up, if a stronger guarantee is required, take an mmap_lock. + /// + /// Return: If a VMA exists which spans @address, return that VMA, read-locked. + /// If no VMA is mapped there or, very unlikely, a reference count overflow + /// occurred, return NULL. + #[inline] + pub fn vma_start_read_unlocked(&self, vma_addr: usize) -> Option> { + // SAFETY: We may invoke `vma_start_read_unlocked` because we know this `mm` has non-zero + // `mm_users`. + let vma = unsafe { bindings::vma_start_read_unlocked(self.as_raw(), vma_addr) }; + if vma.is_null() { + return None; + } + // INVARIANT: We just acquired the VMA read lock. + Some(VmaReadGuard { + // SAFETY: If `vma_start_read_unlocked` returns a non-null ptr, then it points at a + // valid vma. The vma is stable for as long as the vma read lock is held. + vma: unsafe { VmaRef::from_raw(vma) }, + _nts: NotThreadSafe, + }) + } + /// Lock the mmap read lock. #[inline] pub fn mmap_read_lock(&self) -> MmapReadGuard<'_> { From 32dbf83a43f4e59a3afdfb8067a7e9cbc0f56658 Mon Sep 17 00:00:00 2001 From: Dave Hansen Date: Mon, 31 Aug 2026 13:30:56 -0700 Subject: [PATCH 0319/1012] tcp: remove mmap_lock fallback path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Previously, the per-VMA locking could fail in the face of writers which necessitates a fallback to mmap_lock. The new vma_start_read_unlocked() will wait for writers instead of failing. Use the new helper. Wait for writers. Remove the fallback to mmap_lock. The fallback removal does not affect NOMMU case because TCP_ZEROCOPY is gated on CONFIG_MMU. This really is a nice cleanup. It removes the need to pass the lock state back and forth to find_tcp_vma(). Link: https://lore.kernel.org/20260831203056.838265-6-surenb@google.com Signed-off-by: Dave Hansen Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes Acked-by: Vlastimil Babka (SUSE) Tested-by: syzbot@syzkaller.appspotmail.com Cc: Liam R. Howlett Cc: Shakeel Butt Cc: Greg Kroah-Hartman Cc: Arve Hjønnevåg Cc: Todd Kjos Cc: Christian Brauner Cc: Carlos Llamas Cc: Alice Ryhl Cc: David S. Miller Cc: David Ahern Cc: David Hildenbrand (Arm) --- net/ipv4/tcp.c | 31 +++++++++---------------------- 1 file changed, 9 insertions(+), 22 deletions(-) diff --git a/net/ipv4/tcp.c b/net/ipv4/tcp.c index 562752352afe4d..5588310bc64891 100644 --- a/net/ipv4/tcp.c +++ b/net/ipv4/tcp.c @@ -2167,27 +2167,18 @@ static void tcp_zc_finalize_rx_tstamp(struct sock *sk, } static struct vm_area_struct *find_tcp_vma(struct mm_struct *mm, - unsigned long address, - bool *mmap_locked) + unsigned long address) { - struct vm_area_struct *vma = lock_vma_under_rcu(mm, address); + struct vm_area_struct *vma = vma_start_read_unlocked(mm, address); - if (vma) { - if (vma->vm_ops != &tcp_vm_ops) { - vma_end_read(vma); - return NULL; - } - *mmap_locked = false; - return vma; - } + if (!vma) + return NULL; - mmap_read_lock(mm); - vma = vma_lookup(mm, address); - if (!vma || vma->vm_ops != &tcp_vm_ops) { - mmap_read_unlock(mm); + if (vma->vm_ops != &tcp_vm_ops) { + vma_end_read(vma); return NULL; } - *mmap_locked = true; + return vma; } @@ -2208,7 +2199,6 @@ static int tcp_zerocopy_receive(struct sock *sk, u32 seq = tp->copied_seq; u32 total_bytes_to_map; int inq = tcp_inq(sk); - bool mmap_locked; int ret; zc->copybuf_len = 0; @@ -2233,7 +2223,7 @@ static int tcp_zerocopy_receive(struct sock *sk, return 0; } - vma = find_tcp_vma(current->mm, address, &mmap_locked); + vma = find_tcp_vma(current->mm, address); if (!vma) return -EINVAL; @@ -2315,10 +2305,7 @@ static int tcp_zerocopy_receive(struct sock *sk, zc, total_bytes_to_map); } out: - if (mmap_locked) - mmap_read_unlock(current->mm); - else - vma_end_read(vma); + vma_end_read(vma); /* Try to copy straggler data. */ if (!ret) copylen = tcp_zc_handle_leftover(zc, sk, skb, &seq, copybuf_len, tss); From ea26fc1ca54cac28c48fac091a9280d25aaa6fa5 Mon Sep 17 00:00:00 2001 From: Joe Damato Date: Mon, 31 Aug 2026 10:48:35 -0700 Subject: [PATCH 0320/1012] mm: memcontrol: raise MEMCG_MAX for charges that fail without reclaiming Charges that exceed memory.max and return through the nomem label can raise no event and simply return -ENOMEM. A non-blocking charge can hit the limit, get rejected, but is not visible in memory.events. This was noticed in a production setting where bpf_mem_alloc() attempted to refill its per-cpu freelists, which triggered a non-blocking charge while at the limit. Commit d6e103a757fa ("mm: memcontrol: do not miss MEMCG_MAX events for enforced allocations") added raised_max_event to cover charges that are force charged without ever reaching reclaim, but charges that are rejected outright were left out. Getting an allocation failure without the corresponding MEMCG_MAX event is unexpected and makes debugging and monitoring harder. Raise the event on the way out for rejected charges as well, by routing the -ENOMEM return through the same exit path that already covers forced charges. The existing behavior of raising a MEMCG_MAX event on every charge/reclaim/retry iteration is left unchanged. Tested with a module that performs accounted GFP_NOWAIT page allocations from a task in a cgroup at its memory.max, and measures the resulting memory.events:max delta. Without this patch the rejected charges raise no event at all; with it the delta matches the number of rejected charges exactly. A GFP_KERNEL|__GFP_NORETRY control, which reaches reclaim, raises the same two events per failed charge before and after, confirming the existing charge/reclaim/retry accounting is unchanged. Link: https://lore.kernel.org/20260831174836.3102406-1-joe@dama.to Fixes: d6e103a757fa ("mm: memcontrol: do not miss MEMCG_MAX events for enforced allocations") Signed-off-by: Joe Damato Signed-off-by: Andrew Morton Suggested-by: Shakeel Butt Acked-by: Shakeel Butt Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: --- mm/memcontrol.c | 28 ++++++++++++++++------------ 1 file changed, 16 insertions(+), 12 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index d5ebe83eae3efc..aeaa09e01d70ea 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2685,10 +2685,11 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, bool raised_max_event = false; unsigned long pflags; bool allow_spinning = gfpflags_allow_spinning(gfp_mask); + int ret = 0; retry: if (consume_stock(memcg, nr_pages)) - return 0; + return ret; if (!allow_spinning) /* Avoid the refill and flush of the older stock */ @@ -2799,16 +2800,11 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, * put the burden of reclaim on regular allocation requests * and let these go through as privileged allocations. */ - if (!(gfp_mask & (__GFP_NOFAIL | __GFP_HIGH))) - return -ENOMEM; + if (!(gfp_mask & (__GFP_NOFAIL | __GFP_HIGH))) { + ret = -ENOMEM; + goto out; + } force: - /* - * If the allocation has to be enforced, don't forget to raise - * a MEMCG_MAX event. - */ - if (!raised_max_event) - __memcg_memory_event(mem_over_limit, MEMCG_MAX, allow_spinning); - /* * The allocation either can't fail or will lead to more memory * being freed very soon. Allow memory usage go over the limit @@ -2818,7 +2814,15 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, if (do_memsw_account()) page_counter_charge(&memcg->memsw, nr_pages); - return 0; +out: + /* + * Don't forget to raise a MEMCG_MAX event for forced or rejected + * requests. + */ + if (!raised_max_event) + __memcg_memory_event(mem_over_limit, MEMCG_MAX, allow_spinning); + + return ret; done_restock: if (batch > nr_pages) @@ -2877,7 +2881,7 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, !(current->flags & PF_MEMALLOC) && gfpflags_allow_blocking(gfp_mask)) __mem_cgroup_handle_over_high(gfp_mask); - return 0; + return ret; } static inline int try_charge(struct mem_cgroup *memcg, gfp_t gfp_mask, From 7848c017f1982be721e778e4f3baa5de4c7f0e89 Mon Sep 17 00:00:00 2001 From: Jason Angelov Date: Mon, 31 Aug 2026 08:06:48 -0700 Subject: [PATCH 0321/1012] mm/damon/core-kunit: test probe_hits handling at region split and merge Patch series "mm/damon: add kunit tests for probe_hits handling and probe params validation", v2. DAMON recently introduced probes and probe weights. Add kunit tests for the propagation of probe_hits at region split and merge, and the rejection of invalid probe parameters by damon_valid_probe_params(). This patch (of 2): damon_split_region_at() copies probe_hits[] and last_probe_hits[] to the new split region. damon_merge_two_regions() sets probe_hits[] to the size-weighted average of the merged regions. Extend damon_test_split_at() and damon_test_merge_two() tests to cover those fields. Link: https://lore.kernel.org/20260831150650.84829-1-sj@kernel.org Link: https://lore.kernel.org/20260831150650.84829-2-sj@kernel.org Signed-off-by: Jason Angelov Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Gow Cc: Brendan Higgins --- mm/damon/tests/core-kunit.h | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 4a536d41cdb2d0..2db94d49c9bae4 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -152,6 +152,8 @@ static void damon_test_split_at(struct kunit *test) } r->nr_accesses = 42; r->last_nr_accesses = 15; + r->probe_hits[0] = 7; + r->last_probe_hits[0] = 3; r->age = 10; damon_add_region(r, t); damon_split_region_at(t, r, 25); @@ -168,6 +170,8 @@ static void damon_test_split_at(struct kunit *test) KUNIT_EXPECT_EQ(test, r->nr_accesses, r_new->nr_accesses); KUNIT_EXPECT_EQ(test, r->last_nr_accesses, r_new->last_nr_accesses); + KUNIT_EXPECT_EQ(test, r->probe_hits[0], r_new->probe_hits[0]); + KUNIT_EXPECT_EQ(test, r->last_probe_hits[0], r_new->last_probe_hits[0]); KUNIT_EXPECT_EQ(test, r->age, r_new->age); out: @@ -189,6 +193,7 @@ static void damon_test_merge_two(struct kunit *test) kunit_skip(test, "region alloc fail"); } r->nr_accesses = 10; + r->probe_hits[0] = 6; r->age = 9; damon_add_region(r, t); r2 = damon_new_region(100, 300); @@ -197,6 +202,7 @@ static void damon_test_merge_two(struct kunit *test) kunit_skip(test, "second region alloc fail"); } r2->nr_accesses = 20; + r2->probe_hits[0] = 14; r2->age = 21; damon_add_region(r2, t); @@ -204,6 +210,7 @@ static void damon_test_merge_two(struct kunit *test) KUNIT_EXPECT_EQ(test, r->ar.start, 0ul); KUNIT_EXPECT_EQ(test, r->ar.end, 300ul); KUNIT_EXPECT_EQ(test, r->nr_accesses, 16u); + KUNIT_EXPECT_EQ(test, r->probe_hits[0], 11); KUNIT_EXPECT_EQ(test, r->age, 17u); i = 0; From 9bbf11bf9a057bc49ac2728d7f779c3a547f7c61 Mon Sep 17 00:00:00 2001 From: Jason Angelov Date: Mon, 31 Aug 2026 08:06:49 -0700 Subject: [PATCH 0322/1012] mm/damon/core-kunit: test damon_valid_probe_params() damon_valid_probe_params() makes damon_commit_ctx() reject probe configurations that could overflow a probe_hits counter, a single (weight * probe_hits) product, or the sum of those products. Add a kunit test covering each rejection at its boundary: - samples per aggregation interval: U8_MAX is allowed, one more could overflow a probe_hits counter - single weight: the largest whose product fits in unsigned int is allowed, one larger is rejected - multiple probes: each product fits, but their sum overflows - no weight set: the validation is skipped Link: https://lore.kernel.org/20260831150650.84829-3-sj@kernel.org Signed-off-by: Jason Angelov Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Brendan Higgins Cc: David Gow --- mm/damon/tests/core-kunit.h | 57 +++++++++++++++++++++++++++++++++++++ 1 file changed, 57 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 2db94d49c9bae4..b1ca4c8e03f091 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1338,6 +1338,62 @@ static void damon_test_commit_ctx(struct kunit *test) damon_destroy_ctx(dst); } +static void damon_test_valid_probe_params(struct kunit *test) +{ + struct damon_ctx *ctx; + struct damon_probe *probe, *probe2; + + ctx = damon_new_ctx(); + if (!ctx) + kunit_skip(test, "ctx alloc fail"); + probe = damon_new_probe(); + if (!probe) { + damon_destroy_ctx(ctx); + kunit_skip(test, "probe alloc fail"); + } + damon_add_probe(ctx, probe); + + /* Parameters are validated only if any probe weight is set. */ + ctx->attrs.sample_interval = 1; + ctx->attrs.aggr_interval = 1000000; + KUNIT_EXPECT_TRUE(test, damon_valid_probe_params(ctx)); + + /* Up to U8_MAX samples per aggregation interval are allowed. */ + probe->weight = 100; + ctx->attrs.aggr_interval = 255; + KUNIT_EXPECT_TRUE(test, damon_valid_probe_params(ctx)); + + /* More samples could overflow the probe_hits counters. */ + ctx->attrs.aggr_interval = 256; + KUNIT_EXPECT_FALSE(test, damon_valid_probe_params(ctx)); + + /* The largest weight whose weighted hit count fits in unsigned int. */ + ctx->attrs.aggr_interval = 255; + probe->weight = UINT_MAX / 255; + KUNIT_EXPECT_TRUE(test, damon_valid_probe_params(ctx)); + + /* Any larger weight could overflow its weighted hit count. */ + probe->weight = UINT_MAX / 255 + 1; + KUNIT_EXPECT_FALSE(test, damon_valid_probe_params(ctx)); + + /* With one sample per aggregation, even the largest weight fits. */ + ctx->attrs.aggr_interval = 1; + probe->weight = UINT_MAX; + KUNIT_EXPECT_TRUE(test, damon_valid_probe_params(ctx)); + + /* The sum of all probes' weighted hit counts could also overflow. */ + probe2 = damon_new_probe(); + if (!probe2) { + damon_destroy_ctx(ctx); + kunit_skip(test, "probe2 alloc fail"); + } + probe2->weight = 1; + damon_add_probe(ctx, probe2); + KUNIT_EXPECT_FALSE(test, damon_valid_probe_params(ctx)); + + damon_destroy_ctx(ctx); +} + static void damos_test_filter_out(struct kunit *test) { struct damon_target *t; @@ -1664,6 +1720,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damos_test_commit_migrate_hot), KUNIT_CASE(damon_test_commit_target_regions), KUNIT_CASE(damon_test_commit_ctx), + KUNIT_CASE(damon_test_valid_probe_params), KUNIT_CASE(damos_test_filter_out), KUNIT_CASE(damon_test_feed_loop_next_input), KUNIT_CASE(damon_test_set_filters_default_reject), From 35c108feb61c3718dfefe8c974011197589cc97b Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 31 Aug 2026 08:02:24 -0700 Subject: [PATCH 0323/1012] docs/mm/damon/design: accurate semantics of nr_snapshots Patch series "docs/mm/damon/design: add explanation of nr_snapshots", v3. Add an explanation of nr_snapshots to avoid misunderstandings. This patch (of 3): Change "tried to be applied" -> "completely tried to be applied" to maintain consistency between the documentation and the code. Link: https://lore.kernel.org/20260831150227.83416-1-sj@kernel.org Link: https://lore.kernel.org/20260831150227.83416-2-sj@kernel.org Signed-off-by: Liew Rui Yan Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 4 ++-- include/linux/damon.h | 3 ++- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index aed6cb1cf48310..1739aeec6eb95f 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -846,8 +846,8 @@ scheme's execution. - ``nr_applied``: Total number of regions that the scheme is applied. - ``sz_applied``: Total size of regions that the scheme is applied. - ``qt_exceeds``: Total number of times the quota of the scheme has exceeded. -- ``nr_snapshots``: Total number of DAMON snapshots that the scheme is tried to - be applied. +- ``nr_snapshots``: Total number of DAMON snapshots that the scheme is + completely tried to be applied. - ``max_nr_snapshots``: Upper limit of ``nr_snapshots``. "A scheme is tried to be applied to a region" means DAMOS core logic determined diff --git a/include/linux/damon.h b/include/linux/damon.h index 0c8b7ddef9abb3..cbdf5f77978e72 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -357,7 +357,8 @@ struct damos_watermarks { * Total bytes that passed ops layer-handled DAMOS filters. * @qt_exceeds: Total number of times the quota of the scheme has exceeded. * @nr_snapshots: - * Total number of DAMON snapshots that the scheme has tried. + * Total number of DAMON snapshots that the scheme is completely + * tried to be applied. * * "Tried an action to a region" in this context means the DAMOS core logic * determined the region as eligible to apply the action. The access pattern From 773b17d11d99f9cd6291ca70c2dd7d00d9029887 Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 31 Aug 2026 08:02:25 -0700 Subject: [PATCH 0324/1012] docs/mm/damon/design: difference between watermarks and nr_snapshots Explain the difference between nr_snapshots reaches max_nr_snapshots and watermarks. Link: https://lore.kernel.org/20260831150227.83416-3-sj@kernel.org Signed-off-by: Liew Rui Yan Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 1739aeec6eb95f..e7977f005ac06e 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -872,7 +872,8 @@ the action to the region will fail. Unlike normal stats, ``max_nr_snapshots`` is set by users. If it is set as non-zero and ``nr_snapshots`` be same to or greater than ``nr_snapshots``, the -scheme is deactivated. +scheme is deactivated. Note that, unlike watermarks, even if a scheme's +``nr_snapshots`` reaches ``max_nr_snapshots``, monitoring will not stop. To know how user-space can read the stats via :ref:`DAMON sysfs interface `, refer to :ref:s`stats ` part of the From b302dbb1e78b95e4919775e5e399b8b06859e301 Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 31 Aug 2026 08:02:26 -0700 Subject: [PATCH 0325/1012] docs/mm/damon/design: fix typo of max_nr_snapshots Fix a typo (nr_snapshots -> max_nr_snapshots) and corrects a grammar error. Link: https://lore.kernel.org/20260831150227.83416-4-sj@kernel.org Signed-off-by: Liew Rui Yan Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 4 ++-- include/linux/damon.h | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index e7977f005ac06e..d1dd9050ebf40b 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -871,8 +871,8 @@ action is ``pageout`` while all pages of the region are unreclaimable, applying the action to the region will fail. Unlike normal stats, ``max_nr_snapshots`` is set by users. If it is set as -non-zero and ``nr_snapshots`` be same to or greater than ``nr_snapshots``, the -scheme is deactivated. Note that, unlike watermarks, even if a scheme's +non-zero and ``nr_snapshots`` equals or is greater than ``max_nr_snapshots``, +the scheme is deactivated. Note that, unlike watermarks, even if a scheme's ``nr_snapshots`` reaches ``max_nr_snapshots``, monitoring will not stop. To know how user-space can read the stats via :ref:`DAMON sysfs interface diff --git a/include/linux/damon.h b/include/linux/damon.h index cbdf5f77978e72..4b0d2d2e4ea4eb 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -549,7 +549,7 @@ struct damos_migrate_dests { * * After applying the &action to each region, &stat is updated. * - * If &max_nr_snapshots is set as non-zero and &stat.nr_snapshots be same to or + * If &max_nr_snapshots is set as non-zero and &stat.nr_snapshots equals or is * greater than it, the scheme is deactivated. */ struct damos { From 34bd9d753db0a99a9f827f226ca6906c23282639 Mon Sep 17 00:00:00 2001 From: Cheng-Han Wu Date: Mon, 31 Aug 2026 07:57:21 -0700 Subject: [PATCH 0326/1012] mm/damon/core: remove unused damon_targets_empty() Patch series "mm/damon/core: remove unused helper functions", v2. Both damon_targets_empty() and damon_nr_running_ctxs() have had no in-tree users since commit 5ec4333b1967 ("mm/damon: remove DAMON debugfs interface") removed their remaining callers. Remove the unused declarations and definitions. This patch (of 2): damon_targets_empty() has had no in-tree users since commit 5ec4333b1967 ("mm/damon: remove DAMON debugfs interface") removed its last caller. Remove the unused declaration and definition. Link: https://lore.kernel.org/20260831145724.82387-1-sj@kernel.org Link: https://lore.kernel.org/20260831145724.82387-2-sj@kernel.org Signed-off-by: Cheng-Han Wu Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park --- include/linux/damon.h | 1 - mm/damon/core.c | 5 ----- 2 files changed, 6 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 4b0d2d2e4ea4eb..89a41dea1d23f4 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -1055,7 +1055,6 @@ int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src); struct damon_target *damon_new_target(void); void damon_add_target(struct damon_ctx *ctx, struct damon_target *t); -bool damon_targets_empty(struct damon_ctx *ctx); void damon_free_target(struct damon_target *t); void damon_destroy_target(struct damon_target *t, struct damon_ctx *ctx); unsigned int damon_nr_regions(struct damon_target *t); diff --git a/mm/damon/core.c b/mm/damon/core.c index ab3c4d75496449..dc772ceae26ffc 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -795,11 +795,6 @@ void damon_add_target(struct damon_ctx *ctx, struct damon_target *t) list_add_tail(&t->list, &ctx->adaptive_targets); } -bool damon_targets_empty(struct damon_ctx *ctx) -{ - return list_empty(&ctx->adaptive_targets); -} - static void damon_del_target(struct damon_target *t) { list_del(&t->list); From 32af2923fcdfd9dc881cfa9e48f41e032fecd818 Mon Sep 17 00:00:00 2001 From: Cheng-Han Wu Date: Mon, 31 Aug 2026 07:57:22 -0700 Subject: [PATCH 0327/1012] mm/damon/core: remove unused damon_nr_running_ctxs() damon_nr_running_ctxs() has had no in-tree users since commit 5ec4333b1967 ("mm/damon: remove DAMON debugfs interface") removed all of its callers. Remove the unused declaration and definition. Link: https://lore.kernel.org/20260831145724.82387-3-sj@kernel.org Signed-off-by: Cheng-Han Wu Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park --- include/linux/damon.h | 1 - mm/damon/core.c | 14 -------------- 2 files changed, 15 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 89a41dea1d23f4..6993dca0f355cc 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -1065,7 +1065,6 @@ int damon_set_attrs(struct damon_ctx *ctx, struct damon_attrs *attrs); void damon_set_schemes(struct damon_ctx *ctx, struct damos **schemes, ssize_t nr_schemes); int damon_commit_ctx(struct damon_ctx *old_ctx, struct damon_ctx *new_ctx); -int damon_nr_running_ctxs(void); bool damon_is_registered_ops(enum damon_ops_id id); int damon_register_ops(struct damon_operations *ops); int damon_select_ops(struct damon_ctx *ctx, enum damon_ops_id id); diff --git a/mm/damon/core.c b/mm/damon/core.c index dc772ceae26ffc..d832a527bcf623 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1847,20 +1847,6 @@ int damon_commit_ctx(struct damon_ctx *dst, struct damon_ctx *src) return err; } -/** - * damon_nr_running_ctxs() - Return number of currently running contexts. - */ -int damon_nr_running_ctxs(void) -{ - int nr_ctxs; - - mutex_lock(&damon_lock); - nr_ctxs = nr_running_ctxs; - mutex_unlock(&damon_lock); - - return nr_ctxs; -} - /* Returns the size upper limit for each monitoring region */ static unsigned long damon_region_sz_limit(struct damon_ctx *ctx) { From cb0c40ad748e8c3b98a774a57b78ad06eddf4a98 Mon Sep 17 00:00:00 2001 From: Asier Gutierrez Date: Mon, 31 Aug 2026 07:47:28 -0700 Subject: [PATCH 0328/1012] mm/damon: introduce DAMOS_QUOTA_HUGEPAGE auto tuning Patch series "mm/damon: Introduce a huge page collapsing mechanism using auto tuning", v4. Overview ======== This patchset introduces a new autotuning which allows to collapse hot regions into hugepages. Motivation ========== Since TLB is a bottleneck for many systems[1], a way to optimize TLB misses (or hits) is to use huge pages. Unfortunately, using "always" in THP leads to memory fragmentation and memory waste. For this reason, most application guides and system administrators suggest to disable THP. Selective huge page collapse per process is possible using prctl and a launcher. However, this does not solve the issue with hot region detection. Additionally, it the sysadmin should create a launcher that uses PRCTL to enable THP for a particular process. We can use the DAMON support for DAMOS_HUGEPAGE and DAMOS_COLLAPSE, to target a certain process. DAMOS_COLLAPSE can also target the hot regions in that process. Still, there is an issue with the amount of huge page consumption. Since huge pages can lead to memory fragmentation and waste, there should be a way to limit the amount of huge page consumption. There is hugetlbfs, but it requires changes to the application code or the use of libhugetlbfs. DAMON has now a way to autotune some of the variables and adjust quotas automatically, so that DAMON is fired only under the right circumstances. It would be nice to have something similar, but for huge pages. Solution ======== A new autotuning quota goal[2], damos_hugepage_mem_bp, is introduced, which checks the huge page consumption to total memory consumption. This new quota mechanism reuses current autotuning architecture. In order to test this new mechanism, a sample module[3] was created, but not included in this patch series. To demonstrate the tool, damo user space tool was modified[4], which sets up huge pages collapse autotuning. Benchmarks ========== Setup: physical server with arm64 processor with 4 NUMA nodes, 1 TB RAM and running mariaDB 10.5.29. Sysbench was used for the benchmark, with 20 tables and 3 million rows per table. The database was pinned to one of the nodes, and the benchmark framework to a different node. No network traffic involved in the benchmark. Damo user space tool was forked and hugepage_mem_bp support added[4]. DAMON was lauched using this command line: sudo ./damo start $(pidof mariadbd) \ --monitoring_nr_regions_range 10 1000 \ --monitoring_intervals 5000 100000 60000000 \ --damos_quota_time 0 --damos_quota_space 128000000 \ --damos_quota_interval 1000 \ --damos_quota_weights 0 1 1 \ --damos_quota_goal hugepage_mem_bp \ --damos_quota_goal_tuner temporal \ --damos_apply_interval 50000 \ --damos_access_rate 0 max --damos_age 50 max \ --damos_action collapse --debug_damon was 1000 to taget 10% hugepage to total memory ratio, or 2500 to target 25%. Tuner was also tested with consistent and temporal. Results ======= After the last timestamp, there was no change in huge page use, and the total huge page to memory consumption ratio barely moved. hugepage_mem_bp: 1000 goal tuner: temporal +-----------+----------------+----------------+----------------------+ | timestamp | total mem used | huge page used | percentage hugepage | +-----------+----------------+----------------+----------------------+ | 0 | 16945.04297 | 0 | 0 | | 7 | 17008.69531 | 74 | 0.435071583 | | 8 | 17036.40234 | 194 | 1.138738074 | | 9 | 17017.01563 | 314 | 1.845211916 | | 10 | 17029.67969 | 434 | 2.548491856 | | 61 | 17111.30859 | 584 | 3.412947623 | | 120 | 17071.05859 | 694 | 4.065360072 | | 180 | 17133.88281 | 804 | 4.692456513 | | 203 | 17088.16406 | 916 | 5.360435426 | | 204 | 17126.34766 | 1046 | 6.107548562 | | 205 | 17093.84375 | 1176 | 6.879669764 | | 206 | 17142.77734 | 1298 | 7.571701913 | | 209 | 17149.17969 | 1686 | 9.831374041 | | 210 | 17097.30859 | 1754 | 10.25892462 | +-----------+----------------+----------------+----------------------+ hugepage_mem_bp: 1000 goal tuner: consistent +-----------+----------------+----------------+----------------------+ | timestamp | total mem used | huge page used | percentage hugepage | +-----------+----------------+----------------+----------------------+ | 0 | 16955.24609 | 0 | 0 | | 34 | 17039.71875 | 106 | 0.622075995 | | 78 | 17009.47656 | 554 | 3.257007927 | | 90 | 17048.92188 | 596 | 3.495822225 | | 150 | 17092.90625 | 706 | 4.130368409 | | 180 | 17053.08984 | 764 | 4.480126517 | | 233 | 17100.50391 | 1496 | 8.748280216 | | 239 | 17098.89063 | 2216 | 12.95990511 | | 240 | 17135.44531 | 2334 | 13.62088908 | | 245 | 17132.55078 | 2932 | 17.11362212 | | 246 | 17117.95313 | 3052 | 17.82923448 | | 250 | 17163.12109 | 3532 | 20.57900763 | +-----------+----------------+----------------+----------------------+ hugepage_mem_bp: 2500 goal tuner: temporal +-----------+----------------+----------------+----------------------+ | timestamp | total mem used | huge page used | percentage hugepage | +-----------+----------------+----------------+----------------------+ | 0 | 17010.31641 | 0 | 0 | | 9 | 17063.6875 | 50 | 0.2930199 | | 10 | 17051.75781 | 170 | 0.996964664 | | 60 | 17133.85547 | 572 | 3.338419663 | | 90 | 17192.07813 | 626 | 3.641211932 | | 120 | 17221.44531 | 682 | 3.960178647 | | 181 | 17199.76172 | 790 | 4.593086886 | | 208 | 17222.77734 | 1206 | 7.002354939 | | 214 | 17245.17969 | 1904 | 11.04076637 | | 215 | 17240.45703 | 2024 | 11.73982799 | | 220 | 17234.79688 | 2624 | 15.22501262 | | 228 | 17222.83594 | 3584 | 20.80958103 | | 231 | 17247.55469 | 3944 | 22.86700968 | | 235 | 17229.37109 | 4424 | 25.67708349 | +-----------+----------------+----------------+----------------------+ hugepage_mem_bp: 1000 goal tuner: consist +-----------+----------------+----------------+----------------------+ | timestamp | total mem used | huge page used | percentage hugepage | +-----------+----------------+----------------+----------------------+ | 0 | 17125.85156 | 0 | 0 | | 38 | 17081.23438 | 76 | 0.444932716 | | 39 | 17133.11719 | 196 | 1.143983304 | | 40 | 17119.83984 | 316 | 1.84581166 | | 60 | 17109.72656 | 554 | 3.237924335 | | 90 | 17164.11328 | 628 | 3.65879664 | | 180 | 17177.66016 | 792 | 4.610639591 | | 220 | 17180.86719 | 1378 | 8.020549749 | | 226 | 17187.82031 | 1980 | 11.51978531 | | 233 | 17143.48438 | 2818 | 16.4377319 | | 240 | 17137.38281 | 3656 | 21.33347921 | | 250 | 17175.5 | 4856 | 28.27283049 | | 260 | 17199.66406 | 6056 | 35.20999002 | | 270 | 17203.98438 | 7254 | 42.16465118 | | 275 | 17207.21875 | 7762 | 45.10897498 | +-----------+----------------+----------------+----------------------+ More detailed tables are provided here[5] From this, we can conclude that the huge page autotuner works fine, achieving the target. When using consistent autotuner, it actually over-achieves the target, which is expected, since quota esz_bp is not set to 0 to cap the DAMOS policy. Patches Sequence ================ Patch 1 -> Introduce DAMOS_QUOTA_HUGEPAGE_MEM_BP and autotuning Patch 2 -> sysfs support for the new quota goal Patch 3 -> Document hugepage_mem_bp parameter This patch (of 3): Introduce DAMOS_QUOTA_HUGEPAGE_MEM_BP auto tuning. Add a new DAMOS quota goal metric to measure the amount of huge page consumption to total memory consumption ratio. Vmstat may lag, which in some cases may lead to NR_FREE_PAGES being greater than or equal to the amount of RAM in the system. A guard is added to avoid the extremely unlikely case [6]. In the case, return 100% (10000 bp). Link: https://lore.kernel.org/20260831144732.80910-1-sj@kernel.org Link: https://lore.kernel.org/20260831144732.80910-2-sj@kernel.org Link: https://dl.acm.org/doi/pdf/10.1145/3307650.3322227 [1] Link: https://lore.kernel.org/e67f05ad-dbb9-45e6-ba30-b167a99ac67d@huawei-partners.com [2] Link: https://lore.kernel.org/20260616150316.580819-3-gutierrez.asier@huawei-partners.com [3] Link: https://github.com/asierHuawei/damo/commit/79ae1a4ab1c012a7161db85a000d14f08fa36736 [4] Link: https://lore.kernel.org/all/03f678dd-9ef3-4b97-b753-c2e4554c5159@huawei-partners.com/ [5] Link: https://lore.kernel.org/all/20260715151615.99767-1-sj@kernel.org/ [6] Signed-off-by: Asier Gutierrez Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 2 ++ mm/damon/core.c | 19 +++++++++++++++++++ 2 files changed, 21 insertions(+) diff --git a/include/linux/damon.h b/include/linux/damon.h index 6993dca0f355cc..955b9f614e5bcd 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -153,6 +153,7 @@ enum damos_action { * @DAMOS_QUOTA_INACTIVE_MEM_BP: Inactive to total LRU memory ratio. * @DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP: Scheme-eligible memory ratio of a * node in basis points (0-10000). + * @DAMOS_QUOTA_HUGEPAGE_MEM_BP: Huge page to total used memory ratio. * @NR_DAMOS_QUOTA_GOAL_METRICS: Number of DAMOS quota goal metrics. * * Metrics equal to larger than @NR_DAMOS_QUOTA_GOAL_METRICS are unsupported. @@ -167,6 +168,7 @@ enum damos_quota_goal_metric { DAMOS_QUOTA_ACTIVE_MEM_BP, DAMOS_QUOTA_INACTIVE_MEM_BP, DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP, + DAMOS_QUOTA_HUGEPAGE_MEM_BP, NR_DAMOS_QUOTA_GOAL_METRICS, }; diff --git a/mm/damon/core.c b/mm/damon/core.c index d832a527bcf623..153425e416f55e 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2991,6 +2991,22 @@ static unsigned int damos_get_in_active_mem_bp(bool active_ratio) return mult_frac(inactive, 10000, total); } +static unsigned int damos_hugepage_mem_bp(void) +{ + unsigned long thp, total_pages, free_pages; + + total_pages = totalram_pages(); + free_pages = global_zone_page_state(NR_FREE_PAGES); + + if (total_pages <= free_pages) + return 10000; + + thp = global_node_page_state(NR_ANON_THPS) + + global_node_page_state(NR_SHMEM_THPS) + + global_node_page_state(NR_FILE_THPS); + return mult_frac(thp, 10000, total_pages - free_pages); +} + static void damos_set_quota_goal_current_value(struct damon_ctx *c, struct damos *s, struct damos_quota_goal *goal) { @@ -3022,6 +3038,9 @@ static void damos_set_quota_goal_current_value(struct damon_ctx *c, goal->current_value = damos_get_node_eligible_mem_bp(c, s, goal->nid); break; + case DAMOS_QUOTA_HUGEPAGE_MEM_BP: + goal->current_value = damos_hugepage_mem_bp(); + break; default: break; } From a0907f097a80f0f35a378fcf6f50be832189d151 Mon Sep 17 00:00:00 2001 From: Asier Gutierrez Date: Mon, 31 Aug 2026 07:47:29 -0700 Subject: [PATCH 0329/1012] mm/damon/sysfs: support hugepage_mem_bp quota goal metric DAMOS has a new autotune policy metric: DAMOS_QUOTA_HUGEPAGE_MEM_BP. This patch exposes DAMOS_QUOTA_HUGEPAGE_MEM_BP through sysfs. Add the "hugepage_mem_bp" to the sysfs-schemes interface. Link: https://lore.kernel.org/20260831144732.80910-3-sj@kernel.org Signed-off-by: Asier Gutierrez Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs-schemes.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index 32f495a96b17a8..d9b81d7b5910ed 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -1269,6 +1269,10 @@ struct damos_sysfs_qgoal_metric_name damos_sysfs_qgoal_metric_names[] = { .metric = DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP, .name = "node_eligible_mem_bp", }, + { + .metric = DAMOS_QUOTA_HUGEPAGE_MEM_BP, + .name = "hugepage_mem_bp", + }, }; static ssize_t target_metric_show(struct kobject *kobj, From ecea6accd049e020e437e321c435c6e9ba14c760 Mon Sep 17 00:00:00 2001 From: Asier Gutierrez Date: Mon, 31 Aug 2026 07:47:30 -0700 Subject: [PATCH 0330/1012] Docs/mm/damon/design: cocument hugepage_mem_bp target metric Document hugepage_mem_bp metric exposed by sysfs. Link: https://lore.kernel.org/20260831144732.80910-4-sj@kernel.org Signed-off-by: Asier Gutierrez Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 2 ++ 1 file changed, 2 insertions(+) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index d1dd9050ebf40b..63cbb7b536da20 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -713,6 +713,8 @@ mechanism tries to make ``current_value`` of ``target_metric`` be same to bp (1/10,000). - ``node_eligible_mem_bp``: Scheme target access pattern-eligible memory ratio of a node in bp (1/10,000). +- ``hugepage_mem_bp``: Total huge page to total used memory ratio in bp + (1/10,000). ``nid`` is optionally required for ``node_mem_used_bp``, ``node_mem_free_bp``, ``node_memcg_used_bp``, ``node_memcg_free_bp`` and ``node_eligible_mem_bp`` to From 07df3d668bc53be74015ed593a1fcd9ec388168a Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Mon, 31 Aug 2026 07:26:03 -0700 Subject: [PATCH 0331/1012] mm/damon/core: remove declaration of __damon_commit_ctx() Patch series "mm/damon: misc cleanups". Cleanup the code, tests and samples for clarifications and readability. The patches are individually sent by the authors. I'm reposting those as one series for convenience of handling. For this reason, changelog is on each patch's commentary section. This patch (of 7): __damon_commit_ctx() was added by commit b1471afe4d10 ("mm/damon/core: do parameter testing commit on damon_start()") but is actually not needed. Remove it. Link: https://lore.kernel.org/20260831142611.77572-1-sj@kernel.org Link: https://lore.kernel.org/20260831142611.77572-2-sj@kernel.org Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Greg Kroah-Hartman Cc: Shuah Khan Cc: Enze Li Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Li Youhong Cc: zhaozhengzhuo --- mm/damon/core.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 153425e416f55e..fb52d99661bb57 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1935,8 +1935,6 @@ static int __damon_start(struct damon_ctx *ctx) return err; } -static int __damon_commit_ctx(struct damon_ctx *dst, struct damon_ctx *src); - /** * damon_start() - Starts the monitorings for a given group of contexts. * @ctxs: an array of the pointers for contexts to start monitoring From 305a355783c455c5dd1895bab8a71694029e14c3 Mon Sep 17 00:00:00 2001 From: Enze Li Date: Mon, 31 Aug 2026 07:26:04 -0700 Subject: [PATCH 0332/1012] mm/damon/core: introduce damon_set_target_pid() The logic that finds the struct pid for a given pid number and assigns it to a damon_target is duplicated in multiple places. Including damon_sysfs_add_target() of mm/damon/sysfs.c and the start functions of the two sample modules, samples/damon/wsse.c and samples/damon/prcl.c. Add a function that does the work, and replace the duplicated code in the places with calls to the function. Link: https://lore.kernel.org/20260831142611.77572-3-sj@kernel.org Signed-off-by: Enze Li Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo --- include/linux/damon.h | 1 + mm/damon/core.c | 12 ++++++++++++ mm/damon/sysfs.c | 6 ++---- samples/damon/prcl.c | 5 +---- samples/damon/wsse.c | 5 +---- 5 files changed, 17 insertions(+), 12 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 955b9f614e5bcd..7b1b6050a8286f 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -1057,6 +1057,7 @@ int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src); struct damon_target *damon_new_target(void); void damon_add_target(struct damon_ctx *ctx, struct damon_target *t); +int damon_set_target_pid(struct damon_target *t, int pid); void damon_free_target(struct damon_target *t); void damon_destroy_target(struct damon_target *t, struct damon_ctx *ctx); unsigned int damon_nr_regions(struct damon_target *t); diff --git a/mm/damon/core.c b/mm/damon/core.c index fb52d99661bb57..d63d4c6fd3ffdd 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -10,6 +10,7 @@ #include #include #include +#include #include #include #include @@ -795,6 +796,17 @@ void damon_add_target(struct damon_ctx *ctx, struct damon_target *t) list_add_tail(&t->list, &ctx->adaptive_targets); } +/* + * Assign the struct pid of the given pid number to the given target. + */ +int damon_set_target_pid(struct damon_target *t, int pid) +{ + t->pid = find_get_pid(pid); + if (!t->pid) + return -EINVAL; + return 0; +} + static void damon_del_target(struct damon_target *t) { list_del(&t->list); diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index e3858ffab4b227..3c81b4c91ac0dd 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -3,7 +3,6 @@ * DAMON sysfs Interface */ -#include #include #include @@ -2035,9 +2034,8 @@ static int damon_sysfs_add_target(struct damon_sysfs_target *sys_target, return -ENOMEM; damon_add_target(ctx, t); if (damon_target_has_pid(ctx)) { - t->pid = find_get_pid(sys_target->pid); - if (!t->pid) - /* caller will destroy targets */ + /* caller will destroy targets */ + if (damon_set_target_pid(t, sys_target->pid)) return -EINVAL; } t->obsolete = sys_target->obsolete; diff --git a/samples/damon/prcl.c b/samples/damon/prcl.c index 842099bd622861..83ddf12811d57f 100644 --- a/samples/damon/prcl.c +++ b/samples/damon/prcl.c @@ -32,7 +32,6 @@ module_param_cb(enabled, &enabled_param_ops, &enabled, 0600); MODULE_PARM_DESC(enabled, "Enable or disable DAMON_SAMPLE_PRCL"); static struct damon_ctx *ctx; -static struct pid *target_pidp; static int damon_sample_prcl_repeat_call_fn(void *data) { @@ -79,12 +78,10 @@ static int damon_sample_prcl_start(void) return -ENOMEM; } damon_add_target(ctx, target); - target_pidp = find_get_pid(target_pid); - if (!target_pidp) { + if (damon_set_target_pid(target, target_pid)) { damon_destroy_ctx(ctx); return -EINVAL; } - target->pid = target_pidp; scheme = damon_new_scheme( &(struct damos_access_pattern) { diff --git a/samples/damon/wsse.c b/samples/damon/wsse.c index 37fd5da2015885..53944aea8428ea 100644 --- a/samples/damon/wsse.c +++ b/samples/damon/wsse.c @@ -33,7 +33,6 @@ module_param_cb(enabled, &enabled_param_ops, &enabled, 0600); MODULE_PARM_DESC(enabled, "Enable or disable DAMON_SAMPLE_WSSE"); static struct damon_ctx *ctx; -static struct pid *target_pidp; static int damon_sample_wsse_repeat_call_fn(void *data) { @@ -79,12 +78,10 @@ static int damon_sample_wsse_start(void) return -ENOMEM; } damon_add_target(ctx, target); - target_pidp = find_get_pid(target_pid); - if (!target_pidp) { + if (damon_set_target_pid(target, target_pid)) { damon_destroy_ctx(ctx); return -EINVAL; } - target->pid = target_pidp; err = damon_start(&ctx, 1, true); if (err) { From c08e846fd42dc852012acc2991149dd8fb0befb2 Mon Sep 17 00:00:00 2001 From: Li Youhong Date: Mon, 31 Aug 2026 07:26:05 -0700 Subject: [PATCH 0333/1012] mm/damon/ops-common: factor out damon_putback_folio_list() The putback loop is duplicated in damon_migrate_folio_list() and on the invalid-nid path of damon_migrate_pages(). Factor it into a small helper for readability. No functional change. Link: https://lore.kernel.org/20260831142611.77572-4-sj@kernel.org Signed-off-by: Li Youhong Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Enze Li Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo --- mm/damon/ops-common.c | 30 +++++++++++++++--------------- 1 file changed, 15 insertions(+), 15 deletions(-) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index 8fc61d06d35859..7219c608b1952b 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -335,6 +335,19 @@ static unsigned int __damon_migrate_folio_list( return nr_succeeded; } +static void damon_putback_folio_list(struct list_head *folio_list) +{ + struct folio *folio; + + while (!list_empty(folio_list)) { + folio = lru_to_folio(folio_list); + list_del(&folio->lru); + node_stat_sub_folio(folio, NR_ISOLATED_ANON + + folio_is_file_lru(folio)); + folio_putback_lru(folio); + } +} + static unsigned int damon_migrate_folio_list(struct list_head *folio_list, struct pglist_data *pgdat, int target_nid) @@ -376,13 +389,7 @@ static unsigned int damon_migrate_folio_list(struct list_head *folio_list, list_splice(&ret_folios, folio_list); - while (!list_empty(folio_list)) { - folio = lru_to_folio(folio_list); - list_del(&folio->lru); - node_stat_sub_folio(folio, NR_ISOLATED_ANON + - folio_is_file_lru(folio)); - folio_putback_lru(folio); - } + damon_putback_folio_list(folio_list); return nr_migrated; } @@ -399,14 +406,7 @@ unsigned long damon_migrate_pages(struct list_head *folio_list, int target_nid) if (target_nid < 0 || target_nid >= MAX_NUMNODES || !node_state(target_nid, N_MEMORY)) { - while (!list_empty(folio_list)) { - struct folio *folio = lru_to_folio(folio_list); - - list_del(&folio->lru); - node_stat_sub_folio(folio, NR_ISOLATED_ANON + - folio_is_file_lru(folio)); - folio_putback_lru(folio); - } + damon_putback_folio_list(folio_list); return nr_migrated; } From 261e71c00debb94e85927aa194f23744fe207010 Mon Sep 17 00:00:00 2001 From: Hari Mishal Date: Mon, 31 Aug 2026 07:26:06 -0700 Subject: [PATCH 0334/1012] selftests/damon/sysfs.py: clean up sh processes used for obsolete_target test The obsolete_target test spawns three sh processes and uses their pids as DAMON monitoring targets. These processes are never terminated or waited on, so they are left running (or become zombies) as orphaned children after the test program exits. Terminate each process and communicate() with it after the targets are no longer needed, so it exits and gets reaped instead of being leaked. Link: https://lore.kernel.org/20260831142611.77572-5-sj@kernel.org Signed-off-by: Hari Mishal Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Greg Kroah-Hartman Cc: Enze Li Cc: Jaeyeon Lee Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo --- tools/testing/selftests/damon/sysfs.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/tools/testing/selftests/damon/sysfs.py b/tools/testing/selftests/damon/sysfs.py index 3ffa054b63867d..88a26422ff44cf 100755 --- a/tools/testing/selftests/damon/sysfs.py +++ b/tools/testing/selftests/damon/sysfs.py @@ -385,6 +385,10 @@ def main(): assert_ctxs_committed(kdamonds) kdamonds.stop() + for proc in (proc1, proc2, proc3): + proc.terminate() + proc.communicate() + test_memcg_filter_memcg_path_staging() if __name__ == '__main__': From 5d013d41d6cfb5717a0f8eb1d3069128f2862e05 Mon Sep 17 00:00:00 2001 From: Jaeyeon Lee Date: Mon, 31 Aug 2026 07:26:07 -0700 Subject: [PATCH 0335/1012] mm/damon/tests: use scoped_guard() for damon_test_ops_registration Replace manual mutex_lock() and mutex_unlock() calls with the scoped_guard() macro. This simplifies the code, improves readability, and ensures that the lock is automatically released when the scope ends, preventing potential lock leaks in the future. Link: https://lore.kernel.org/20260831142611.77572-6-sj@kernel.org Signed-off-by: Jaeyeon Lee Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Enze Li Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo --- mm/damon/tests/core-kunit.h | 21 ++++++++++----------- 1 file changed, 10 insertions(+), 11 deletions(-) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index b1ca4c8e03f091..7071ec277b0072 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -441,17 +441,17 @@ static void damon_test_ops_registration(struct kunit *test) KUNIT_EXPECT_EQ(test, damon_select_ops(c, NR_DAMON_OPS), -EINVAL); /* Registration should success after unregistration */ - mutex_lock(&damon_ops_lock); - bak = damon_registered_ops[DAMON_OPS_VADDR]; - damon_registered_ops[DAMON_OPS_VADDR] = (struct damon_operations){}; - mutex_unlock(&damon_ops_lock); + scoped_guard(mutex, &damon_ops_lock) { + bak = damon_registered_ops[DAMON_OPS_VADDR]; + damon_registered_ops[DAMON_OPS_VADDR] = + (struct damon_operations){}; + } ops.id = DAMON_OPS_VADDR; KUNIT_EXPECT_EQ(test, damon_register_ops(&ops), 0); - mutex_lock(&damon_ops_lock); - damon_registered_ops[DAMON_OPS_VADDR] = bak; - mutex_unlock(&damon_ops_lock); + scoped_guard(mutex, &damon_ops_lock) + damon_registered_ops[DAMON_OPS_VADDR] = bak; /* Check double-registration failure again */ KUNIT_EXPECT_EQ(test, damon_register_ops(&ops), -EINVAL); @@ -459,10 +459,9 @@ static void damon_test_ops_registration(struct kunit *test) damon_destroy_ctx(c); if (need_cleanup) { - mutex_lock(&damon_ops_lock); - damon_registered_ops[DAMON_OPS_VADDR] = - (struct damon_operations){}; - mutex_unlock(&damon_ops_lock); + scoped_guard(mutex, &damon_ops_lock) + damon_registered_ops[DAMON_OPS_VADDR] = + (struct damon_operations){}; } } From 816af2b3fda1f7f081e307eb345a24b4680ede75 Mon Sep 17 00:00:00 2001 From: zhaozhengzhuo Date: Mon, 31 Aug 2026 07:26:08 -0700 Subject: [PATCH 0336/1012] selftests/damon: prevent remaining cross-object state pollution _damon_sysfs.py defines constructors with mutable default arguments, including DamosAccessPattern(), DamosQuota(), DamosWatermarks(), DamosDests(), IntervalsGoal(), and empty lists. Default arguments are evaluated once at function definition time. Damos() instances created without explicit arguments therefore share the same DamosQuota(), and the other default-constructed sub-objects and lists are shared in the same way. The sub-objects keep back-pointers to their owner scheme, so constructing the second Damos() rebinds the shared quota's scheme pointer to the second object. An item appended to one object's default contexts or filters list is also visible from other default-constructed objects. The shared state can corrupt test configurations. DamosQuota.sysfs_dir() derives the sysfs directory from its scheme pointer, so operating on the first scheme's default quota may write to the second scheme's directory. The wrong values often match the defaults, so tests still pass, but the behavior depends on object creation order. Commit 8319dadcbd81 ("selftests/damon: prevent cross-context state pollution in DamonCtx") fixed the same pattern in DamonCtx only. Fix the remaining constructors by defaulting to None and creating fresh objects or lists inside each constructor. Explicit arguments keep their previous behavior. Link: https://lore.kernel.org/20260831142611.77572-7-sj@kernel.org Signed-off-by: zhaozhengzhuo Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Enze Li Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" --- tools/testing/selftests/damon/_damon_sysfs.py | 38 ++++++++++++++----- 1 file changed, 28 insertions(+), 10 deletions(-) diff --git a/tools/testing/selftests/damon/_damon_sysfs.py b/tools/testing/selftests/damon/_damon_sysfs.py index e6a2265d721e8b..f604b7d6530b3c 100644 --- a/tools/testing/selftests/damon/_damon_sysfs.py +++ b/tools/testing/selftests/damon/_damon_sysfs.py @@ -321,8 +321,10 @@ class DamosFilters: filters = None scheme = None # owner scheme - def __init__(self, name, filters=[]): + def __init__(self, name, filters=None): self.name = name + if filters is None: + filters = [] self.filters = filters for idx, filter_ in enumerate(self.filters): filter_.idx = idx @@ -368,7 +370,9 @@ class DamosDests: dests = None scheme = None # owner scheme - def __init__(self, dests=[]): + def __init__(self, dests=None): + if dests is None: + dests = [] self.dests = dests for idx, dest in enumerate(self.dests): dest.idx = idx @@ -426,15 +430,21 @@ class Damos: stats = None tried_regions = None - def __init__(self, action='stat', access_pattern=DamosAccessPattern(), - quota=DamosQuota(), watermarks=DamosWatermarks(), - core_filters=[], ops_filters=[], filters=[], target_nid=0, - dests=DamosDests(), apply_interval_us=0): + def __init__(self, action='stat', access_pattern=None, quota=None, + watermarks=None, core_filters=None, ops_filters=None, + filters=None, target_nid=0, dests=None, + apply_interval_us=0): self.action = action + if access_pattern is None: + access_pattern = DamosAccessPattern() self.access_pattern = access_pattern self.access_pattern.scheme = self + if quota is None: + quota = DamosQuota() self.quota = quota self.quota.scheme = self + if watermarks is None: + watermarks = DamosWatermarks() self.watermarks = watermarks self.watermarks.scheme = self @@ -448,6 +458,8 @@ def __init__(self, action='stat', access_pattern=DamosAccessPattern(), self.filters.scheme = self self.target_nid = target_nid + if dests is None: + dests = DamosDests() self.dests = dests self.dests.scheme = self @@ -568,10 +580,12 @@ class DamonAttrs: context = None def __init__(self, sample_us=5000, aggr_us=100000, - intervals_goal=IntervalsGoal(), update_us=1000000, - min_nr_regions=10, max_nr_regions=1000): + intervals_goal=None, update_us=1000000, min_nr_regions=10, + max_nr_regions=1000): self.sample_us = sample_us self.aggr_us = aggr_us + if intervals_goal is None: + intervals_goal = IntervalsGoal() self.intervals_goal = intervals_goal self.intervals_goal.attrs = self self.update_us = update_us @@ -703,7 +717,9 @@ class Kdamond: idx = None # index of this kdamond between siblings kdamonds = None # parent - def __init__(self, contexts=[], refresh_ms=None): + def __init__(self, contexts=None, refresh_ms=None): + if contexts is None: + contexts = [] self.contexts = contexts self.refresh_ms = refresh_ms for idx, context in enumerate(self.contexts): @@ -853,7 +869,9 @@ def commit_schemes_quota_goals(self): class Kdamonds: kdamonds = [] - def __init__(self, kdamonds=[]): + def __init__(self, kdamonds=None): + if kdamonds is None: + kdamonds = [] self.kdamonds = kdamonds for idx, kdamond in enumerate(self.kdamonds): kdamond.idx = idx From 9b8f7a6b583a41581fed45615f1d6bd701d76c39 Mon Sep 17 00:00:00 2001 From: Enze Li Date: Mon, 31 Aug 2026 07:26:09 -0700 Subject: [PATCH 0337/1012] samples/damon/mtier: add comment for struct region_range The mtier sample defines a local struct region_range using phys_addr_t instead of damon_addr_range which uses unsigned long. Add a comment explaining the rationale: on 32-bit systems with more than 4GiB memory, phys_addr_t will be 64-bit while unsigned long is 32-bit. Link: https://lore.kernel.org/20260831142611.77572-8-sj@kernel.org Signed-off-by: Enze Li Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Greg Kroah-Hartman Cc: Hari Mishal Cc: Jaeyeon Lee Cc: Li Youhong Cc: Shuah Khan Cc: "Zenghui Yu (Huawei)" Cc: zhaozhengzhuo --- samples/damon/mtier.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/samples/damon/mtier.c b/samples/damon/mtier.c index d1123ebbfab906..bea45c87cc9bed 100644 --- a/samples/damon/mtier.c +++ b/samples/damon/mtier.c @@ -52,6 +52,11 @@ module_param(detect_node_addresses, bool, 0600); static struct damon_ctx *ctxs[2]; +/* + * Use phys_addr_t instead of damon_addr_range (unsigned long) for physical + * addresses. On 32-bit systems with more than 4GB memory, phys_addr_t will + * be 64-bit while unsigned long is 32-bit. + */ struct region_range { phys_addr_t start; phys_addr_t end; From a15241efc4a3c489002c0117caeb43db2102d5e1 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:32 +0800 Subject: [PATCH 0338/1012] mm: fix stale ZONE_DEVICE refcount comment Patch series "mm: optimize zone-device memmap initialization", v11. memmap_init_zone_device() can take a noticeable amount of time when large pmem namespaces are bound or rebound, because it initializes nearly identical struct page descriptors one PFN at a time. This series reduces that ZONE_DEVICE memmap initialization overhead by reusing prepared struct page templates and, on x86, using memcpy_nontemporal() for the template copy path. The main target is large fsdax/devdax pmem configurations, where the cost of initializing the memmap shows up directly in nd_pmem/dax_pmem bind and rebind latency. This matters because the cost is paid in the synchronous probe/bind path for large DAX/PMEM ZONE_DEVICE mappings. Userspace workflows such as provisioning or reconfiguring nd_pmem/dax_pmem namespaces, bringing hot-added PMEM-backed capacity online, and recovering or rebinding a device after driver or device changes all wait for this initialization to finish. Reducing this cost will yield benefits as lower user-visible provisioning, hot-add, recovery, and rebind latency for large DAX/PMEM devices. Patches 1-2 are preparatory cleanups and helper extraction. Patches 3-4 add the template-copy path for head pages and compound tails. Patch 5 introduces memcpy_nontemporal(). Patch 6 switches the ZONE_DEVICE template-copy path over to memcpy_nontemporal(). Patch 7 extends the x86 fixed-size memcpy_flushcache() inline cases used by the x86 memcpy_nontemporal() backend for struct page sized copies. Architectures without a specialized memcpy_nontemporal() backend fall back to memcpy(), so the generic template-copy optimization remains available without arch-specific support. On x86, memcpy_nontemporal() maps to the existing memcpy_flushcache() backend and can use the fixed-size MOVNTI paths added by this series for struct page sized copies. memcpy_nontemporal() is only a copy primitive. It does not imply a drain or a publication barrier. Callers that use it before a producer-consumer or device-visible handoff must provide the required ordering. The ZONE_DEVICE template-copy path uses it only while initializing struct page metadata, so the copy primitive itself does not grow a separate drain contract. The numbers below measure the time spent in memmap_init_zone_device() during driver bind/rebind. They are not measurements of the full nd_pmem or dax_pmem bind/rebind operation. Tested in an x86_64 QEMU/KVM VM with a 100 GB fsdax namespace device configured with map=dev and a 100 GB devdax namespace (align=2097152) on Intel Ice Lake server. Test procedure: Rebind the nd_pmem and dax_pmem drivers 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.3-rc1): Average of nd_pmem rebinds: 221.07 ms Average of dax_pmem rebinds: 191.20 ms With this series applied: Average of nd_pmem rebinds: 71.93 ms Average of dax_pmem rebinds: 87.37 ms This reduces the average memmap initialization time measured during rebind by about 67.5% for nd_pmem and 54.3% for dax_pmem. As an additional x86_64 data point, I also ran measurements on the same physical host with a 100 GB PMEM region created via the memmap= kernel command line, configured as fsdax and devdax namespaces with map=dev and 2 MiB alignment. For brevity, the individual patches keep only the VM results rather than including a second set of physical-host measurements throughout the series. The physical-host numbers below are included only as supplemental evidence that the same optimization also provides a similar benefit on a non-virtualized system. Test procedure: Reconfigure the namespace mode, rebind the nd_pmem or dax_pmem driver 30 times, and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base (v7.3-rc1): nd_pmem / fsdax: 205.90 ms dax_pmem / devdax: 225.43 ms With this series applied: nd_pmem / fsdax: 69.13 ms dax_pmem / devdax: 90.67 ms This reduces the measured memmap initialization time during rebind by about 66.4% for nd_pmem and 59.8% for dax_pmem on that setup, which is broadly consistent with the VM results above. As another supplemental data point, I measured the test_hmm.ko module on the same physical x86_64 host, using the test_hmm.ko setup from the previous discussion that times ten 64 GB memremap_pages()/memunmap_pages() iterations during module insertion[1]. By default, module insertion initializes two DEVICE_PRIVATE dmirror devices, so two avg memremap values are reported; each value is the average for one 64 GB chunk. This is not the primary target workload of the series, but it exercises the same large ZONE_DEVICE memmap initialization path and shows the same direction of improvement. Base (v7.3-rc1): avg memremap reported during module insertion: 116500596 ns, 116438028 ns With this series applied: avg memremap reported during module insertion: 46953088 ns, 46428399 ns This corresponds to about a 59.9% reduction based on the mean of the reported values, which is again consistent with the pmem bind/rebind results above. I also include an arm64 data point for the generic template-copy part. It was measured on an arm64 QEMU virt VM with 64 KB pages and a 100 GB ACPI NVDIMM sparse backend. This setup does not use the x86 MOVNTI fast paths, so it exercises the architecture-independent part of the optimization. For devdax, 2 MiB alignment is rejected in this 64 KB page setup, so the devdax namespace was tested with the supported default 512 MiB alignment. Base (v7.3-rc1): Average of rebinds for nd_pmem driver: 27.93 ms Average of rebinds for dax_pmem driver: 27.87 ms With this series applied: Average of rebinds for nd_pmem driver: 14.53 ms Average of rebinds for dax_pmem driver: 16.27 ms This reduces the average memmap initialization time measured during rebind by about 48.0% for nd_pmem and 41.6% for dax_pmem on that arm64 VM setup. Since this arm64 setup does not use the x86 MOVNTI fast paths, the result also suggests that the generic template-copy optimization can benefit architectures without an architecture-specific memcpy_nontemporal() backend. This patch (of 7): The comment in __init_zone_device_page() still uses the old MEMORY_TYPE_* names and implies that FS_DAX pages regain a refcount of 1 in the free path. That no longer matches the code. Update the comment to describe the current policy correctly: MEMORY_DEVICE_GENERIC pages regain a refcount of 1 in the free path, while the remaining ZONE_DEVICE types start from 0 here and raise the count again when the allocator or driver hands the page out. No functional change intended. Link: https://lore.kernel.org/20260831111638.76012-1-lizhe.67@bytedance.com Link: https://lore.kernel.org/20260831111638.76012-2-lizhe.67@bytedance.com Link: https://lore.kernel.org/all/aiEoByaQdRR3xtM5@nvdebian.thelocal/ [1] Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Alistair Popple Reviewed-by: Muchun Song Reviewed-by: Mike Rapoport (Microsoft) Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: Ingo Molnar Cc: Kees Cook --- mm/mm_init.c | 10 +++------- 1 file changed, 3 insertions(+), 7 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 304c88da5cce21..951e6fc17f581b 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1012,13 +1012,9 @@ static void __ref __init_zone_device_page(struct page *page, unsigned long pfn, page->zone_device_data = NULL; /* - * ZONE_DEVICE pages other than MEMORY_TYPE_GENERIC are released - * directly to the driver page allocator which will set the page count - * to 1 when allocating the page. - * - * MEMORY_TYPE_GENERIC and MEMORY_TYPE_FS_DAX pages automatically have - * their refcount reset to one whenever they are freed (ie. after - * their refcount drops to 0). + * MEMORY_DEVICE_GENERIC pages regain a refcount of 1 in the free + * path. The remaining ZONE_DEVICE types start from 0 here and raise + * the count again when the allocator or driver hands the page out. */ switch (pgmap->type) { case MEMORY_DEVICE_FS_DAX: From 26c63c4428d0d2a00a140cdca03593eb02249e9e Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:33 +0800 Subject: [PATCH 0339/1012] mm: add a set_page_section_from_pfn() helper Callers that want to update section bits from a PFN currently need to open-code: set_page_section(page, pfn_to_section_nr(pfn)); and guard that sequence with #ifdef SECTION_IN_PAGE_FLAGS. Add set_page_section_from_pfn() to wrap that update in one place. When section bits are stored in page flags, the helper derives the section number from the PFN and updates the page flags. Otherwise keep it as a no-op so callers can use one helper without open-coding SECTION_IN_PAGE_FLAGS. Convert set_page_links() to use the new helper so later ZONE_DEVICE fast-path patches can also update section bits without open-coding SECTION_IN_PAGE_FLAGS at each callsite. This keeps the PFN-to-section translation local to the configurations that actually store section bits in struct page flags, and avoids exposing that detail to generic callers. No functional change intended. Link: https://lore.kernel.org/20260831111638.76012-3-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Muchun Song Reviewed-by: Balbir Singh Cc: Alistair Popple Cc: Arnd Bergmann Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook --- include/linux/mm.h | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index a9fbe26536f450..1b28e6fc8d5dd1 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -2632,12 +2632,23 @@ static inline void set_page_section(struct page *page, unsigned long section) page->flags.f |= (section & SECTIONS_MASK) << SECTIONS_PGSHIFT; } +static inline void set_page_section_from_pfn(struct page *page, + unsigned long pfn) +{ + set_page_section(page, pfn_to_section_nr(pfn)); +} + static inline unsigned long memdesc_section(const memdesc_flags_t *mdf) { ASSERT_EXCLUSIVE_BITS(mdf->f, SECTIONS_MASK << SECTIONS_PGSHIFT); return (mdf->f >> SECTIONS_PGSHIFT) & SECTIONS_MASK; } #else /* !SECTION_IN_PAGE_FLAGS */ +static inline void set_page_section_from_pfn(struct page *page, + unsigned long pfn) +{ +} + static inline unsigned long memdesc_section(const memdesc_flags_t *mdf) { return 0; @@ -2860,9 +2871,7 @@ static inline void set_page_links(struct page *page, enum zone_type zone, { set_page_zone(page, zone); set_page_node(page, node); -#ifdef SECTION_IN_PAGE_FLAGS - set_page_section(page, pfn_to_section_nr(pfn)); -#endif + set_page_section_from_pfn(page, pfn); } /** From 7cf6f67420f2f1b13c6ce2d2f7194f717b25a146 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:34 +0800 Subject: [PATCH 0340/1012] mm: add a template-based fast path for zone-device page init memmap_init_zone_device() repeats nearly identical head-page initialization for each PFN. Initialize the first real ZONE_DEVICE head page through the existing path, copy that final state into a reusable template, refresh the PFN-dependent fields in that template before each copy, and copy it into the remaining destination pages. Use the template path unconditionally. The page_ref_set tracepoint is primarily a debugging aid, while this code is still initializing struct pages before they are handed out. From the perspective of users of those pages, the initialization-time refcount transitions are not part of the observable page lifetime. This means page_ref_set will no longer observe every initialization-time refcount assignment for copied ZONE_DEVICE head pages. The impact is controlled because the final initialized struct page state is unchanged, and keeping a separate non-template path only for this local tracepoint observability would add complexity to the common path. This patch accelerates head-page initialization. The pfns_per_compound == 1 case gets the full benefit here, compound tails are handled in the next patch. Tested in a VM with a 100 GB fsdax namespace device configured with map=dev on Intel Ice Lake server. This test exercises the nd_pmem rebind path (pfns_per_compound == 1). Test procedure: Rebind the nd_pmem driver 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.3-rc1): Average of rebinds for nd_pmem driver: 221.07 ms With this patch and its prerequisites applied: Average of rebinds for nd_pmem driver: 155.00 ms This reduces the average memmap initialization time measured during rebind from 221.07 ms to 155.00 ms, or about 29.9%. Link: https://lore.kernel.org/20260831111638.76012-4-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Muchun Song --- mm/mm_init.c | 38 +++++++++++++++++++++++++++++++++++--- 1 file changed, 35 insertions(+), 3 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 951e6fc17f581b..0494a795f5fa58 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1029,6 +1029,17 @@ static void __ref __init_zone_device_page(struct page *page, unsigned long pfn, } } +static void zone_device_page_init_from_template(struct page *page, + unsigned long pfn, struct page *template) +{ + set_page_section_from_pfn(template, pfn); +#ifdef WANT_PAGE_VIRTUAL + if (!is_highmem_idx(ZONE_DEVICE)) + set_page_address(template, __va(pfn << PAGE_SHIFT)); +#endif + memcpy(page, template, sizeof(*page)); +} + /* * With compound page geometry and when struct pages are stored in ram most * tail pages are reused. Consequently, the amount of unique struct pages to @@ -1091,6 +1102,8 @@ void __ref memmap_init_zone_device(struct zone *zone, unsigned long zone_idx = zone_idx(zone); unsigned long start = jiffies; int nid = pgdat->node_id; + struct page template; + struct page *page; if (WARN_ON_ONCE(!pgmap || zone_idx != ZONE_DEVICE)) return; @@ -1105,10 +1118,29 @@ void __ref memmap_init_zone_device(struct zone *zone, nr_pages = end_pfn - start_pfn; } - for (pfn = start_pfn; pfn < end_pfn; pfn += pfns_per_compound) { - struct page *page = pfn_to_page(pfn); + if (!nr_pages) + return; - __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); + /* + * Seed the reusable head-page template from the first real struct + * page. The normal page-init and refcount helpers must operate on + * a real memmap entry rather than a stack object. + */ + pfn = start_pfn; + page = pfn_to_page(pfn); + __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); + memcpy(&template, page, sizeof(*page)); + if (pfns_per_compound != 1) + memmap_init_compound(page, pfn, zone_idx, nid, pgmap, + compound_nr_pages(pfn, altmap, pgmap)); + pfn += pfns_per_compound; + + /* Initialize the remaining head pages from template. */ + for (; pfn < end_pfn; pfn += pfns_per_compound) { + page = pfn_to_page(pfn); + + zone_device_page_init_from_template(page, pfn, + &template); if (IS_ALIGNED(pfn, PAGES_PER_SECTION)) cond_resched(); From 8dbe62ec77c8c5581194178e4d58845b3f8d8bd5 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Thu, 3 Sep 2026 10:58:06 +0800 Subject: [PATCH 0341/1012] mm-add-a-template-based-fast-path-for-zone-device-page-init-fix whitespace fix, per Mike Link: https://lore.kernel.org/20260903025806.70825-1-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Cc: Mike Rapoport (Microsoft) --- mm/mm_init.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 0494a795f5fa58..af11885f17ef05 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1139,8 +1139,7 @@ void __ref memmap_init_zone_device(struct zone *zone, for (; pfn < end_pfn; pfn += pfns_per_compound) { page = pfn_to_page(pfn); - zone_device_page_init_from_template(page, pfn, - &template); + zone_device_page_init_from_template(page, pfn, &template); if (IS_ALIGNED(pfn, PAGES_PER_SECTION)) cond_resched(); From a8659aa604c9ddecfdffd4c386899fafc5db6069 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:35 +0800 Subject: [PATCH 0342/1012] mm: extend the template fast path to zone-device compound tails The template fast path from the previous patch only accelerates head pages. Compound tails in memmap_init_compound() still go through the normal initialization path one by one. Build separate head and tail templates and reuse one prepared tail template across the tail pages in a compound range. Head pages preserve the existing refcount policy, while compound tails always start with a refcount of 0 after prep_compound_tail(). This extends the template-copy fast path to pfns_per_compound > 1. Tail-page PFN-dependent fields are refreshed in the reusable tail template before each copy. Do not keep a separate non-template fallback for compound tails either. These pages are still under memmap initialization, and the initialization-time refcount updates are not part of the observable lifetime of pages handed out later. The impact is controlled for the same reason as for head pages. The first tail page still seeds the reusable tail template through the normal tail initialization sequence, and the copied tail pages have the same final initialized state except for the PFN-dependent fields refreshed before each copy. Tested in a VM with a 100 GB devdax namespace (align=2097152) on Intel Ice Lake server. This test exercises the dax_pmem rebind path and measures memmap initialization latency. Test procedure: Unbind and rebind the dax_pmem driver 30 times, collect memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.3-rc1): Average of rebinds for dax_pmem driver: 191.20 ms With this patch and its prerequisites applied: Average of rebinds for dax_pmem driver: 176.87 ms This reduces the average memmap initialization time measured during rebind from 191.20 ms to 176.87 ms, or about 7.5%. Link: https://lore.kernel.org/20260831111638.76012-5-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Muchun Song --- mm/mm_init.c | 24 ++++++++++++++++++------ 1 file changed, 18 insertions(+), 6 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index af11885f17ef05..e2b0952023b3b5 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1072,6 +1072,8 @@ static void __ref memmap_init_compound(struct page *head, { unsigned long pfn, end_pfn = head_pfn + nr_pages; unsigned int order = pgmap->vmemmap_shift; + struct page template; + struct page *page; /* * We have to initialize the pages, including setting up page links. @@ -1080,13 +1082,23 @@ static void __ref memmap_init_compound(struct page *head, * the pages in the same go. */ __SetPageHead(head); - for (pfn = head_pfn + 1; pfn < end_pfn; pfn++) { - struct page *page = pfn_to_page(pfn); - __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); - prep_compound_tail(page, head, order); - set_page_count(page, 0); - } + /* + * All tails of the same compound page share the state established by + * prep_compound_tail(). Reuse one tail template for the whole range and + * refresh only the PFN-dependent fields in that template before each copy. + */ + pfn = head_pfn + 1; + page = pfn_to_page(pfn); + __init_zone_device_page(page, pfn, zone_idx, nid, pgmap); + prep_compound_tail(page, head, order); + set_page_count(page, 0); + memcpy(&template, page, sizeof(*page)); + + /* Initialize the remaining tail pages from template. */ + for (pfn = head_pfn + 2; pfn < end_pfn; pfn++) + zone_device_page_init_from_template(pfn_to_page(pfn), pfn, + &template); prep_compound_head(head, order); } From 25199bfbc633d068cc0f4646fd53e4546712020f Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:36 +0800 Subject: [PATCH 0343/1012] string: introduce memcpy_nontemporal() Introduce memcpy_nontemporal() for write-once copy sites that want a named non-temporal copy primitive. On x86_64, override the helper in arch/x86/include/asm/string_64.h using the usual self-macro pattern, next to the existing memcpy_flushcache() backend that memcpy_nontemporal() wraps. include/linux/string.h provides the generic memcpy_nontemporal() fallback as #define memcpy_nontemporal(dst, src, len) \ ((void)memcpy(dst, src, len)) instead of an inline wrapper, so architectures without a specialized backend keep the usual memcpy() FORTIFY coverage when the compiler can still see object sizes at the original call site. It also makes the memcpy_nontemporal() API uniformly void, matching memcpy_flushcache() and the x86 backend, so callers cannot accidentally depend on a return value on fallback architectures. memcpy_nontemporal() is only a copy primitive. It does not imply a drain or a publication barrier. Callers that use it before a producer-consumer or device-visible handoff must provide the required ordering at that handoff point. The immediate user is the ZONE_DEVICE template-copy path. It populates struct page descriptors in a write-once pattern, so a regular cached memcpy() can incur avoidable write-allocate traffic and cache pollution for data with little near-term reuse. Link: https://lore.kernel.org/20260831111638.76012-6-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Mike Rapoport (Microsoft) Cc: Muchun Song --- arch/x86/include/asm/string_64.h | 12 ++++++++++++ include/linux/string.h | 13 +++++++++++++ 2 files changed, 25 insertions(+) diff --git a/arch/x86/include/asm/string_64.h b/arch/x86/include/asm/string_64.h index 4635616863f53d..21ae515ae35a3d 100644 --- a/arch/x86/include/asm/string_64.h +++ b/arch/x86/include/asm/string_64.h @@ -100,6 +100,18 @@ static __always_inline void memcpy_flushcache(void *dst, const void *src, size_t } __memcpy_flushcache(dst, src, cnt); } + +#define memcpy_nontemporal memcpy_nontemporal +/* + * Reuse the existing x86 flushcache backend as the non-temporal copy + * primitive. + */ +static __always_inline void memcpy_nontemporal(void *dst, const void *src, + size_t cnt) +{ + memcpy_flushcache(dst, src, cnt); +} + #endif #endif /* __KERNEL__ */ diff --git a/include/linux/string.h b/include/linux/string.h index 5702daca4326b7..6cb5cdd01158b2 100644 --- a/include/linux/string.h +++ b/include/linux/string.h @@ -278,6 +278,19 @@ static inline void memcpy_flushcache(void *dst, const void *src, size_t cnt) } #endif +#ifndef memcpy_nontemporal +/* + * memcpy_nontemporal() requests a non-temporal copy when the + * architecture has a suitable backend. Architectures without a + * specialized backend fall back to memcpy(). Keep this as a + * function-like macro so the compiler can still see the original + * memcpy() call site and preserve the usual FORTIFY coverage when + * object sizes remain visible there, while keeping the API void. + */ +#define memcpy_nontemporal(dst, src, len) \ + ((void)memcpy(dst, src, len)) +#endif + void *memchr_inv(const void *s, int c, size_t n); char *strreplace(char *str, char old, char new); From dce96d2860a36e396c4a0e0fef57c9ff944e3b68 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:37 +0800 Subject: [PATCH 0344/1012] mm: use memcpy_nontemporal() in zone-device template copies The template fast path currently uses memcpy() for the actual struct page copy. Switch zone_device_page_init_from_template() to memcpy_nontemporal(). ZONE_DEVICE memmap initialization is largely write-once: each struct page is populated once, and most destination cachelines are not expected to be reused immediately afterwards. On x86, a regular cached memcpy() can therefore incur write-allocate traffic by pulling destination cachelines into the cache before writeback, and can populate the cache with data that has little near-term reuse. Using memcpy_nontemporal() lets this path request nontemporal stores for that copy pattern, which can reduce cache pollution and avoid part of the associated write-allocate overhead, while architectures without a specialized backend still fall back to memcpy(). Do not add a KASAN/KMSAN-specific fallback around this call site. As Muchun pointed out, special KASAN handling for memcpy_flushcache() or memcpy_nontemporal(), if needed, belongs in the low-level helper rather than in this ZONE_DEVICE caller. No separate drain is added here. memcpy_nontemporal() is used only as the copy primitive while memmap_init_zone_device() is still initializing the struct page array. The ordinary stores that follow in this path, such as compound-page setup, are part of the same CPU's initialization sequence; they are not used as a publication store that tells another CPU or device to consume data written by the non-temporal copy. Therefore this call site does not need a helper-level drain for correctness. Callers that use memcpy_nontemporal() as part of a producer-consumer or device-visible handoff must add the required ordering themselves. Tested in a VM with a 100 GB fsdax namespace device configured with map=dev and a 100 GB devdax namespace (align=2097152) on Intel Ice Lake server. Test procedure: Rebind the nd_pmem and dax_pmem driver 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). Base(v7.3-rc1): Average of rebinds for nd_pmem driver: 221.07 ms Average of rebinds for dax_pmem driver: 191.20 ms With this patch and its prerequisites applied: Average of rebinds for nd_pmem driver: 150.40 ms Average of rebinds for dax_pmem driver: 161.83 ms This reduces the average memmap initialization time measured during rebind by about 32.0% for nd_pmem and 15.4% for dax_pmem. Link: https://lore.kernel.org/20260831111638.76012-7-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: "Borislav Petkov (AMD)" Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Mike Rapoport (Microsoft) Cc: Muchun Song --- mm/mm_init.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index e2b0952023b3b5..97e0158d2aca5b 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1037,7 +1037,7 @@ static void zone_device_page_init_from_template(struct page *page, if (!is_highmem_idx(ZONE_DEVICE)) set_page_address(template, __va(pfn << PAGE_SHIFT)); #endif - memcpy(page, template, sizeof(*page)); + memcpy_nontemporal(page, template, sizeof(*page)); } /* From 17f0917a561ffad60afee9d9a43ac72c806e4819 Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 31 Aug 2026 19:16:38 +0800 Subject: [PATCH 0345/1012] x86/string: extend memcpy_flushcache() fixed-size fastpaths The x86 memcpy_nontemporal() helper maps to memcpy_flushcache(), and the ZONE_DEVICE template-copy path uses it to copy one struct page at a time. The relevant copy size is sizeof(struct page). On x86_64, the base struct page layout is 64 bytes. Adding either the KMSAN metadata pointers or an out-of-flags last_cpupid field can make it 80 bytes after alignment, and enabling both can make it 96 bytes. memcpy_flushcache() currently only has inline fixed-size cases for 4, 8, and 16 bytes. As a result, these constant-sized struct page copies fall through to __memcpy_flushcache() even though the compiler knows the copy size at the call site. Add fixed-size MOVNTI cases up to 96 bytes so the ZONE_DEVICE template-copy path can keep these struct page copies in the inline memcpy_flushcache() path. This matters for ZONE_DEVICE memmap initialization because the copy happens once per initialized struct page. For a 100 GB fsdax namespace with map=dev, this is about 25 million struct page copies during nd_pmem binding or rebinding. Tested in a VM with a 100 GB fsdax namespace device configured with map=dev and a 100 GB devdax namespace (align=2097152) on Intel Ice Lake server. Test procedure: Rebind the nd_pmem and dax_pmem drivers 30 times and collect the memmap initialization time from the pr_debug() output of memmap_init_zone_device(). With memcpy_nontemporal() used by the ZONE_DEVICE template-copy path: Average of rebinds for nd_pmem driver: 150.40 ms Average of rebinds for dax_pmem driver: 161.83 ms With this x86 fixed-size fastpath patch applied: Average of rebinds for nd_pmem driver: 71.93 ms Average of rebinds for dax_pmem driver: 87.37 ms This further reduces the average memmap initialization time measured during rebind by about 52.2% for nd_pmem and 46.0% for dax_pmem. Link: https://lore.kernel.org/20260831111638.76012-8-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Suggested-by: Borislav Petkov Acked-by: Borislav Petkov (AMD) Acked-by: Dave Hansen Cc: Alistair Popple Cc: Arnd Bergmann Cc: Balbir Singh Cc: David Hildenbrand (Arm) Cc: Ingo Molnar Cc: Kees Cook Cc: Mike Rapoport (Microsoft) Cc: Muchun Song --- arch/x86/include/asm/string_64.h | 71 +++++++++++++++++++++++++------- 1 file changed, 56 insertions(+), 15 deletions(-) diff --git a/arch/x86/include/asm/string_64.h b/arch/x86/include/asm/string_64.h index 21ae515ae35a3d..831d3dda3b380e 100644 --- a/arch/x86/include/asm/string_64.h +++ b/arch/x86/include/asm/string_64.h @@ -82,23 +82,64 @@ int strcmp(const char *cs, const char *ct); #ifdef CONFIG_ARCH_HAS_UACCESS_FLUSHCACHE #define __HAVE_ARCH_MEMCPY_FLUSHCACHE 1 void __memcpy_flushcache(void *dst, const void *src, size_t cnt); -static __always_inline void memcpy_flushcache(void *dst, const void *src, size_t cnt) + +static __always_inline void movnti_4(void *dst, const void *src) +{ + asm volatile("movntil %1, %0" + : "=m"(*(u32 *)dst) + : "r"(*(const u32 *)src) + : "memory"); +} + +static __always_inline void movnti_8(void *dst, const void *src) +{ + asm volatile("movntiq %1, %0" + : "=m"(*(u64 *)dst) + : "r"(*(const u64 *)src) + : "memory"); +} + +static __always_inline void movnti_16(void *dst, const void *src) +{ + movnti_8(dst, src); + movnti_8(dst + 8, src + 8); +} + +static __always_inline void movnti_32(void *dst, const void *src) +{ + movnti_16(dst, src); + movnti_16(dst + 16, src + 16); +} + +static __always_inline void movnti_64(void *dst, const void *src) +{ + movnti_32(dst, src); + movnti_32(dst + 32, src + 32); +} + +static __always_inline void memcpy_flushcache(void *dst, const void *src, + size_t cnt) { - if (__builtin_constant_p(cnt)) { - switch (cnt) { - case 4: - asm ("movntil %1, %0" : "=m"(*(u32 *)dst) : "r"(*(u32 *)src)); - return; - case 8: - asm ("movntiq %1, %0" : "=m"(*(u64 *)dst) : "r"(*(u64 *)src)); - return; - case 16: - asm ("movntiq %1, %0" : "=m"(*(u64 *)dst) : "r"(*(u64 *)src)); - asm ("movntiq %1, %0" : "=m"(*(u64 *)(dst + 8)) : "r"(*(u64 *)(src + 8))); - return; - } + if (!__builtin_constant_p(cnt)) + return __memcpy_flushcache(dst, src, cnt); + + /* + * The relevant fixed-size copies here are the x86_64 struct page sizes: + * 64, 80, and 96 bytes. Keep 32-byte and 48-byte copies inline as well + * instead of sending those nearby fixed-size cases back to + * __memcpy_flushcache(). + */ + switch (cnt) { + case 4: movnti_4(dst, src); break; + case 8: movnti_8(dst, src); break; + case 16: movnti_16(dst, src); break; + case 32: movnti_32(dst, src); break; + case 48: movnti_32(dst, src); movnti_16(dst + 32, src + 32); break; + case 64: movnti_64(dst, src); break; + case 80: movnti_64(dst, src); movnti_16(dst + 64, src + 64); break; + case 96: movnti_64(dst, src); movnti_32(dst + 64, src + 64); break; + default: __memcpy_flushcache(dst, src, cnt); break; } - __memcpy_flushcache(dst, src, cnt); } #define memcpy_nontemporal memcpy_nontemporal From f58f4ae5bcda943c3d2fe879f74d1e653fb7c6a5 Mon Sep 17 00:00:00 2001 From: Dev Jain Date: Mon, 31 Aug 2026 08:28:47 +0000 Subject: [PATCH 0346/1012] mm/rmap: remove stale hugetlb check in try_to_unmap_one Post commit d4ec5572825a ("mm/rmap: add try_to_unmap_poisoned_hugetlb_one") try_to_unmap_one() cannot be called with a hugetlb folio. Therefore remove the folio_test_hugetlb() check. Link: https://lore.kernel.org/20260831082849.3573957-1-dev.jain@arm.com Signed-off-by: Dev Jain Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Lance Yang Reviewed-by: Kunwu Chan Acked-by: David Hildenbrand (Arm) Cc: Harry Yoo Cc: Jann Horn Cc: Liam R. Howlett Cc: Rik van Riel Cc: Vlastimil Babka --- mm/rmap.c | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/mm/rmap.c b/mm/rmap.c index 3c67ad0e95620f..0a3952706faf5c 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -2301,11 +2301,8 @@ static bool try_to_unmap_one(struct folio *folio, struct vm_area_struct *vma, VM_BUG_ON_FOLIO(!pvmw.pte, folio); address = pvmw.address; - if (folio_test_hugetlb(folio)) { - pteval = huge_ptep_get(mm, address, pvmw.pte); - } else { - pteval = ptep_get(pvmw.pte); - } + pteval = ptep_get(pvmw.pte); + if (likely(pte_present(pteval))) { pfn = pte_pfn(pteval); } else { From 92fcc5afc5d43fae816134ab9fa877125e54568b Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Mon, 31 Aug 2026 15:43:04 +0530 Subject: [PATCH 0347/1012] mm/gup_test: report actual pinned bytes __gup_test_ioctl() advances addr to the end of the current batch before checking if GUP pinned the entire requested batch. If GUP pins more than 0 pages but less than the requested batch size, addr still advances by the requested batch size. The next iteration detects the partial pinning and breaks out of the loop. Again gup->size is calculated using addr - gup->addr, so it also includes the unpinned pages of the requested batch. Calculate gup->size using the actual number of pages pinned multiplied by PAGE_SIZE. Link: https://lore.kernel.org/20260831101304.162867-1-sarthak.sharma@arm.com Fixes: 64c349f4ae78 ("mm: add infrastructure for get_user_pages_fast() benchmarking") Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu --- mm/gup_test.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/gup_test.c b/mm/gup_test.c index 44c1cdfb9c3717..185ba3bb8ed10b 100644 --- a/mm/gup_test.c +++ b/mm/gup_test.c @@ -188,7 +188,7 @@ static int __gup_test_ioctl(unsigned int cmd, nr_pages = i; gup->get_delta_usec = ktime_us_delta(end_time, start_time); - gup->size = addr - gup->addr; + gup->size = nr_pages * PAGE_SIZE; /* * Take an un-benchmark-timed moment to verify DMA pinned From 58819f88c724987d8280f2089e314e0512022002 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 31 Aug 2026 10:15:13 +0100 Subject: [PATCH 0348/1012] mm/huge_memory: do not touch frozen folios in deferred_split_isolate() Patch series "Fix deferred_split_isolate() and drop the split workaround", v2. deferred_split_isolate() probes each queued folio with folio_try_get(). folio_try_get() failure is treated as a lost race with folio_put(). It leads to wrong results when !folio_try_get() was not caused by folio_put(): for a frozen folio, PG_partially_mapped gets wrongfully cleared and the folio dropped from the queue. It came up in the review of my collapse RFC series: https://lore.kernel.org/all/20260824131224.73344-1-lance.yang@linux.dev/ The bug is inert in upstream code: - __folio_split() works around it; - __folio_migrate_mapping() freezes a folio it is about to replace; - reclaim freezes only what try_to_unmap() already unmapped. No cc:stable needed. But my collapse rework steps on it, so it is worth fixing. The branch the first patch removes also hid an inert, pre-existing bug in the zone device path: https://lore.kernel.org/all/20260827163838.1813081-1-usama.arif@linux.dev/ The first patch fixes deferred_split_isolate(). The second patch removes the workaround for this deferred_split_isolate() behaviour from __folio_freeze_and_split_unmapped(). Tested in a VM: split_huge_page_test, folio_split_race_test and cow pass. Also ran a test that leaves 16 partially mapped THPs on the deferred split queue and drives thp-deferred_split through debugfs, checking nr_anon_partially_mapped. This patch (of 2): deferred_split_isolate() probes each queued folio with folio_try_get(). folio_try_get() failure is treated as a lost race with folio_put(): clear PG_partially_mapped, correct MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, take the folio off the queue. The folio_put() race is the most common case for !folio_try_get(), but it is not the only option. Another scenario is folio_ref_freeze(). A zero refcount in such cases does not mean the folio is going away. It means "don't touch me" and current deferred_split_isolate() doesn't respect it. It can lead to unqueueing folios from the deferred list for no reason: CPU 0 CPU 1 --------------------------- ------------------------------ freeze a mapped folio deferred_split_scan() folio_ref_freeze() folio_try_get() fails folio_clear_partially_mapped() NR_ANON_PARTIALLY_MAPPED-- folio off the queue give up, put it back folio_ref_unfreeze() The folio is still partially mapped, but it is no longer a split candidate. Nothing queues it again until part of it is unmapped once more. Skip the folio instead: whoever freezes the folio, owns it and owner is responsible for its fate. It also covers the folio_put() case: __folio_put() unqueues the folio via folio_unqueue_deferred_split(). Nothing is lost by skipping. Everything that frees a queued folio unqueues it first, and folio_unqueue_deferred_split() clears PG_partially_mapped and brings MTHP_STAT_NR_ANON_PARTIALLY_MAPPED down on the way: __folio_put(), folios_put_refs() mm/folio.c __folio_migrate_mapping() mm/migrate.c shrink_folio_list() mm/vmscan.c __folio_freeze_and_split_unmapped() does the same by hand, under the list_lru lock it holds across the freeze. A freeze that ends in folio_ref_unfreeze() leaves a folio that is still partially mapped and still belongs on the queue. Link: https://lore.kernel.org/20260831091514.1879786-1-kirill@shutemov.name Link: https://lore.kernel.org/20260831091514.1879786-2-kirill@shutemov.name Fixes: 8422acdc97ed ("mm: introduce a pageflag for partially mapped folios") Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reported-by: Lance Yang Closes: https://lore.kernel.org/all/20260824131224.73344-1-lance.yang@linux.dev/ Reviewed-by: Zi Yan Reviewed-by: Johannes Weiner Reviewed-by: Lance Yang Acked-by: Usama Arif Reviewed-by: Baolin Wang Acked-by: David Hildenbrand (Arm) Assisted-by: Claude-Code:claude-opus-5 Cc: Balbir Singh Cc: Barry Song Cc: Dev Jain Cc: Hugh Dickins Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Ryan Roberts --- mm/huge_memory.c | 19 ++++--------------- 1 file changed, 4 insertions(+), 15 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 54494c3fa9835e..779c02e0bf6cc3 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4638,22 +4638,11 @@ static enum lru_status deferred_split_isolate(struct list_head *item, struct folio *folio = container_of(item, struct folio, _deferred_list); struct list_head *freeable = cb_arg; - if (folio_try_get(folio)) { - list_lru_isolate_move(lru, item, freeable); - return LRU_REMOVED; - } + /* Lost race to folio_put() or the folio is under folio_ref_freeze() */ + if (!folio_try_get(folio)) + return LRU_SKIP; - /* - * We lost race with folio_put(). Read folio state before the - * isolate: folio_unqueue_deferred_split() checks list_empty() - * locklessly, so once removed the folio can be freed any time. - */ - if (folio_test_partially_mapped(folio)) { - folio_clear_partially_mapped(folio); - mod_mthp_stat(folio_order(folio), - MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); - } - list_lru_isolate(lru, item); + list_lru_isolate_move(lru, item, freeable); return LRU_REMOVED; } From 77c87d7ec24bae451393f585f5ce75b8175f33a6 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 31 Aug 2026 10:15:14 +0100 Subject: [PATCH 0349/1012] mm/huge_memory: dequeue the deferred split after the split freeze __folio_freeze_and_split_unmapped() takes the deferred split list_lru lock across the freeze. It is only there to stop deferred_split_scan() from touching the folio under split. With deferred_split_isolate() fixed, the workaround can be dropped. Unqueue the folio after folio_ref_freeze(), the way __folio_migrate_mapping() does: folio_unqueue_deferred_split() needs a zero refcount and a memcg still set, and both hold there. If the split is called from deferred_split_scan(), the unqueue is a no-op -- the folio is already removed from the list. But PG_partially_mapped is still set, so it has to be cleared here or MTHP_STAT_NR_ANON_PARTIALLY_MAPPED never comes back down. Link: https://lore.kernel.org/20260831091514.1879786-3-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Johannes Weiner Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Assisted-by: Claude-Code:claude-opus-5 Cc: Balbir Singh Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Hugh Dickins Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Ryan Roberts Cc: Usama Arif --- mm/huge_memory.c | 46 ++++++++++++++-------------------------------- 1 file changed, 14 insertions(+), 32 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 779c02e0bf6cc3..c5d11147b69aec 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3979,41 +3979,27 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n struct folio *end_folio = folio_next(folio); struct folio *new_folio, *next; int old_order = folio_order(folio); - struct list_lru_one *lru; - bool dequeue_deferred; int ret = 0; VM_WARN_ON_ONCE(!mapping && end); - /* - * If this folio can be on the deferred split queue, lock out - * the shrinker before freezing the ref. If the shrinker sees - * a 0-ref folio, it assumes it beat folio_put() to the list - * lock and must clean up the LRU state - the same dequeue we - * will do below as part of the split. - */ - dequeue_deferred = folio_test_anon(folio) && old_order > 1; - if (dequeue_deferred) { - struct mem_cgroup *memcg; - - rcu_read_lock(); - memcg = folio_memcg(folio); - lru = list_lru_lock(&deferred_split_lru, - folio_nid(folio), &memcg); - } + if (folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { struct swap_cluster_info *ci = NULL; struct lruvec *lruvec; - if (dequeue_deferred) { - __list_lru_del(&deferred_split_lru, lru, - &folio->_deferred_list, folio_nid(folio)); - if (folio_test_partially_mapped(folio)) { - folio_clear_partially_mapped(folio); - mod_mthp_stat(old_order, - MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); - } - list_lru_unlock(lru); - rcu_read_unlock(); + /* Take off the deferred split queue while frozen and memcg set */ + folio_unqueue_deferred_split(folio); + + /* + * deferred_split_scan() takes the folio off the queue before it + * splits it, so the unqueue above finds an empty list and + * leaves PG_partially_mapped set. + * Clear it here: the flag does not survive the split. + */ + if (folio_test_partially_mapped(folio)) { + folio_clear_partially_mapped(folio); + mod_mthp_stat(old_order, + MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); } if (mapping) { @@ -4115,10 +4101,6 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n if (ci) swap_cluster_unlock(ci); } else { - if (dequeue_deferred) { - list_lru_unlock(lru); - rcu_read_unlock(); - } return -EAGAIN; } From aa066cd75ecb369e545d3373bd0a8b0338ae3dd3 Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Mon, 31 Aug 2026 21:35:18 +0800 Subject: [PATCH 0350/1012] mm/hugetlb: preserve source surplus accounting during demotion Patch series "mm/hugetlb: fix surplus accounting and availability checks during demotion", v2. Fix surplus accounting and availability checks in the hugetlb demote path. Patch 1 fixes source hstate accounting when the free folio selected for demotion accounts for a surplus page. Patch 2 prevents demotion from removing free huge pages that back reservations. Both fixes were tested with x86_64 QEMU guests. The commands below use: hstate=/sys/kernel/mm/hugepages/hugepages-1048576kB Patch 1: surplus accounting A vmemmap restoration failure is difficult to trigger deterministically. For this test only, add a one-shot fault injection that makes the first attempt to restore the vmemmap of an optimized 1 GiB folio fail: /* TEST ONLY: fail the first optimized 1G folio restore. */ static atomic_t fail_next_1g_restore = ATOMIC_INIT(1); /* In __hugetlb_vmemmap_restore_folio(). */ if (huge_page_size(h) == SZ_1G && atomic_cmpxchg(&fail_next_1g_restore, 1, 0) == 1) { pr_info("TEST ONLY: forcing one 1G vmemmap " "restore failure\n"); return -ENOMEM; } The injection does not modify the demotion or accounting code. It is one-shot so that the later restore performed during demotion can succeed. 1. Boot QEMU with: hugepagesz=1G hugepages=0 hugetlb_cma=1G hugetlb_free_vmemmap=on 2. Enable overcommit: echo 1 > "$hstate/nr_overcommit_hugepages" 3. Allocate one 1 GiB huge page: nr=1 surplus=1 free=0 resv=0 4. Unmap it. The forced restoration failure leaves the folio on the freelist while it is still accounted as surplus: nr=1 surplus=1 free=1 resv=0 5. Demote one page: echo 1 > "$hstate/demote" Before this fix: nr=0 surplus=1 free=0 resv=0 surplus > nr After this fix: nr=0 surplus=0 free=0 resv=0 Patch 2: cap demotion This reproducer requires no kernel instrumentation. 1. Boot QEMU with: hugepagesz=1G hugepages=2 nr=2 surplus=0 free=2 resv=0 2. Reserve one 1 GiB huge page with an untouched hugetlbfs mapping: nr=2 surplus=0 free=2 resv=1 3. Request demotion of two pages: echo 2 > "$hstate/demote" Before this fix: nr=0 surplus=0 free=0 resv=1 resv > free After this fix: nr=1 surplus=0 free=1 resv=1 resv == free 4. Touch the reserved page and let the process exit. Before this fix, the access fails with SIGBUS and leaves: nr=0 surplus=0 free=0 resv=0 After this fix, the access succeeds and leaves: nr=1 surplus=0 free=1 resv=0 This patch (of 2): demote_pool_huge_page() currently removes every source folio as a persistent folio. A free folio can instead account for one of the source hstate's surplus pages, for example after a vmemmap restoration failure. Removing such a folio without adjusting surplus_huge_pages makes the persistent count underflow, and later subtracting it from max_huge_pages can underflow that counter as well. Classify selected folios against the node's surplus count while holding hugetlb_lock, and preserve that classification on rollback. Track the number of successfully demoted persistent folios separately so only those folios reduce the source max_huge_pages target. All successfully demoted folios still increase the destination target because the new destination folios are added as persistent pages. Testing: Tested on an x86_64 QEMU guest booted with: hugepagesz=1G hugepages=0 hugetlb_cma=1G hugetlb_free_vmemmap=on For testing only, add a one-shot fault injection that makes the first call to __hugetlb_vmemmap_restore_folio() for an optimized 1 GiB folio return -ENOMEM. Set nr_overcommit_hugepages to 1, then allocate one 1 GiB huge page: nr=1 surplus=1 free=0 resv=0 Unmap it. The failed restoration leaves the folio on the freelist while it is still accounted as surplus: nr=1 surplus=1 free=1 resv=0 Demote one page. Before this fix, the result is: nr=0 surplus=1 free=0 resv=0 After this fix, the result is: nr=0 surplus=0 free=0 resv=0 The fault injection is one-shot, so the restore performed during demotion can succeed. Link: https://lore.kernel.org/20260831133519.2505020-2-xialonglong2025@163.com Fixes: 8531fc6f52f5 ("hugetlb: add hugetlb demote page support") Signed-off-by: Longlong Xia Signed-off-by: Andrew Morton Assisted-by: Codex:gpt-5.6-sol Cc: David Hildenbrand Cc: Muchun Song Cc: Oscar Salvador Cc: Yu Zhao --- mm/hugetlb.c | 35 +++++++++++++++++++++++++++++++---- 1 file changed, 31 insertions(+), 4 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 03182cc28a7dbc..ea79d31e6160ab 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -4011,6 +4011,7 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, struct hstate *dst; long rc = 0; long nr_demoted = 0; + long nr_persistent = 0; lockdep_assert_held(&hugetlb_lock); @@ -4023,22 +4024,40 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, for_each_node_mask_to_free(src, nr_nodes, node, nodes_allowed) { LIST_HEAD(list); + LIST_HEAD(surplus_list); struct folio *folio, *next; list_for_each_entry_safe(folio, next, &src->hugepage_freelists[node], lru) { + bool adjust_surplus; + if (folio_test_hwpoison(folio)) continue; - remove_hugetlb_folio(src, folio, false); - list_add(&folio->lru, &list); + /* Surplus accounting is maintained per node, not per folio. */ + adjust_surplus = src->surplus_huge_pages_node[node] > 0; + remove_hugetlb_folio(src, folio, adjust_surplus); + list_add(&folio->lru, adjust_surplus ? &surplus_list : &list); + if (!adjust_surplus) + nr_persistent++; if (++nr_demoted == nr_to_demote) break; } + if (list_empty(&list) && list_empty(&surplus_list)) + continue; + spin_unlock_irq(&hugetlb_lock); - rc = demote_free_hugetlb_folios(src, dst, &list); + if (!list_empty(&list)) + rc = demote_free_hugetlb_folios(src, dst, &list); + if (!list_empty(&surplus_list)) { + long tmp_rc; + + tmp_rc = demote_free_hugetlb_folios(src, dst, &surplus_list); + if (rc >= 0) + rc = tmp_rc; + } spin_lock_irq(&hugetlb_lock); @@ -4046,6 +4065,14 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, list_del(&folio->lru); add_hugetlb_folio(src, folio, false); + nr_demoted--; + nr_persistent--; + } + + list_for_each_entry_safe(folio, next, &surplus_list, lru) { + list_del(&folio->lru); + add_hugetlb_folio(src, folio, true); + nr_demoted--; } @@ -4057,7 +4084,7 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, * Not absolutely necessary, but for consistency update max_huge_pages * based on pool changes for the demoted page. */ - src->max_huge_pages -= nr_demoted; + src->max_huge_pages -= nr_persistent; dst->max_huge_pages += nr_demoted << (huge_page_order(src) - huge_page_order(dst)); if (rc < 0) From d79974291ef929a0b9fe6a6eb25bb322790e3a74 Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Mon, 31 Aug 2026 21:35:19 +0800 Subject: [PATCH 0351/1012] mm/hugetlb: cap demotion at currently available free pages Demotion must not remove free huge pages that back existing reservations. The sysfs path checks whether any page is available, but passes the entire request to demote_pool_huge_page(). For example, with two free pages and one reservation, a request for two pages removes both and leaves the reservation without a backing page. Cap the sysfs request by both global availability and the selected node's free pages. Recheck global availability in demote_pool_huge_page() before each node batch because that function drops hugetlb_lock while restoring vmemmap and reservations can change before the next batch. Testing: Tested on an x86_64 QEMU guest booted with: hugepagesz=1G hugepages=2 Reserve one 1 GiB huge page with an untouched hugetlbfs mapping: nr=2 surplus=0 free=2 resv=1 Request demotion of two pages. Before this fix, both free pages are demoted: nr=0 surplus=0 free=0 resv=1 Touching the reserved mapping then fails with SIGBUS. After this fix, the request is capped at the single available page: nr=1 surplus=0 free=1 resv=1 Touching the reserved mapping succeeds. After the process exits, the counters are: nr=1 surplus=0 free=1 resv=0 Link: https://lore.kernel.org/20260831133519.2505020-3-xialonglong2025@163.com Fixes: c0f398c3b2cf ("mm/hugetlb_vmemmap: batch HVO work when demoting") Signed-off-by: Longlong Xia Signed-off-by: Andrew Morton Assisted-by: Codex:gpt-5.6-sol Cc: David Hildenbrand Cc: Muchun Song Cc: Oscar Salvador Cc: Yu Zhao --- mm/hugetlb.c | 22 +++++++++++++++++++++- mm/hugetlb_sysfs.c | 10 +++++----- 2 files changed, 26 insertions(+), 6 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index ea79d31e6160ab..7edc2a860a4007 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -4026,6 +4026,26 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, LIST_HEAD(list); LIST_HEAD(surplus_list); struct folio *folio, *next; + unsigned long nr_available, nr_target; + + /* + * Re-check available each node batch: the previous + * batch released hugetlb_lock for vmemmap restore/split, + * and a new reservation could have been added in that + * window, shrinking the budget. available is global + * (resv is not per-node), so 0 means no node can + * contribute -- stop the whole scan. + */ + nr_available = available_huge_pages(src); + if (!nr_available) + break; + + /* + * Cap this batch at the current budget; expressed as a + * cumulative stop point because nr_demoted is running. + */ + nr_target = nr_demoted + min_t(unsigned long, + nr_to_demote - nr_demoted, nr_available); list_for_each_entry_safe(folio, next, &src->hugepage_freelists[node], lru) { bool adjust_surplus; @@ -4040,7 +4060,7 @@ long demote_pool_huge_page(struct hstate *src, nodemask_t *nodes_allowed, if (!adjust_surplus) nr_persistent++; - if (++nr_demoted == nr_to_demote) + if (++nr_demoted == nr_target) break; } diff --git a/mm/hugetlb_sysfs.c b/mm/hugetlb_sysfs.c index 79ece91406bfa4..326a54b4d991c9 100644 --- a/mm/hugetlb_sysfs.c +++ b/mm/hugetlb_sysfs.c @@ -211,15 +211,15 @@ static ssize_t demote_store(struct kobject *kobj, * Check for available pages to demote each time thorough the * loop as demote_pool_huge_page will drop hugetlb_lock. */ + nr_available = h->free_huge_pages - h->resv_huge_pages; if (nid != NUMA_NO_NODE) - nr_available = h->free_huge_pages_node[nid]; - else - nr_available = h->free_huge_pages; - nr_available -= h->resv_huge_pages; + nr_available = min(nr_available, + h->free_huge_pages_node[nid]); if (!nr_available) break; - rc = demote_pool_huge_page(h, n_mask, nr_demote); + rc = demote_pool_huge_page(h, n_mask, + min(nr_demote, nr_available)); if (rc < 0) { err = rc; break; From 15e8b294f06c3c0b25b667d3743873ff97d584d4 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:23 +0530 Subject: [PATCH 0352/1012] mm: make ptval_to_str() generally available Patch series "mm: Drop pxd_ERROR()". pxd_ERROR() macros have been provided by all platforms, which are very much identical and can be dropped off completely if these pgtable printing could be moved to callers in generic MM aka all pxd_clear_bad(). But first cleanups and re-organizations are required in some platforms that are using these macros internally. Afterwards [pte|pmd|pud|p4d|pgd]_ERROR() macros have been completely dropped from the entire tree. This patch (of 8): Move ptval_to_str() inside a header thus making the helper more generally available for new users which are being added later. While here, also move another related string size macro PTVAL_STR_MAX inside the header as well. Link: https://lore.kernel.org/20260831054331.625505-1-anshuman.khandual@arm.com Link: https://lore.kernel.org/20260831054331.625505-2-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Lorenzo Stoakes Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: John Paul Adrian Glaubitz Cc: Lorenzo Stoakes Cc: Rich Felker Cc: Samuel Holland Cc: WANG Xuerui Cc: Yoshinori Sato Cc: Geert Uytterhoeven --- include/linux/pgtable.h | 14 ++++++++++++++ mm/memory.c | 15 +-------------- 2 files changed, 15 insertions(+), 14 deletions(-) diff --git a/include/linux/pgtable.h b/include/linux/pgtable.h index 8c093c119e5a82..e3c8ab96941c5e 100644 --- a/include/linux/pgtable.h +++ b/include/linux/pgtable.h @@ -2313,6 +2313,20 @@ static inline const char *pgtable_level_to_str(enum pgtable_level level) } } +void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size); + +#define ptval_to_str(buf, val) \ + do { \ + auto __val = (val); \ + \ + ptval_bytes_to_hex_str((buf), sizeof(buf), &__val, sizeof(__val)); \ + } while (0) + +#if defined(__SIZEOF_INT128__) +#define PTVAL_STR_MAX (32 + 1) /* Max 128-bit value in hex + NUL */ +#else +#define PTVAL_STR_MAX (16 + 1) /* Max 64-bit value in hex + NUL */ +#endif #endif /* !__ASSEMBLER__ */ #if !defined(MAX_POSSIBLE_PHYSMEM_BITS) && !defined(CONFIG_64BIT) diff --git a/mm/memory.c b/mm/memory.c index bc14cae3c49d72..ec63dd6212ac5a 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -495,7 +495,7 @@ static inline void add_mm_rss_vec(struct mm_struct *mm, int *rss) /* Allow a burst of 60 bad page map reports per minute. */ static DEFINE_RATELIMIT_STATE(bad_page_map_ratelimit, 60 * HZ, 60); -static void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size) +void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry, size_t entry_size) { if (WARN_ON_ONCE(buf_size < entry_size * 2 + 1)) { snprintf(buf, buf_size, "overflow"); @@ -522,19 +522,6 @@ static void ptval_bytes_to_hex_str(char *buf, size_t buf_size, const void *entry } } -#define ptval_to_str(buf, val) \ - do { \ - auto __val = (val); \ - \ - ptval_bytes_to_hex_str((buf), sizeof(buf), &__val, sizeof(__val)); \ - } while (0) - -#if defined(__SIZEOF_INT128__) -#define PTVAL_STR_MAX (32 + 1) /* Max 128-bit value in hex + NUL */ -#else -#define PTVAL_STR_MAX (16 + 1) /* Max 64-bit value in hex + NUL */ -#endif - static void __print_bad_page_map_pgtable(struct mm_struct *mm, unsigned long addr) { char pgd_str[PTVAL_STR_MAX]; From e482f6ec2252de20acc50d80e0c2fdf5f573534e Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:24 +0530 Subject: [PATCH 0353/1012] mm: stop using pxd_ERROR() pxd_ERROR() has been used in generic mm just to print the page table entry in pxd_clear_bad() before clearing those out with pxd_clear() later. These pxd_ERROR() macros have been provided by all platforms which basically did the same thing. Make pxd_clear_bad() use recently added ptval_to_str() instead for printing page table entries thus completely dropping dependency on platform provided pxd_ERROR() macros which can then be dropped off later on. Link: https://lore.kernel.org/20260831054331.625505-3-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Lorenzo Stoakes Cc: Geert Uytterhoeven Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: John Paul Adrian Glaubitz Cc: Rich Felker Cc: Samuel Holland Cc: WANG Xuerui Cc: Yoshinori Sato --- mm/pgtable-generic.c | 20 ++++++++++++++++---- 1 file changed, 16 insertions(+), 4 deletions(-) diff --git a/mm/pgtable-generic.c b/mm/pgtable-generic.c index cd227fc05d2d8f..b45e891d1193fd 100644 --- a/mm/pgtable-generic.c +++ b/mm/pgtable-generic.c @@ -26,14 +26,20 @@ void pgd_clear_bad(pgd_t *pgd) { - pgd_ERROR(*pgd); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, pgd_val(*pgd)); + pr_err("bad pgd %s.\n", str); pgd_clear(pgd); } #ifndef __PAGETABLE_P4D_FOLDED void p4d_clear_bad(p4d_t *p4d) { - p4d_ERROR(*p4d); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, p4d_val(*p4d)); + pr_err("bad p4d %s.\n", str); p4d_clear(p4d); } #endif @@ -41,7 +47,10 @@ void p4d_clear_bad(p4d_t *p4d) #ifndef __PAGETABLE_PUD_FOLDED void pud_clear_bad(pud_t *pud) { - pud_ERROR(*pud); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, pud_val(*pud)); + pr_err("bad pud %s.\n", str); pud_clear(pud); } #endif @@ -53,7 +62,10 @@ void pud_clear_bad(pud_t *pud) */ void pmd_clear_bad(pmd_t *pmd) { - pmd_ERROR(*pmd); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, pmd_val(*pmd)); + pr_err("bad pmd %s.\n", str); pmd_clear(pmd); } From 06d8b5d3cdada86bc829459a960517a02c392865 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:25 +0530 Subject: [PATCH 0354/1012] loongarch/mm: stop using pte_ERROR() Directly use pr_err() in __set_fixmap() and drop pte_ERROR() which helps in eventually dropping pte_ERROR() macro across the tree. In this new printing __FILE__ and __LINE__ has been dropped because they are always the same and don't really add any value. The new ptval_to_str() helper is being used for converting pgtable entry value into a string. The error message itself has been cleaned up as well. Link: https://lore.kernel.org/20260831054331.625505-4-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Huacai Chen Cc: WANG Xuerui Cc: Geert Uytterhoeven Cc: Helge Deller Cc: James Bottomley Cc: John Paul Adrian Glaubitz Cc: Lorenzo Stoakes Cc: Rich Felker Cc: Samuel Holland Cc: Yoshinori Sato --- arch/loongarch/include/asm/pgtable.h | 2 -- arch/loongarch/mm/init.c | 4 +++- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/arch/loongarch/include/asm/pgtable.h b/arch/loongarch/include/asm/pgtable.h index a05f6a4928dc68..eddd8906b77dfc 100644 --- a/arch/loongarch/include/asm/pgtable.h +++ b/arch/loongarch/include/asm/pgtable.h @@ -135,8 +135,6 @@ struct vm_area_struct; #define ptep_get(ptep) READ_ONCE(*(ptep)) #define pmdp_get(pmdp) READ_ONCE(*(pmdp)) -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %016lx.\n", __FILE__, __LINE__, pte_val(e)) #ifndef __PAGETABLE_PMD_FOLDED #define pmd_ERROR(e) \ pr_err("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) diff --git a/arch/loongarch/mm/init.c b/arch/loongarch/mm/init.c index 4b46c5d30708d8..f801f7097f0379 100644 --- a/arch/loongarch/mm/init.c +++ b/arch/loongarch/mm/init.c @@ -197,13 +197,15 @@ void __init __set_fixmap(enum fixed_addresses idx, phys_addr_t phys, pgprot_t flags) { unsigned long addr = __fix_to_virt(idx); + char str[PTVAL_STR_MAX]; pte_t *ptep; BUG_ON(idx <= FIX_HOLE || idx >= __end_of_fixed_addresses); ptep = populate_kernel_pte(addr); if (!pte_none(ptep_get(ptep))) { - pte_ERROR(*ptep); + ptval_to_str(str, pte_val(*ptep)); + pr_err("unexpected set PTE at %lx in %s: %s\n", addr, __func__, str); return; } From 97045c70a52031148cfc67abf1f4d7a6797f5d6a Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:26 +0530 Subject: [PATCH 0355/1012] parisc/mm: directly use generic [pmd|pgd]_clear_bad() Drop [pmd|pgd]_ERROR() followed by [pmd|pgd]_clear() instances. But instead directly use semantically equivalent generic helpers [pmd|pgd]_clear_bad() in unmap_uncached_[pte|pmd]() which helps in dropping their corresponding [pmd|pgd]_ERROR() macros across the tree. Link: https://lore.kernel.org/20260831054331.625505-5-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: James E.J. Bottomley Cc: Helge Deller Cc: Geert Uytterhoeven Cc: Huacai Chen Cc: John Paul Adrian Glaubitz Cc: Lorenzo Stoakes Cc: Rich Felker Cc: Samuel Holland Cc: WANG Xuerui Cc: Yoshinori Sato --- arch/parisc/kernel/pci-dma.c | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/arch/parisc/kernel/pci-dma.c b/arch/parisc/kernel/pci-dma.c index bf9f192c826ebe..84e7309a826696 100644 --- a/arch/parisc/kernel/pci-dma.c +++ b/arch/parisc/kernel/pci-dma.c @@ -160,8 +160,7 @@ static inline void unmap_uncached_pte(pmd_t * pmd, unsigned long vaddr, if (pmd_none(*pmd)) return; if (pmd_bad(*pmd)) { - pmd_ERROR(*pmd); - pmd_clear(pmd); + pmd_clear_bad(pmd); return; } pte = pte_offset_kernel(pmd, vaddr); @@ -196,8 +195,7 @@ static inline void unmap_uncached_pmd(pgd_t * dir, unsigned long vaddr, if (pgd_none(*dir)) return; if (pgd_bad(*dir)) { - pgd_ERROR(*dir); - pgd_clear(dir); + pgd_clear_bad(dir); return; } pmd = pmd_offset(pud_offset(p4d_offset(dir, vaddr), vaddr), vaddr); From 512adfb1370346f070a9e84cac9a9d3372a318cd Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:27 +0530 Subject: [PATCH 0356/1012] sh/mm: stop using pte_ERROR() Directly use pr_err() in set_pte_phys() and drop pte_ERROR() which helps in eventually dropping pte_ERROR() macro across the tree. In this new printing __FILE__ and __LINE__ has been dropped because they are always the same and don't really add any value. Besides ptrval_to_str() has been able to handle different PTE representation with and without CONFIG_X2TLB, which helped in unifying error message printing. Link: https://lore.kernel.org/20260831054331.625505-6-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Yoshinori Sato Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Geert Uytterhoeven Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: Lorenzo Stoakes Cc: Samuel Holland Cc: WANG Xuerui --- arch/sh/include/asm/pgtable_32.h | 5 ----- arch/sh/mm/init.c | 6 +++++- 2 files changed, 5 insertions(+), 6 deletions(-) diff --git a/arch/sh/include/asm/pgtable_32.h b/arch/sh/include/asm/pgtable_32.h index 5f51af18997b57..c8eb9a7a4c4c78 100644 --- a/arch/sh/include/asm/pgtable_32.h +++ b/arch/sh/include/asm/pgtable_32.h @@ -401,14 +401,9 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) #define pmd_page(pmd) (virt_to_page(pmd_val(pmd))) #ifdef CONFIG_X2TLB -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %p(%08lx%08lx).\n", __FILE__, __LINE__, \ - &(e), (e).pte_high, (e).pte_low) #define pgd_ERROR(e) \ printk("%s:%d: bad pgd %016llx.\n", __FILE__, __LINE__, pgd_val(e)) #else -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) #define pgd_ERROR(e) \ printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) #endif diff --git a/arch/sh/mm/init.c b/arch/sh/mm/init.c index 110308bdef01d0..9466ae6f9f164d 100644 --- a/arch/sh/mm/init.c +++ b/arch/sh/mm/init.c @@ -84,7 +84,11 @@ static void set_pte_phys(unsigned long addr, unsigned long phys, pgprot_t prot) pte = __get_pte_phys(addr); if (!pte_none(*pte)) { - pte_ERROR(*pte); + char str[PTVAL_STR_MAX]; + + ptval_to_str(str, pte_val(*pte)); + pr_err("unexpected set PTE at %lx in %s: bad pte %p(%s).\n", + addr, __func__, pte, str); return; } From fab9aac2da11410eada01141f68058aed3885c5d Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:28 +0530 Subject: [PATCH 0357/1012] sh/mm: stop using [p4d|pud|pmd]_ERROR() Stop using [p4d|pud|pmd]_ERROR() in __get_pte_phys() as the pgtable entries are known to be NULL and hence could not really be accessed. Link: https://lore.kernel.org/20260831054331.625505-7-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Yoshinori Sato Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Geert Uytterhoeven Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: Lorenzo Stoakes Cc: Samuel Holland Cc: WANG Xuerui --- arch/sh/mm/init.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/arch/sh/mm/init.c b/arch/sh/mm/init.c index 9466ae6f9f164d..8d65e60688dcbd 100644 --- a/arch/sh/mm/init.c +++ b/arch/sh/mm/init.c @@ -59,19 +59,19 @@ static pte_t *__get_pte_phys(unsigned long addr) p4d = p4d_alloc(NULL, pgd, addr); if (unlikely(!p4d)) { - p4d_ERROR(*p4d); + pr_err("allocating p4d table failed\n"); return NULL; } pud = pud_alloc(NULL, p4d, addr); if (unlikely(!pud)) { - pud_ERROR(*pud); + pr_err("allocating pud table failed\n"); return NULL; } pmd = pmd_alloc(NULL, pud, addr); if (unlikely(!pmd)) { - pmd_ERROR(*pmd); + pr_err("allocating pmd table failed\n"); return NULL; } From 310e8df92ee73b07578747b1ab396fc9449b0f9b Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:29 +0530 Subject: [PATCH 0358/1012] sh/mm: stop using pgd_ERROR() Stop using pgd_ERROR() in __get_pte_phys() when page table entry is already known to be empty. Link: https://lore.kernel.org/20260831054331.625505-8-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Yoshinori Sato Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Geert Uytterhoeven Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: Lorenzo Stoakes Cc: Samuel Holland Cc: WANG Xuerui --- arch/sh/mm/init.c | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/arch/sh/mm/init.c b/arch/sh/mm/init.c index 8d65e60688dcbd..93921109f4e64b 100644 --- a/arch/sh/mm/init.c +++ b/arch/sh/mm/init.c @@ -52,10 +52,8 @@ static pte_t *__get_pte_phys(unsigned long addr) pmd_t *pmd; pgd = pgd_offset_k(addr); - if (pgd_none(*pgd)) { - pgd_ERROR(*pgd); + if (pgd_none(*pgd)) return NULL; - } p4d = p4d_alloc(NULL, pgd, addr); if (unlikely(!p4d)) { From d5266c1c505ce23548f2ed783ae12d78ca74f7ee Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Mon, 31 Aug 2026 11:13:30 +0530 Subject: [PATCH 0359/1012] mm: drop pxd_ERROR() There are no more users left for any pxd_ERROR() either in generic MM or in the platform MM. Hence all these platform macros along with their generic fallback could be dropped across the tree. Link: https://lore.kernel.org/20260831054331.625505-9-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Acked-by: Geert Uytterhoeven # m68k Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Helge Deller Cc: Huacai Chen Cc: James Bottomley Cc: John Paul Adrian Glaubitz Cc: Lorenzo Stoakes Cc: Rich Felker Cc: Samuel Holland Cc: WANG Xuerui Cc: Yoshinori Sato --- arch/alpha/include/asm/pgtable.h | 7 ------- arch/arc/include/asm/pgtable-levels.h | 11 ----------- arch/arm/include/asm/pgtable.h | 7 ------- arch/arm/kernel/traps.c | 17 ----------------- arch/arm64/include/asm/pgtable.h | 15 --------------- arch/csky/include/asm/pgtable.h | 4 ---- arch/hexagon/include/asm/pgtable.h | 3 --- arch/loongarch/include/asm/pgtable.h | 11 ----------- arch/m68k/include/asm/mcf_pgtable.h | 6 ------ arch/m68k/include/asm/motorola_pgtable.h | 8 -------- arch/m68k/include/asm/sun3_pgtable.h | 7 ------- arch/microblaze/include/asm/pgtable.h | 7 ------- arch/mips/include/asm/pgtable-32.h | 10 ---------- arch/mips/include/asm/pgtable-64.h | 13 ------------- arch/nios2/include/asm/pgtable.h | 7 ------- arch/openrisc/include/asm/pgtable.h | 7 ------- arch/parisc/include/asm/pgtable.h | 9 --------- arch/powerpc/include/asm/book3s/32/pgtable.h | 2 -- arch/powerpc/include/asm/book3s/64/pgtable.h | 7 ------- arch/powerpc/include/asm/nohash/32/pgtable.h | 2 -- .../powerpc/include/asm/nohash/64/pgtable-4k.h | 3 --- arch/powerpc/include/asm/nohash/64/pgtable.h | 5 ----- arch/riscv/include/asm/page.h | 6 ------ arch/riscv/include/asm/pgtable-64.h | 9 --------- arch/riscv/include/asm/pgtable.h | 4 ---- arch/s390/include/asm/pgtable.h | 11 ----------- arch/sh/include/asm/pgtable-3level.h | 3 --- arch/sh/include/asm/pgtable_32.h | 8 -------- arch/sparc/include/asm/pgtable_32.h | 3 --- arch/sparc/include/asm/pgtable_64.h | 10 ---------- arch/um/include/asm/pgtable-2level.h | 7 ------- arch/um/include/asm/pgtable-4level.h | 13 ------------- arch/x86/include/asm/pgtable-2level.h | 5 ----- arch/x86/include/asm/pgtable-3level.h | 11 ----------- arch/x86/include/asm/pgtable_64.h | 18 ------------------ arch/xtensa/include/asm/pgtable.h | 4 ---- include/asm-generic/pgtable-nop4d.h | 1 - include/asm-generic/pgtable-nopmd.h | 1 - include/asm-generic/pgtable-nopud.h | 1 - 39 files changed, 283 deletions(-) diff --git a/arch/alpha/include/asm/pgtable.h b/arch/alpha/include/asm/pgtable.h index 8e00cf9dc39dea..7cac8241ee674c 100644 --- a/arch/alpha/include/asm/pgtable.h +++ b/arch/alpha/include/asm/pgtable.h @@ -357,13 +357,6 @@ static inline pte_t pte_swp_clear_exclusive(pte_t pte) return pte; } -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %016lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %016lx.\n", __FILE__, __LINE__, pgd_val(e)) - extern void paging_init(void); /* We have our own get_unmapped_area */ diff --git a/arch/arc/include/asm/pgtable-levels.h b/arch/arc/include/asm/pgtable-levels.h index c8f9273372c073..167b82fcfafe36 100644 --- a/arch/arc/include/asm/pgtable-levels.h +++ b/arch/arc/include/asm/pgtable-levels.h @@ -98,8 +98,6 @@ /* * 1st level paging: pgd */ -#define pgd_ERROR(e) \ - pr_crit("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) #if CONFIG_PGTABLE_LEVELS > 3 @@ -115,9 +113,6 @@ /* * 2nd level paging: pud */ -#define pud_ERROR(e) \ - pr_crit("%s:%d: bad pud %08lx.\n", __FILE__, __LINE__, pud_val(e)) - #endif #if CONFIG_PGTABLE_LEVELS > 2 @@ -137,9 +132,6 @@ /* * 3rd level paging: pmd */ -#define pmd_ERROR(e) \ - pr_crit("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, pmd_val(e)) - #define pmd_pfn(pmd) ((pmd_val(pmd) & PMD_MASK) >> PAGE_SHIFT) #define pfn_pmd(pfn,prot) __pmd(((pfn) << PAGE_SHIFT) | pgprot_val(prot)) @@ -165,9 +157,6 @@ /* * 4th level paging: pte */ -#define pte_ERROR(e) \ - pr_crit("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) - #define PFN_PTE_SHIFT PAGE_SHIFT #define pte_none(x) (!pte_val(x)) #define pte_present(x) (pte_val(x) & _PAGE_PRESENT) diff --git a/arch/arm/include/asm/pgtable.h b/arch/arm/include/asm/pgtable.h index 982795cf45637e..8dd17d20faa33f 100644 --- a/arch/arm/include/asm/pgtable.h +++ b/arch/arm/include/asm/pgtable.h @@ -44,13 +44,6 @@ #define LIBRARY_TEXT_START 0x0c000000 #ifndef __ASSEMBLY__ -extern void __pte_error(const char *file, int line, pte_t); -extern void __pmd_error(const char *file, int line, pmd_t); -extern void __pgd_error(const char *file, int line, pgd_t); - -#define pte_ERROR(pte) __pte_error(__FILE__, __LINE__, pte) -#define pmd_ERROR(pmd) __pmd_error(__FILE__, __LINE__, pmd) -#define pgd_ERROR(pgd) __pgd_error(__FILE__, __LINE__, pgd) /* * This is the lowest virtual address we can permit any user space diff --git a/arch/arm/kernel/traps.c b/arch/arm/kernel/traps.c index afbd2ebe5c39dc..ad04c806cc9d8f 100644 --- a/arch/arm/kernel/traps.c +++ b/arch/arm/kernel/traps.c @@ -753,23 +753,6 @@ void __readwrite_bug(const char *fn) } EXPORT_SYMBOL(__readwrite_bug); -#ifdef CONFIG_MMU -void __pte_error(const char *file, int line, pte_t pte) -{ - pr_err("%s:%d: bad pte %08llx.\n", file, line, (long long)pte_val(pte)); -} - -void __pmd_error(const char *file, int line, pmd_t pmd) -{ - pr_err("%s:%d: bad pmd %08llx.\n", file, line, (long long)pmd_val(pmd)); -} - -void __pgd_error(const char *file, int line, pgd_t pgd) -{ - pr_err("%s:%d: bad pgd %08llx.\n", file, line, (long long)pgd_val(pgd)); -} -#endif - asmlinkage void __div0(void) { pr_err("Division by zero in kernel.\n"); diff --git a/arch/arm64/include/asm/pgtable.h b/arch/arm64/include/asm/pgtable.h index 6000905a2e865e..e89ec5f4787b49 100644 --- a/arch/arm64/include/asm/pgtable.h +++ b/arch/arm64/include/asm/pgtable.h @@ -107,9 +107,6 @@ static inline void arch_leave_lazy_mmu_mode(void) __flush_tlb_range(vma, address, address + PMD_SIZE, PMD_SIZE, 2, \ TLBF_NOBROADCAST | TLBF_NONOTIFY | TLBF_NOWALKCACHE) -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %016llx.\n", __FILE__, __LINE__, pte_val(e)) - #ifdef CONFIG_ARM64_PA_BITS_52 static inline phys_addr_t __pte_to_phys(pte_t pte) { @@ -866,9 +863,6 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) #if CONFIG_PGTABLE_LEVELS > 2 -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %016llx.\n", __FILE__, __LINE__, pmd_val(e)) - #define pud_none(pud) (!pud_val(pud)) #define pud_bad(pud) ((pud_val(pud) & PUD_TYPE_MASK) != \ PUD_TYPE_TABLE) @@ -960,9 +954,6 @@ static inline bool mm_pud_folded(const struct mm_struct *mm) } #define mm_pud_folded mm_pud_folded -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %016llx.\n", __FILE__, __LINE__, pud_val(e)) - #define p4d_none(p4d) (pgtable_l4_enabled() && !p4d_val(p4d)) #define p4d_bad(p4d) (pgtable_l4_enabled() && \ ((p4d_val(p4d) & P4D_TYPE_MASK) != \ @@ -1088,9 +1079,6 @@ static inline bool mm_p4d_folded(const struct mm_struct *mm) } #define mm_p4d_folded mm_p4d_folded -#define p4d_ERROR(e) \ - pr_err("%s:%d: bad p4d %016llx.\n", __FILE__, __LINE__, p4d_val(e)) - #define pgd_none(pgd) (pgtable_l5_enabled() && !pgd_val(pgd)) #define pgd_bad(pgd) (pgtable_l5_enabled() && \ ((pgd_val(pgd) & PGD_TYPE_MASK) != \ @@ -1217,9 +1205,6 @@ p4d_t *p4d_offset_lockless_folded(pgd_t *pgdp, pgd_t pgd, unsigned long addr) #endif /* CONFIG_PGTABLE_LEVELS > 4 */ -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %016llx.\n", __FILE__, __LINE__, pgd_val(e)) - #define pgd_set_fixmap(addr) ((pgd_t *)set_fixmap_offset(FIX_PGD, addr)) #define pgd_clear_fixmap() clear_fixmap(FIX_PGD) diff --git a/arch/csky/include/asm/pgtable.h b/arch/csky/include/asm/pgtable.h index bafcd5823531a5..5ca77ff89ef939 100644 --- a/arch/csky/include/asm/pgtable.h +++ b/arch/csky/include/asm/pgtable.h @@ -23,10 +23,6 @@ #define PTRS_PER_PMD 1 #define PTRS_PER_PTE (PAGE_SIZE / sizeof(pte_t)) -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, (e).pte_low) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) #define PFN_PTE_SHIFT PAGE_SHIFT #define pmd_pfn(pmd) (pmd_phys(pmd) >> PAGE_SHIFT) diff --git a/arch/hexagon/include/asm/pgtable.h b/arch/hexagon/include/asm/pgtable.h index 27b269e2870d37..2fdb27afe70329 100644 --- a/arch/hexagon/include/asm/pgtable.h +++ b/arch/hexagon/include/asm/pgtable.h @@ -94,9 +94,6 @@ #endif /* Any bigger and the PTE disappears. */ -#define pgd_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__,\ - pgd_val(e)) /* * Page Protection Constants. Includes (in this variant) cache attributes. diff --git a/arch/loongarch/include/asm/pgtable.h b/arch/loongarch/include/asm/pgtable.h index eddd8906b77dfc..cf29a4c8ac593a 100644 --- a/arch/loongarch/include/asm/pgtable.h +++ b/arch/loongarch/include/asm/pgtable.h @@ -135,17 +135,6 @@ struct vm_area_struct; #define ptep_get(ptep) READ_ONCE(*(ptep)) #define pmdp_get(pmdp) READ_ONCE(*(pmdp)) -#ifndef __PAGETABLE_PMD_FOLDED -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) -#endif -#ifndef __PAGETABLE_PUD_FOLDED -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %016lx.\n", __FILE__, __LINE__, pud_val(e)) -#endif -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %016lx.\n", __FILE__, __LINE__, pgd_val(e)) - extern pte_t invalid_pte_table[PTRS_PER_PTE]; #ifndef __PAGETABLE_PUD_FOLDED diff --git a/arch/m68k/include/asm/mcf_pgtable.h b/arch/m68k/include/asm/mcf_pgtable.h index 189bb7b1e6630f..f45a882238dbb6 100644 --- a/arch/m68k/include/asm/mcf_pgtable.h +++ b/arch/m68k/include/asm/mcf_pgtable.h @@ -137,12 +137,6 @@ static inline int pmd_bad2(pmd_t *pmd) { return 0; } #define pmd_present(pmd) (!pmd_none2(&(pmd))) static inline void pmd_clear(pmd_t *pmdp) { pmd_val(*pmdp) = 0; } -#define pte_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pte %08lx.\n", \ - __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pgd %08lx.\n", \ - __FILE__, __LINE__, pgd_val(e)) /* * The following only work if pte_present() is true. diff --git a/arch/m68k/include/asm/motorola_pgtable.h b/arch/m68k/include/asm/motorola_pgtable.h index dcf6829b3eab97..d9393b310add6a 100644 --- a/arch/m68k/include/asm/motorola_pgtable.h +++ b/arch/m68k/include/asm/motorola_pgtable.h @@ -131,14 +131,6 @@ static inline void pud_set(pud_t *pudp, pmd_t *pmdp) #define pud_clear(pudp) ({ pud_val(*pudp) = 0; }) #define pud_page(pud) (mem_map + ((unsigned long)(__va(pud_val(pud)) - PAGE_OFFSET) >> PAGE_SHIFT)) -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - - /* * The following only work if pte_present() is true. * Undefined behaviour if not.. diff --git a/arch/m68k/include/asm/sun3_pgtable.h b/arch/m68k/include/asm/sun3_pgtable.h index 80ca185a18a193..704442a391fd74 100644 --- a/arch/m68k/include/asm/sun3_pgtable.h +++ b/arch/m68k/include/asm/sun3_pgtable.h @@ -119,13 +119,6 @@ static inline int pmd_present2 (pmd_t *pmd) { return pmd_val (*pmd) & SUN3_PMD_V #define pmd_present(pmd) (!pmd_none2(&(pmd))) static inline void pmd_clear (pmd_t *pmdp) { pmd_val (*pmdp) = 0; } - -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - - /* * The following only work if pte_present() is true. * Undefined behaviour if not... diff --git a/arch/microblaze/include/asm/pgtable.h b/arch/microblaze/include/asm/pgtable.h index 7678c040a2fd36..72708f9af1c0b8 100644 --- a/arch/microblaze/include/asm/pgtable.h +++ b/arch/microblaze/include/asm/pgtable.h @@ -103,13 +103,6 @@ extern pte_t *va_to_pte(unsigned long address); #define USER_PGD_PTRS (PAGE_OFFSET >> PGDIR_SHIFT) #define KERNEL_PGD_PTRS (PTRS_PER_PGD-USER_PGD_PTRS) -#define pte_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pte "PTE_FMT".\n", \ - __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pgd %08lx.\n", \ - __FILE__, __LINE__, pgd_val(e)) - /* * Bits in a linux-style PTE. These match the bits in the * (hardware-defined) PTE as closely as possible. diff --git a/arch/mips/include/asm/pgtable-32.h b/arch/mips/include/asm/pgtable-32.h index 92b7591aac2acd..ef1001ab09c5be 100644 --- a/arch/mips/include/asm/pgtable-32.h +++ b/arch/mips/include/asm/pgtable-32.h @@ -104,16 +104,6 @@ extern int add_temporary_entry(unsigned long entrylo0, unsigned long entrylo1, # define VMALLOC_END (FIXADDR_START-2*PAGE_SIZE) #endif -#ifdef CONFIG_PHYS_ADDR_T_64BIT -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %016Lx.\n", __FILE__, __LINE__, pte_val(e)) -#else -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#endif -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - extern void load_pgd(unsigned long pg_dir); extern pte_t invalid_pte_table[PTRS_PER_PTE]; diff --git a/arch/mips/include/asm/pgtable-64.h b/arch/mips/include/asm/pgtable-64.h index 6e854bb11f37de..785fc37bab9417 100644 --- a/arch/mips/include/asm/pgtable-64.h +++ b/arch/mips/include/asm/pgtable-64.h @@ -151,19 +151,6 @@ #define MODULES_END (FIXADDR_START-2*PAGE_SIZE) #endif -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %016lx.\n", __FILE__, __LINE__, pte_val(e)) -#ifndef __PAGETABLE_PMD_FOLDED -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) -#endif -#ifndef __PAGETABLE_PUD_FOLDED -#define pud_ERROR(e) \ - printk("%s:%d: bad pud %016lx.\n", __FILE__, __LINE__, pud_val(e)) -#endif -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %016lx.\n", __FILE__, __LINE__, pgd_val(e)) - extern pte_t invalid_pte_table[PTRS_PER_PTE]; #ifndef __PAGETABLE_PUD_FOLDED diff --git a/arch/nios2/include/asm/pgtable.h b/arch/nios2/include/asm/pgtable.h index d389aa9ca57ce0..272707d48f1bb1 100644 --- a/arch/nios2/include/asm/pgtable.h +++ b/arch/nios2/include/asm/pgtable.h @@ -223,13 +223,6 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) return pmd_val(pmd); } -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %08lx.\n", \ - __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", \ - __FILE__, __LINE__, pgd_val(e)) - /* * Encode/decode swap entries and swap PTEs. Swap PTEs are all PTEs that * are !pte_none() && !pte_present(). diff --git a/arch/openrisc/include/asm/pgtable.h b/arch/openrisc/include/asm/pgtable.h index 6b89996d0b628e..13afcc0bd8631b 100644 --- a/arch/openrisc/include/asm/pgtable.h +++ b/arch/openrisc/include/asm/pgtable.h @@ -338,13 +338,6 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) #define pte_pfn(x) ((unsigned long)(((x).pte)) >> PAGE_SHIFT) #define pfn_pte(pfn, prot) __pte((((pfn) << PAGE_SHIFT)) | pgprot_val(prot)) -#define pte_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pte %p(%08lx).\n", \ - __FILE__, __LINE__, &(e), pte_val(e)) -#define pgd_ERROR(e) \ - printk(KERN_ERR "%s:%d: bad pgd %p(%08lx).\n", \ - __FILE__, __LINE__, &(e), pgd_val(e)) - extern pgd_t swapper_pg_dir[PTRS_PER_PGD]; /* defined in head.S */ struct vm_area_struct; diff --git a/arch/parisc/include/asm/pgtable.h b/arch/parisc/include/asm/pgtable.h index 467b8547ac8bfa..f6899375cb4393 100644 --- a/arch/parisc/include/asm/pgtable.h +++ b/arch/parisc/include/asm/pgtable.h @@ -75,15 +75,6 @@ extern void __update_cache(pte_t pte); #endif /* !__ASSEMBLER__ */ -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#if CONFIG_PGTABLE_LEVELS == 3 -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, (unsigned long)pmd_val(e)) -#endif -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, (unsigned long)pgd_val(e)) - /* This is the size of the initially mapped kernel memory */ #if defined(CONFIG_64BIT) || defined(CONFIG_KALLSYMS) #define KERNEL_INITIAL_ORDER 26 /* 1<<26 = 64MB */ diff --git a/arch/powerpc/include/asm/book3s/32/pgtable.h b/arch/powerpc/include/asm/book3s/32/pgtable.h index e18a4fa282a1b6..835e84caee13fa 100644 --- a/arch/powerpc/include/asm/book3s/32/pgtable.h +++ b/arch/powerpc/include/asm/book3s/32/pgtable.h @@ -203,8 +203,6 @@ void unmap_kernel_page(unsigned long va); /* Bits to mask out from a PGD to get to the PUD page */ #define PGD_MASKED_BITS 0 -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) /* * Bits in a linux-style PTE. These match the bits in the * (hardware-defined) PowerPC PTE as closely as possible. diff --git a/arch/powerpc/include/asm/book3s/64/pgtable.h b/arch/powerpc/include/asm/book3s/64/pgtable.h index f4db7d7fbd5c62..dff8790a047db5 100644 --- a/arch/powerpc/include/asm/book3s/64/pgtable.h +++ b/arch/powerpc/include/asm/book3s/64/pgtable.h @@ -991,13 +991,6 @@ static inline pmd_t *pud_pgtable(pud_t pud) return (pmd_t *)__va(pud_val(pud) & ~PUD_MASKED_BITS); } -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %08lx.\n", __FILE__, __LINE__, pud_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - static inline int map_kernel_page(unsigned long ea, unsigned long pa, pgprot_t prot) { if (radix_enabled()) { diff --git a/arch/powerpc/include/asm/nohash/32/pgtable.h b/arch/powerpc/include/asm/nohash/32/pgtable.h index 496ecc65ac255a..f17afde89fa13d 100644 --- a/arch/powerpc/include/asm/nohash/32/pgtable.h +++ b/arch/powerpc/include/asm/nohash/32/pgtable.h @@ -51,8 +51,6 @@ #define USER_PTRS_PER_PGD (TASK_SIZE / PGDIR_SIZE) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08llx.\n", __FILE__, __LINE__, (unsigned long long)pgd_val(e)) /* * This is the bottom of the PKMAP area with HIGHMEM or an arbitrary diff --git a/arch/powerpc/include/asm/nohash/64/pgtable-4k.h b/arch/powerpc/include/asm/nohash/64/pgtable-4k.h index fb6fa1d4e0749a..75cf3c331b9292 100644 --- a/arch/powerpc/include/asm/nohash/64/pgtable-4k.h +++ b/arch/powerpc/include/asm/nohash/64/pgtable-4k.h @@ -82,9 +82,6 @@ extern struct page *p4d_page(p4d_t p4d); #endif /* !__ASSEMBLER__ */ -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %08lx.\n", __FILE__, __LINE__, pud_val(e)) - /* * On all 4K setups, remap_4k_pfn() equates to remap_pfn_range() */ #define remap_4k_pfn(vma, addr, pfn, prot) \ diff --git a/arch/powerpc/include/asm/nohash/64/pgtable.h b/arch/powerpc/include/asm/nohash/64/pgtable.h index 661eb3820d1291..446dde8b6ead54 100644 --- a/arch/powerpc/include/asm/nohash/64/pgtable.h +++ b/arch/powerpc/include/asm/nohash/64/pgtable.h @@ -159,11 +159,6 @@ static inline void huge_ptep_set_wrprotect(struct mm_struct *mm, __young; \ }) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %08lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) - /* * Encode/decode swap entries and swap PTEs. Swap PTEs are all PTEs that * are !pte_none() && !pte_present(). diff --git a/arch/riscv/include/asm/page.h b/arch/riscv/include/asm/page.h index 709a36fb432343..b4bbae55e93111 100644 --- a/arch/riscv/include/asm/page.h +++ b/arch/riscv/include/asm/page.h @@ -76,12 +76,6 @@ typedef struct page *pgtable_t; #define __pgd(x) ((pgd_t) { (x) }) #define __pgprot(x) ((pgprot_t) { (x) }) -#ifdef CONFIG_64BIT -#define PTE_FMT "%016lx" -#else -#define PTE_FMT "%08lx" -#endif - #if defined(CONFIG_64BIT) && defined(CONFIG_MMU) /* * We override this value as its generic definition uses __pa too early in diff --git a/arch/riscv/include/asm/pgtable-64.h b/arch/riscv/include/asm/pgtable-64.h index 6e789fa58514c7..ae23182b572cdd 100644 --- a/arch/riscv/include/asm/pgtable-64.h +++ b/arch/riscv/include/asm/pgtable-64.h @@ -264,15 +264,6 @@ static inline unsigned long _pmd_pfn(pmd_t pmd) return __page_val_to_pfn(pmd_val(pmd)); } -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) - -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %016lx.\n", __FILE__, __LINE__, pud_val(e)) - -#define p4d_ERROR(e) \ - pr_err("%s:%d: bad p4d %016lx.\n", __FILE__, __LINE__, p4d_val(e)) - static inline void set_p4d(p4d_t *p4dp, p4d_t p4d) { if (pgtable_l4_enabled) diff --git a/arch/riscv/include/asm/pgtable.h b/arch/riscv/include/asm/pgtable.h index 40b1ed4f3ea893..4c8fc684550311 100644 --- a/arch/riscv/include/asm/pgtable.h +++ b/arch/riscv/include/asm/pgtable.h @@ -556,10 +556,6 @@ static inline pte_t pte_modify(pte_t pte, pgprot_t newprot) return __pte((pte_val(pte) & _PAGE_CHG_MASK) | newprot_val); } -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd " PTE_FMT ".\n", __FILE__, __LINE__, pgd_val(e)) - - /* Commit new configuration to MMU hardware */ static inline void update_mmu_cache_range(struct vm_fault *vmf, struct vm_area_struct *vma, unsigned long address, diff --git a/arch/s390/include/asm/pgtable.h b/arch/s390/include/asm/pgtable.h index e882663a58e776..2d5c2ab06de988 100644 --- a/arch/s390/include/asm/pgtable.h +++ b/arch/s390/include/asm/pgtable.h @@ -68,17 +68,6 @@ extern unsigned long zero_page_mask; /* TODO: s390 cannot support io_remap_pfn_range... */ -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %016lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %016lx.\n", __FILE__, __LINE__, pmd_val(e)) -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %016lx.\n", __FILE__, __LINE__, pud_val(e)) -#define p4d_ERROR(e) \ - pr_err("%s:%d: bad p4d %016lx.\n", __FILE__, __LINE__, p4d_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %016lx.\n", __FILE__, __LINE__, pgd_val(e)) - /* * The vmalloc and module area will always be on the topmost area of the * kernel mapping. 512GB are reserved for vmalloc by default. diff --git a/arch/sh/include/asm/pgtable-3level.h b/arch/sh/include/asm/pgtable-3level.h index d1ce73f3bd85ef..3f4d747f30f40b 100644 --- a/arch/sh/include/asm/pgtable-3level.h +++ b/arch/sh/include/asm/pgtable-3level.h @@ -25,9 +25,6 @@ #define PTRS_PER_PMD ((1 << PGDIR_SHIFT) / PMD_SIZE) -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %016llx.\n", __FILE__, __LINE__, pmd_val(e)) - typedef union { struct { unsigned long pmd_low; diff --git a/arch/sh/include/asm/pgtable_32.h b/arch/sh/include/asm/pgtable_32.h index c8eb9a7a4c4c78..cde1bf0c67342b 100644 --- a/arch/sh/include/asm/pgtable_32.h +++ b/arch/sh/include/asm/pgtable_32.h @@ -400,14 +400,6 @@ static inline unsigned long pmd_page_vaddr(pmd_t pmd) #define pmd_pfn(pmd) (__pa(pmd_val(pmd)) >> PAGE_SHIFT) #define pmd_page(pmd) (virt_to_page(pmd_val(pmd))) -#ifdef CONFIG_X2TLB -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %016llx.\n", __FILE__, __LINE__, pgd_val(e)) -#else -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %08lx.\n", __FILE__, __LINE__, pgd_val(e)) -#endif - /* * Encode/decode swap entries and swap PTEs. Swap PTEs are all PTEs that * are !pte_none() && !pte_present(). diff --git a/arch/sparc/include/asm/pgtable_32.h b/arch/sparc/include/asm/pgtable_32.h index f89b1250661dff..5a5f54a090f5b2 100644 --- a/arch/sparc/include/asm/pgtable_32.h +++ b/arch/sparc/include/asm/pgtable_32.h @@ -40,9 +40,6 @@ void load_mmu(void); unsigned long calc_highpages(void); unsigned long __init bootmem_init(unsigned long *pages_avail); -#define pte_ERROR(e) __builtin_trap() -#define pmd_ERROR(e) __builtin_trap() -#define pgd_ERROR(e) __builtin_trap() #define PTRS_PER_PTE 64 #define PTRS_PER_PMD 64 diff --git a/arch/sparc/include/asm/pgtable_64.h b/arch/sparc/include/asm/pgtable_64.h index 0837ebbc5dce63..44d1333065a6ae 100644 --- a/arch/sparc/include/asm/pgtable_64.h +++ b/arch/sparc/include/asm/pgtable_64.h @@ -96,16 +96,6 @@ bool kern_addr_valid(unsigned long addr); #define PTRS_PER_PUD (1UL << PUD_BITS) #define PTRS_PER_PGD (1UL << PGDIR_BITS) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %p(%016lx) seen at (%pS)\n", \ - __FILE__, __LINE__, &(e), pmd_val(e), __builtin_return_address(0)) -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %p(%016lx) seen at (%pS)\n", \ - __FILE__, __LINE__, &(e), pud_val(e), __builtin_return_address(0)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %p(%016lx) seen at (%pS)\n", \ - __FILE__, __LINE__, &(e), pgd_val(e), __builtin_return_address(0)) - #endif /* !(__ASSEMBLER__) */ /* PTE bits which are the same in SUN4U and SUN4V format. */ diff --git a/arch/um/include/asm/pgtable-2level.h b/arch/um/include/asm/pgtable-2level.h index 14ec16f92ce408..fa625f5b5ef750 100644 --- a/arch/um/include/asm/pgtable-2level.h +++ b/arch/um/include/asm/pgtable-2level.h @@ -24,13 +24,6 @@ #define USER_PTRS_PER_PGD ((TASK_SIZE + (PGDIR_SIZE - 1)) / PGDIR_SIZE) #define PTRS_PER_PGD 1024 -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %p(%08lx).\n", __FILE__, __LINE__, &(e), \ - pte_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %p(%08lx).\n", __FILE__, __LINE__, &(e), \ - pgd_val(e)) - static inline int pgd_needsync(pgd_t pgd) { return 0; } static inline void pgd_mkuptodate(pgd_t pgd) { } diff --git a/arch/um/include/asm/pgtable-4level.h b/arch/um/include/asm/pgtable-4level.h index 7a271b7b83d2bd..ff82f99c80fa10 100644 --- a/arch/um/include/asm/pgtable-4level.h +++ b/arch/um/include/asm/pgtable-4level.h @@ -42,19 +42,6 @@ #define USER_PTRS_PER_PGD ((TASK_SIZE + (PGDIR_SIZE - 1)) / PGDIR_SIZE) -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %p(%016lx).\n", __FILE__, __LINE__, &(e), \ - pte_val(e)) -#define pmd_ERROR(e) \ - printk("%s:%d: bad pmd %p(%016lx).\n", __FILE__, __LINE__, &(e), \ - pmd_val(e)) -#define pud_ERROR(e) \ - printk("%s:%d: bad pud %p(%016lx).\n", __FILE__, __LINE__, &(e), \ - pud_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd %p(%016lx).\n", __FILE__, __LINE__, &(e), \ - pgd_val(e)) - #define pud_none(x) (!(pud_val(x) & ~_PAGE_NEEDSYNC)) #define pud_bad(x) ((pud_val(x) & (~PAGE_MASK & ~_PAGE_USER)) != _KERNPG_TABLE) #define pud_present(x) (pud_val(x) & _PAGE_PRESENT) diff --git a/arch/x86/include/asm/pgtable-2level.h b/arch/x86/include/asm/pgtable-2level.h index e9482a11ac52d6..83427765cfbfd2 100644 --- a/arch/x86/include/asm/pgtable-2level.h +++ b/arch/x86/include/asm/pgtable-2level.h @@ -2,11 +2,6 @@ #ifndef _ASM_X86_PGTABLE_2LEVEL_H #define _ASM_X86_PGTABLE_2LEVEL_H -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %08lx\n", __FILE__, __LINE__, (e).pte_low) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %08lx\n", __FILE__, __LINE__, pgd_val(e)) - /* * Certain architectures need to do special things when PTEs * within a page table are directly modified. Thus, the following diff --git a/arch/x86/include/asm/pgtable-3level.h b/arch/x86/include/asm/pgtable-3level.h index dabafba957ea6f..d6729911e09a28 100644 --- a/arch/x86/include/asm/pgtable-3level.h +++ b/arch/x86/include/asm/pgtable-3level.h @@ -8,17 +8,6 @@ * * Copyright (C) 1999 Ingo Molnar */ - -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %p(%08lx%08lx)\n", \ - __FILE__, __LINE__, &(e), (e).pte_high, (e).pte_low) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %p(%016Lx)\n", \ - __FILE__, __LINE__, &(e), pmd_val(e)) -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %p(%016Lx)\n", \ - __FILE__, __LINE__, &(e), pgd_val(e)) - #define pxx_xchg64(_pxx, _ptr, _val) ({ \ _pxx##val_t *_p = (_pxx##val_t *)_ptr; \ _pxx##val_t _o = *_p; \ diff --git a/arch/x86/include/asm/pgtable_64.h b/arch/x86/include/asm/pgtable_64.h index ce45882ccd071b..c861f3832bed2e 100644 --- a/arch/x86/include/asm/pgtable_64.h +++ b/arch/x86/include/asm/pgtable_64.h @@ -29,24 +29,6 @@ extern pgd_t init_top_pgt[]; extern void paging_init(void); static inline void sync_initial_page_table(void) { } -#define pte_ERROR(e) \ - pr_err("%s:%d: bad pte %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), pte_val(e)) -#define pmd_ERROR(e) \ - pr_err("%s:%d: bad pmd %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), pmd_val(e)) -#define pud_ERROR(e) \ - pr_err("%s:%d: bad pud %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), pud_val(e)) - -#define p4d_ERROR(e) \ - pr_err("%s:%d: bad p4d %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), p4d_val(e)) - -#define pgd_ERROR(e) \ - pr_err("%s:%d: bad pgd %p(%016lx)\n", \ - __FILE__, __LINE__, &(e), pgd_val(e)) - struct mm_struct; #define mm_p4d_folded mm_p4d_folded diff --git a/arch/xtensa/include/asm/pgtable.h b/arch/xtensa/include/asm/pgtable.h index f00a879dc298a5..60fb67a9972907 100644 --- a/arch/xtensa/include/asm/pgtable.h +++ b/arch/xtensa/include/asm/pgtable.h @@ -204,10 +204,6 @@ */ #ifndef __ASSEMBLER__ -#define pte_ERROR(e) \ - printk("%s:%d: bad pte %08lx.\n", __FILE__, __LINE__, pte_val(e)) -#define pgd_ERROR(e) \ - printk("%s:%d: bad pgd entry %08lx.\n", __FILE__, __LINE__, pgd_val(e)) #ifdef CONFIG_MMU extern pgd_t swapper_pg_dir[PAGE_SIZE/sizeof(pgd_t)]; diff --git a/include/asm-generic/pgtable-nop4d.h b/include/asm-generic/pgtable-nop4d.h index 89c21f84cffbe2..1cf739ee38aa15 100644 --- a/include/asm-generic/pgtable-nop4d.h +++ b/include/asm-generic/pgtable-nop4d.h @@ -22,7 +22,6 @@ static inline int pgd_none(pgd_t pgd) { return 0; } static inline int pgd_bad(pgd_t pgd) { return 0; } static inline int pgd_present(pgd_t pgd) { return 1; } static inline void pgd_clear(pgd_t *pgd) { } -#define p4d_ERROR(p4d) (pgd_ERROR((p4d).pgd)) #define pgd_populate(mm, pgd, p4d) do { } while (0) #define pgd_populate_safe(mm, pgd, p4d) do { } while (0) diff --git a/include/asm-generic/pgtable-nopmd.h b/include/asm-generic/pgtable-nopmd.h index 36b6490ed18081..ff4235cf84d772 100644 --- a/include/asm-generic/pgtable-nopmd.h +++ b/include/asm-generic/pgtable-nopmd.h @@ -33,7 +33,6 @@ static inline int pud_present(pud_t pud) { return 1; } static inline int pud_user(pud_t pud) { return 0; } static inline int pud_leaf(pud_t pud) { return 0; } static inline void pud_clear(pud_t *pud) { } -#define pmd_ERROR(pmd) (pud_ERROR((pmd).pud)) #define pud_populate(mm, pmd, pte) do { } while (0) diff --git a/include/asm-generic/pgtable-nopud.h b/include/asm-generic/pgtable-nopud.h index 356cbfbaab2476..eedee8e3ad68fd 100644 --- a/include/asm-generic/pgtable-nopud.h +++ b/include/asm-generic/pgtable-nopud.h @@ -29,7 +29,6 @@ static inline int p4d_none(p4d_t p4d) { return 0; } static inline int p4d_bad(p4d_t p4d) { return 0; } static inline int p4d_present(p4d_t p4d) { return 1; } static inline void p4d_clear(p4d_t *p4d) { } -#define pud_ERROR(pud) (p4d_ERROR((pud).p4d)) #define p4d_populate(mm, p4d, pud) do { } while (0) #define p4d_populate_safe(mm, p4d, pud) do { } while (0) From 552dd6e9534502dbedcd302e1c959fdc1dc85ea5 Mon Sep 17 00:00:00 2001 From: Johannes Weiner Date: Sun, 30 Aug 2026 12:29:17 +0800 Subject: [PATCH 0360/1012] mm: add page_counter_margin() Patch series "mm: avoid large folio splits when swap is unavailable", v7. This is v7 of Barry's original RFC patch, "mm: Avoiding split large folios if swap has no space": https://lore.kernel.org/r/20260618221720.71768-1-baohua@kernel.org Barry's RFC showed the no-swap case with MADV_PAGEOUT on 16KB mTHP: the large-folio split counter increased by 1024 even though no swapout progress was possible. Skipping the split in that case kept the counter at 0. This series makes folio_alloc_swap() classify failures according to whether splitting a large folio might allow swapout to make progress. Callers can then avoid destroying the large folio when neither global swap availability nor the folio's memcg swap hierarchy has capacity for even a smaller folio. Patch #1 adds page_counter_margin(), a small helper that computes the minimum remaining chargeable space across a page_counter hierarchy. Patch #2 establishes the folio_alloc_swap() return-value contract: - -E2BIG: splitting may let smaller folios make progress - -ENOSPC: no global swap space is available - -ENOMEM: splitting is not expected to help, including when the folio's memcg swap hierarchy has no remaining capacity Patch #3 makes vmscan split a large folio only when folio_alloc_swap() returns -E2BIG. Other failures keep the existing activation path and avoid destroying the large folio when no smaller part can be backed by swap either. Patch #4 applies the same contract to shmem_writeout(), which currently splits a large folio on every folio_alloc_swap() failure. It now enters the split fallback only on -E2BIG; other failures redirty and reactivate the folio as before. Testing: With a 1GB anonymous mapping backed by 16KB mTHPs and memory.swap.max=0, the patch reduced the median latency of 30 process_madvise(MADV_PAGEOUT) runs from 743.8 ms to 181.7 ms, while the number of large-folio splits per run dropped from 65536 to 0. Neither kernel swapped out any pages. I also ran DaCapo h2 under swap pressure and found no statistically significant change in wall time or CPU time. The overall benefit appears minor and workload-dependent. This patch (of 4): mem_cgroup_get_nr_swap_pages() open-codes the remaining capacity across the memcg swap counter hierarchy. Add page_counter_margin() to return the minimum usable space from a page counter to the root, and use it in mem_cgroup_get_nr_swap_pages(). This is a pure refactoring with no intended behavior change. Link: https://lore.kernel.org/20260830042920.2280454-1-xueyuan.chen21@gmail.com Link: https://lore.kernel.org/20260830042920.2280454-2-xueyuan.chen21@gmail.com Signed-off-by: Johannes Weiner Signed-off-by: Xueyuan Chen Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Barry Song Cc: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: Hugh Dickins Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Nhat Pham Cc: Roman Gushchin Cc: Shakeel Butt Cc: Nanzhe Zhao Cc: Youngjun Park --- include/linux/page_counter.h | 1 + mm/memcontrol.c | 9 +++------ mm/page_counter.c | 20 ++++++++++++++++++++ 3 files changed, 24 insertions(+), 6 deletions(-) diff --git a/include/linux/page_counter.h b/include/linux/page_counter.h index d649b6bbbc871b..07b7cb12249c7c 100644 --- a/include/linux/page_counter.h +++ b/include/linux/page_counter.h @@ -68,6 +68,7 @@ static inline unsigned long page_counter_read(struct page_counter *counter) return atomic_long_read(&counter->usage); } +long page_counter_margin(struct page_counter *counter); void page_counter_cancel(struct page_counter *counter, unsigned long nr_pages); void page_counter_charge(struct page_counter *counter, unsigned long nr_pages); bool page_counter_try_charge(struct page_counter *counter, diff --git a/mm/memcontrol.c b/mm/memcontrol.c index aeaa09e01d70ea..7f63bf9e8ef140 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5832,12 +5832,9 @@ long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) { long nr_swap_pages = get_nr_swap_pages(); - if (mem_cgroup_disabled() || do_memsw_account()) - return nr_swap_pages; - for (; !mem_cgroup_is_root(memcg); memcg = parent_mem_cgroup(memcg)) - nr_swap_pages = min_t(long, nr_swap_pages, - READ_ONCE(memcg->swap.max) - - page_counter_read(&memcg->swap)); + if (!mem_cgroup_disabled() && !do_memsw_account()) + nr_swap_pages = min(nr_swap_pages, page_counter_margin(&memcg->swap)); + return nr_swap_pages; } diff --git a/mm/page_counter.c b/mm/page_counter.c index 661e0f2a5127a5..450543f4b318b6 100644 --- a/mm/page_counter.c +++ b/mm/page_counter.c @@ -46,6 +46,26 @@ static void propagate_protected_usage(struct page_counter *c, } } +/** + * page_counter_margin - remaining usable space within hierarchical limits + * @counter: counter + * + * Return: The minimum value of max minus usage across @counter and all of + * its ancestors. The value may be negative during a concurrent charge. + */ +long page_counter_margin(struct page_counter *counter) +{ + long margin = PAGE_COUNTER_MAX; + + do { + long m = READ_ONCE(counter->max) - page_counter_read(counter); + + margin = min(margin, m); + } while ((counter = counter->parent)); + + return margin; +} + /** * page_counter_cancel - take pages out of the local counter * @counter: counter From 37b8f80716ecea66a121da9de73e1f9b72c5c44a Mon Sep 17 00:00:00 2001 From: Xueyuan Chen Date: Sun, 30 Aug 2026 12:29:18 +0800 Subject: [PATCH 0361/1012] mm: distinguish large folio swap allocation failures folio_alloc_swap() reports most failures with generic negative error codes. Reclaim callers consequently cannot tell whether splitting a large folio could make progress, or whether no swap space is available for even a single page. Classify failures using both the global free swap count and the remaining capacity in the folio's memcg swap hierarchy. Return -ENOSPC when global swap space is exhausted, -ENOMEM when splitting cannot overcome the failure, and -E2BIG for a large folio when allocating or charging a smaller folio might still succeed. Use this classification for all folio_alloc_swap() failure paths, including capability rejection, swap slot allocation failure, and memcg swap charge failure. Callers are updated separately to split large folios only on -E2BIG. Link: https://lore.kernel.org/20260830042920.2280454-3-xueyuan.chen21@gmail.com Signed-off-by: Xueyuan Chen Signed-off-by: Andrew Morton Suggested-by: Kairui Song Suggested-by: Barry Song Suggested-by: Youngjun Park Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: Hugh Dickins Cc: Johannes Weiner Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Nanzhe Zhao Cc: Nhat Pham Cc: Roman Gushchin Cc: Shakeel Butt --- include/linux/swap.h | 6 ++++++ mm/memcontrol.c | 23 +++++++++++++++++++++++ mm/swapfile.c | 26 +++++++++++++++++++------- 3 files changed, 48 insertions(+), 7 deletions(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index 5658a1634b85ea..7a43409879caed 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -508,6 +508,7 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_p __mem_cgroup_uncharge_swap(id, nr_pages); } +long mem_cgroup_get_folio_swap_margin(struct folio *folio); extern long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg); extern bool mem_cgroup_swap_full(struct folio *folio); #else @@ -521,6 +522,11 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, { } +static inline long mem_cgroup_get_folio_swap_margin(struct folio *folio) +{ + return PAGE_COUNTER_MAX; +} + static inline long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) { return get_nr_swap_pages(); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 7f63bf9e8ef140..bd1e7e15442659 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5838,6 +5838,29 @@ long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) return nr_swap_pages; } +/** + * mem_cgroup_get_folio_swap_margin - get a folio's memcg swap margin + * @folio: folio whose memcg margin is queried + * + * Return: Remaining chargeable pages in the folio's memcg hierarchy. + */ +long mem_cgroup_get_folio_swap_margin(struct folio *folio) +{ + struct mem_cgroup *memcg; + long margin; + + if (mem_cgroup_disabled() || do_memsw_account() || + !folio_memcg_charged(folio)) + return PAGE_COUNTER_MAX; + + rcu_read_lock(); + memcg = folio_memcg(folio); + margin = page_counter_margin(&memcg->swap); + rcu_read_unlock(); + + return margin; +} + bool mem_cgroup_swap_full(struct folio *folio) { struct mem_cgroup *memcg; diff --git a/mm/swapfile.c b/mm/swapfile.c index 408f6c72fb5a69..01e7b6b046b67d 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1735,7 +1735,9 @@ static int swap_dup_entries_cluster(struct swap_info_struct *si, * swap cache. * * Context: Caller needs to hold the folio lock. - * Return: Whether the folio was added to the swap cache. + * Return: %0 on success, %-E2BIG if splitting the folio might allow swapout, + * %-ENOSPC if no global swap space is available, or %-ENOMEM if splitting + * would not help. */ int folio_alloc_swap(struct folio *folio) { @@ -1747,11 +1749,11 @@ int folio_alloc_swap(struct folio *folio) if (order) { /* - * Reject large allocation when THP_SWAP is disabled, - * the caller should split the folio and try again. + * Reject large allocation when THP_SWAP is disabled. Check below + * whether splitting and retrying can make progress. */ if (!IS_ENABLED(CONFIG_THP_SWAP)) - return -EAGAIN; + goto failed; /* * Allocation size should never exceed cluster size @@ -1759,7 +1761,7 @@ int folio_alloc_swap(struct folio *folio) */ if (size > SWAPFILE_CLUSTER) { VM_WARN_ON_ONCE(1); - return -EINVAL; + goto failed; } } @@ -1775,13 +1777,23 @@ int folio_alloc_swap(struct folio *folio) } /* Need to call this even if allocation failed, for MEMCG_SWAP_FAIL. */ - if (unlikely(mem_cgroup_try_charge_swap(folio))) + if (unlikely(mem_cgroup_try_charge_swap(folio))) { swap_cache_del_folio(folio); + goto failed; + } if (unlikely(!folio_test_swapcache(folio))) - return -ENOMEM; + goto failed; return 0; + +failed: + if (get_nr_swap_pages() <= 0) + return -ENOSPC; + if (mem_cgroup_get_folio_swap_margin(folio) <= 0) + return -ENOMEM; + + return order ? -E2BIG : -ENOMEM; } /** From e5eb1781c8f48cef95595169ad2f70326490d16b Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Sun, 30 Aug 2026 12:29:19 +0800 Subject: [PATCH 0362/1012] mm/vmscan: avoid pointless large folio splits without swap When swap is disabled, exhausted, or unavailable due to memcg swap limits, splitting a large anonymous folio cannot make swapout progress. The fallback only destroys the large folio and inflates split statistics. Use -E2BIG from folio_alloc_swap() as the explicit signal that splitting the folio might allow swapout of smaller pieces. For other allocation failures, keep the existing activation path and avoid the split. This preserves the split fallback for fragmented or partially available swap, while avoiding it when there is no backing space for any part of the folio. Link: https://lore.kernel.org/20260830042920.2280454-4-xueyuan.chen21@gmail.com Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Xueyuan Chen Signed-off-by: Andrew Morton Reported-by: Nanzhe Zhao Acked-by: David Hildenbrand (Arm) Reviewed-by: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: Hugh Dickins Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Nhat Pham Cc: Roman Gushchin Cc: Shakeel Butt Cc: Youngjun Park --- mm/vmscan.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 42fcdcd3d2e49d..ce3bab78af3cd1 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -1259,6 +1259,8 @@ static unsigned int shrink_folio_list(struct list_head *folio_list, */ if (folio_test_anon(folio) && folio_test_swapbacked(folio) && !folio_test_swapcache(folio)) { + int ret; + if (!(sc->gfp_mask & __GFP_IO)) goto keep_locked; if (folio_maybe_dma_pinned(folio)) @@ -1277,11 +1279,14 @@ static unsigned int shrink_folio_list(struct list_head *folio_list, split_folio_to_list(folio, folio_list)) goto activate_locked; } - if (folio_alloc_swap(folio)) { + ret = folio_alloc_swap(folio); + if (ret) { int __maybe_unused order = folio_order(folio); if (!folio_test_large(folio)) goto activate_locked_split; + if (ret != -E2BIG) + goto activate_locked; /* Fallback to swap normal pages */ if (split_folio_to_list(folio, folio_list)) goto activate_locked; From 44774e2aaab5deb4dd8a059ae51e71aeaa1170b3 Mon Sep 17 00:00:00 2001 From: Xueyuan Chen Date: Sun, 30 Aug 2026 12:29:20 +0800 Subject: [PATCH 0363/1012] mm/shmem: split large folios only on -E2BIG shmem_writeout() currently splits a large folio on every folio_alloc_swap() failure. With the refined return-value contract, only -E2BIG indicates that splitting might allow smaller folios to be swapped out. Enter the split fallback only for -E2BIG. For -ENOSPC and -ENOMEM, redirty and reactivate the folio as before. Link: https://lore.kernel.org/20260830042920.2280454-5-xueyuan.chen21@gmail.com Signed-off-by: Xueyuan Chen Signed-off-by: Andrew Morton Suggested-by: Baolin Wang Reviewed-by: Baolin Wang Acked-by: David Hildenbrand (Arm) Reviewed-by: Barry Song Cc: Baoquan He Cc: Chris Li Cc: Hugh Dickins Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Nanzhe Zhao Cc: Nhat Pham Cc: Roman Gushchin Cc: Shakeel Butt Cc: Youngjun Park --- mm/shmem.c | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/mm/shmem.c b/mm/shmem.c index d3f24b5977bd5f..84f0a2eb85fecd 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -1813,7 +1813,7 @@ int shmem_writeout(struct swap_io_ctx *ctx, struct folio *folio, struct shmem_inode_info *info = SHMEM_I(inode); struct shmem_sb_info *sbinfo = SHMEM_SB(inode->i_sb); pgoff_t index; - int nr_pages; + int nr_pages, ret; bool split = false; if ((info->flags & SHMEM_F_LOCKED) || sbinfo->noswap) @@ -1894,7 +1894,8 @@ int shmem_writeout(struct swap_io_ctx *ctx, struct folio *folio, folio_mark_uptodate(folio); } - if (!folio_alloc_swap(folio)) { + ret = folio_alloc_swap(folio); + if (!ret) { bool first_swapped = shmem_recalc_inode(inode, 0, nr_pages); int error; @@ -1947,7 +1948,7 @@ int shmem_writeout(struct swap_io_ctx *ctx, struct folio *folio, swap_cache_del_folio(folio); goto redirty; } - if (nr_pages > 1) + if (nr_pages > 1 && ret == -E2BIG) goto try_split; redirty: folio_mark_dirty(folio); From d93e815ea34493fc4fb622297eacc15ca4993dce Mon Sep 17 00:00:00 2001 From: Aristeu Rozanski Date: Thu, 27 Aug 2026 21:55:43 -0400 Subject: [PATCH 0364/1012] mm: gup: move pmd_protnone() into gup_fast_pmd_leaf() Patch series "mm: gup: cleanup gup_fast call chain", v3. These two patches implement the refactor in gup_fast call chain David Hildenbrand mentioned in [1]. This patch (of 2): Make pmd handling match pud handling by calling pmd_protnone() inside gup_fast_pmd_leaf(). Link: https://lore.kernel.org/20260828015542.125576330@ruivo.org Link: https://lore.kernel.org/20260828015542.245315718@ruivo.org Link: https://lore.kernel.org/all/85e760cf-b994-40db-8d13-221feee55c60@redhat.com/T/#u [1] Signed-off-by: Aristeu Rozanski Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand Link: https://lore.kernel.org/all/85e760cf-b994-40db-8d13-221feee55c60@redhat.com/T/#u Acked-by: David Hildenbrand (Arm) Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu --- mm/gup.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/mm/gup.c b/mm/gup.c index eb898ea1ee22e5..bd981ab3975143 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -2929,6 +2929,10 @@ static int gup_fast_pmd_leaf(pmd_t orig, pmd_t *pmdp, unsigned long addr, struct folio *folio; int refs; + /* See gup_fast_pte_range() */ + if (pmd_protnone(orig)) + return 0; + if (!pmd_access_permitted(orig, flags & FOLL_WRITE)) return 0; @@ -3024,10 +3028,6 @@ static int gup_fast_pmd_range(pud_t *pudp, pud_t pud, unsigned long addr, return 0; if (unlikely(pmd_leaf(pmd))) { - /* See gup_fast_pte_range() */ - if (pmd_protnone(pmd)) - return 0; - if (!gup_fast_pmd_leaf(pmd, pmdp, addr, next, flags, pages, nr)) return 0; From 55d6966edb4f4b0bf2015a1119d7b9b84c0d4484 Mon Sep 17 00:00:00 2001 From: Aristeu Rozanski Date: Thu, 27 Aug 2026 21:55:44 -0400 Subject: [PATCH 0365/1012] mm: gup: cleanup the gup_fast_*() call chain Refactor gup_fast functions so each step of the way returns the number of pages pinned. Because the previous step of the chain knows what the number it should be, less indicates an error. This way there's no need to pass *nr along. Link: https://lore.kernel.org/20260828015542.334186653@ruivo.org Signed-off-by: Aristeu Rozanski Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand Link: https://lore.kernel.org/all/85e760cf-b994-40db-8d13-221feee55c60@redhat.com/T/#u Acked-by: David Hildenbrand (Arm) Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu --- mm/gup.c | 179 +++++++++++++++++++++++++++++-------------------------- 1 file changed, 94 insertions(+), 85 deletions(-) diff --git a/mm/gup.c b/mm/gup.c index bd981ab3975143..a4036c02e2137f 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -2826,11 +2826,11 @@ static bool gup_fast_folio_allowed(struct folio *folio, unsigned int flags) * also check pmd here to make sure pmd doesn't change (corresponds to * pmdp_collapse_flush() in the THP collapse code path). */ -static int gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { - int ret = 0; + unsigned long nr_pages = 0; pte_t *ptep, *ptem; ptem = ptep = pte_offset_map(&pmd, addr); @@ -2892,15 +2892,13 @@ static int gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, unsigned long addr, goto pte_unmap; } folio_set_referenced(folio); - pages[*nr] = page; - (*nr)++; + pages[nr_pages] = page; + nr_pages++; } while (ptep++, addr += PAGE_SIZE, addr != end); - ret = 1; - pte_unmap: pte_unmap(ptem); - return ret; + return nr_pages; } #else @@ -2913,21 +2911,21 @@ static int gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, unsigned long addr, * get_user_pages_fast_only implementation that can pin pages. Thus it's still * useful to have gup_fast_pmd_leaf even if we can't operate on ptes. */ -static int gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pte_range(pmd_t pmd, pmd_t *pmdp, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { return 0; } #endif /* CONFIG_ARCH_HAS_PTE_SPECIAL */ -static int gup_fast_pmd_leaf(pmd_t orig, pmd_t *pmdp, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pmd_leaf(pmd_t orig, pmd_t *pmdp, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { struct page *page; struct folio *folio; - int refs; + unsigned long nr_pages, i; /* See gup_fast_pte_range() */ if (pmd_protnone(orig)) @@ -2939,42 +2937,40 @@ static int gup_fast_pmd_leaf(pmd_t orig, pmd_t *pmdp, unsigned long addr, if (pmd_special(orig)) return 0; - refs = (end - addr) >> PAGE_SHIFT; + nr_pages = (end - addr) >> PAGE_SHIFT; page = pmd_page(orig) + ((addr & ~PMD_MASK) >> PAGE_SHIFT); - folio = try_grab_folio_fast(page, refs, flags); + folio = try_grab_folio_fast(page, nr_pages, flags); if (!folio) return 0; if (unlikely(pmd_val(orig) != pmd_val(pmdp_get_lockless(pmdp)))) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } if (!gup_fast_folio_allowed(folio, flags)) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } if (!pmd_write(orig) && gup_must_unshare(NULL, flags, &folio->page)) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } - pages += *nr; - *nr += refs; - for (; refs; refs--) + for (i = 0; i < nr_pages; i++) *(pages++) = page++; folio_set_referenced(folio); - return 1; + return nr_pages; } -static int gup_fast_pud_leaf(pud_t orig, pud_t *pudp, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pud_leaf(pud_t orig, pud_t *pudp, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { struct page *page; struct folio *folio; - int refs; + unsigned long nr_pages, i; if (!pud_access_permitted(orig, flags & FOLL_WRITE)) return 0; @@ -2982,41 +2978,39 @@ static int gup_fast_pud_leaf(pud_t orig, pud_t *pudp, unsigned long addr, if (pud_special(orig)) return 0; - refs = (end - addr) >> PAGE_SHIFT; + nr_pages = (end - addr) >> PAGE_SHIFT; page = pud_page(orig) + ((addr & ~PUD_MASK) >> PAGE_SHIFT); - folio = try_grab_folio_fast(page, refs, flags); + folio = try_grab_folio_fast(page, nr_pages, flags); if (!folio) return 0; if (unlikely(pud_val(orig) != pud_val(pudp_get(pudp)))) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } if (!gup_fast_folio_allowed(folio, flags)) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } if (!pud_write(orig) && gup_must_unshare(NULL, flags, &folio->page)) { - gup_put_folio(folio, refs, flags); + gup_put_folio(folio, nr_pages, flags); return 0; } - pages += *nr; - *nr += refs; - for (; refs; refs--) + for (i = 0; i < nr_pages; i++) *(pages++) = page++; folio_set_referenced(folio); - return 1; + return nr_pages; } -static int gup_fast_pmd_range(pud_t *pudp, pud_t pud, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pmd_range(pud_t *pudp, pud_t pud, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { - unsigned long next; + unsigned long next, nr_pages = 0, chunk_nr_pages; pmd_t *pmdp; pmdp = pmd_offset_lockless(pudp, pud, addr); @@ -3025,26 +3019,30 @@ static int gup_fast_pmd_range(pud_t *pudp, pud_t pud, unsigned long addr, next = pmd_addr_end(addr, end); if (!pmd_present(pmd)) - return 0; + break; if (unlikely(pmd_leaf(pmd))) { - if (!gup_fast_pmd_leaf(pmd, pmdp, addr, next, flags, - pages, nr)) - return 0; - - } else if (!gup_fast_pte_range(pmd, pmdp, addr, next, flags, - pages, nr)) - return 0; + chunk_nr_pages = gup_fast_pmd_leaf(pmd, pmdp, addr, + next, flags, + &pages[nr_pages]); + + } else + chunk_nr_pages = gup_fast_pte_range(pmd, pmdp, addr, + next, flags, + &pages[nr_pages]); + nr_pages += chunk_nr_pages; + if (chunk_nr_pages != (next - addr) >> PAGE_SHIFT) + break; } while (pmdp++, addr = next, addr != end); - return 1; + return nr_pages; } -static int gup_fast_pud_range(p4d_t *p4dp, p4d_t p4d, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_pud_range(p4d_t *p4dp, p4d_t p4d, + unsigned long addr, unsigned long end, + unsigned int flags, struct page **pages) { - unsigned long next; + unsigned long next, nr_pages = 0, chunk_nr_pages; pud_t *pudp; pudp = pud_offset_lockless(p4dp, p4d, addr); @@ -3053,24 +3051,27 @@ static int gup_fast_pud_range(p4d_t *p4dp, p4d_t p4d, unsigned long addr, next = pud_addr_end(addr, end); if (unlikely(!pud_present(pud))) - return 0; - if (unlikely(pud_leaf(pud))) { - if (!gup_fast_pud_leaf(pud, pudp, addr, next, flags, - pages, nr)) - return 0; - } else if (!gup_fast_pmd_range(pudp, pud, addr, next, flags, - pages, nr)) - return 0; + break; + if (unlikely(pud_leaf(pud))) + chunk_nr_pages = gup_fast_pud_leaf(pud, pudp, addr, + next, flags, + &pages[nr_pages]); + else + chunk_nr_pages = gup_fast_pmd_range(pudp, pud, addr, + next, flags, + &pages[nr_pages]); + nr_pages += chunk_nr_pages; + if (chunk_nr_pages != (next - addr) >> PAGE_SHIFT) + break; } while (pudp++, addr = next, addr != end); - return 1; + return nr_pages; } -static int gup_fast_p4d_range(pgd_t *pgdp, pgd_t pgd, unsigned long addr, - unsigned long end, unsigned int flags, struct page **pages, - int *nr) +static unsigned long gup_fast_p4d_range(pgd_t *pgdp, pgd_t pgd, unsigned long addr, + unsigned long end, unsigned int flags, struct page **pages) { - unsigned long next; + unsigned long next, nr_pages = 0, chunk_nr_pages; p4d_t *p4dp; p4dp = p4d_offset_lockless(pgdp, pgd, addr); @@ -3079,20 +3080,23 @@ static int gup_fast_p4d_range(pgd_t *pgdp, pgd_t pgd, unsigned long addr, next = p4d_addr_end(addr, end); if (!p4d_present(p4d)) - return 0; + break; BUILD_BUG_ON(p4d_leaf(p4d)); - if (!gup_fast_pud_range(p4dp, p4d, addr, next, flags, - pages, nr)) - return 0; + chunk_nr_pages = gup_fast_pud_range(p4dp, p4d, addr, next, + flags, &pages[nr_pages]); + nr_pages += chunk_nr_pages; + if (chunk_nr_pages != (next - addr) >> PAGE_SHIFT) + break; } while (p4dp++, addr = next, addr != end); - return 1; + return nr_pages; } -static void gup_fast_pgd_range(unsigned long addr, unsigned long end, - unsigned int flags, struct page **pages, int *nr) +static unsigned long gup_fast_pgd_range(unsigned long addr, + unsigned long end, unsigned int flags, + struct page **pages) { - unsigned long next; + unsigned long next, nr_pages = 0, chunk_nr_pages; pgd_t *pgdp; pgdp = pgd_offset(current->mm, addr); @@ -3101,17 +3105,23 @@ static void gup_fast_pgd_range(unsigned long addr, unsigned long end, next = pgd_addr_end(addr, end); if (pgd_none(pgd)) - return; + break; BUILD_BUG_ON(pgd_leaf(pgd)); - if (!gup_fast_p4d_range(pgdp, pgd, addr, next, flags, - pages, nr)) - return; + chunk_nr_pages = gup_fast_p4d_range(pgdp, pgd, addr, next, + flags, &pages[nr_pages]); + nr_pages += chunk_nr_pages; + if (chunk_nr_pages != (next - addr) >> PAGE_SHIFT) + break; } while (pgdp++, addr = next, addr != end); + + return nr_pages; } #else -static inline void gup_fast_pgd_range(unsigned long addr, unsigned long end, - unsigned int flags, struct page **pages, int *nr) +static inline unsigned long gup_fast_pgd_range(unsigned long addr, + unsigned long end, unsigned int flags, + struct page **pages) { + return 0; } #endif /* CONFIG_HAVE_GUP_FAST */ @@ -3129,8 +3139,7 @@ static bool gup_fast_permitted(unsigned long start, unsigned long end) static unsigned long gup_fast(unsigned long start, unsigned long end, unsigned int gup_flags, struct page **pages) { - unsigned long flags; - int nr_pinned = 0; + unsigned long flags, nr_pinned; unsigned seq; if (!IS_ENABLED(CONFIG_HAVE_GUP_FAST) || @@ -3154,7 +3163,7 @@ static unsigned long gup_fast(unsigned long start, unsigned long end, * that come from callers of tlb_remove_table_sync_one(). */ local_irq_save(flags); - gup_fast_pgd_range(start, end, gup_flags, pages, &nr_pinned); + nr_pinned = gup_fast_pgd_range(start, end, gup_flags, pages); local_irq_restore(flags); /* From 573ef31c4a7754aa5140c731a9101db16f94e356 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Wed, 19 Aug 2026 10:55:15 +0800 Subject: [PATCH 0366/1012] mm/page_table_check: add explicit pmd_none check in pte_clear_range In __page_table_check_pte_clear_range(), the condition to determine whether to iterate over PTEs only checked pmd_bad() and pmd_leaf(). This relies on the implicit assumption that pmd_none() is always a subset of pmd_bad() on all architectures supporting PAGE_TABLE_CHECK. While this assumption currently holds for x86_64, arm64, s390, riscv, and powerpc, it is an architecture-dependent behavior that may not hold for future architectures. Add an explicit pmd_none() check to make the intent clear and avoid calling pte_offset_map() on an empty PMD, which could lead to undefined behavior. Link: https://lore.kernel.org/20260819025516.2967199-1-ye.liu@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexandre Ghiti Cc: Palmer Dabbelt Cc: Pasha Tatashin --- mm/page_table_check.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/page_table_check.c b/mm/page_table_check.c index 6ffc536359cd06..ed3e1a76f26643 100644 --- a/mm/page_table_check.c +++ b/mm/page_table_check.c @@ -278,7 +278,7 @@ void __page_table_check_pte_clear_range(struct mm_struct *mm, if (&init_mm == mm) return; - if (!pmd_bad(pmd) && !pmd_leaf(pmd)) { + if (!pmd_none(pmd) && !pmd_bad(pmd) && !pmd_leaf(pmd)) { pte_t *ptep = pte_offset_map(&pmd, addr); unsigned long i; From 856107a09dadabf95366d43c590196adaf6ba171 Mon Sep 17 00:00:00 2001 From: Zhiling Zou Date: Tue, 21 Jul 2026 23:56:38 +0800 Subject: [PATCH 0367/1012] mm/page_table_check: skip zero pages page_table_check_set() accounts pages by whether they are PageAnon(). The shared zero page is a special page, not an ordinary file-backed page. Read faults on private anonymous mappings can install many read-only PTEs that point at the zero page, but page_table_check currently accounts them in file_map_count. That lets an unprivileged process populate enough zero-page mappings to overflow file_map_count and trip the BUG_ON() in page_table_check_set(). Skip zero pages in page_table_check accounting. They do not need the anonymous/file mapping conflict checks that page_table_check performs for ordinary pages, and this keeps the existing counter size and page_ext layout unchanged. Link: https://lore.kernel.org/1f8848512d2e3ded944f8d595c29faee8fdaeab0.1784645969.git.roxy520tt@gmail.com Fixes: df4e817b7108 ("mm: page table check") Signed-off-by: Zhiling Zou Signed-off-by: Ren Wei Signed-off-by: Andrew Morton Reported-by: Vega Assisted-by: Codex:gpt-5.4 Cc: Ye Liu Cc: --- mm/page_table_check.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/page_table_check.c b/mm/page_table_check.c index ed3e1a76f26643..143a918c0bde22 100644 --- a/mm/page_table_check.c +++ b/mm/page_table_check.c @@ -67,7 +67,7 @@ static void page_table_check_clear(unsigned long pfn, unsigned long pgcnt) struct page *page; bool anon; - if (!pfn_valid(pfn)) + if (!pfn_valid(pfn) || is_zero_pfn(pfn) || is_huge_zero_pfn(pfn)) return; page = pfn_to_page(pfn); @@ -102,7 +102,7 @@ static void page_table_check_set(unsigned long pfn, unsigned long pgcnt, struct page *page; bool anon; - if (!pfn_valid(pfn)) + if (!pfn_valid(pfn) || is_zero_pfn(pfn) || is_huge_zero_pfn(pfn)) return; page = pfn_to_page(pfn); From f0df882e706b1c0d06375b1f1b679776e9177119 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:42 -0700 Subject: [PATCH 0368/1012] mm/damon/core: skip applying scheme if region split for quota fails Patch series "mm/damon: fix DAMOS bugs in core, paddr and vaddr". Fix misc bugs of DAMOS. Patch 1 makes DAMOS less stress memory allocator under extreme situation. Patches 2 and 3 fix wrong folios walking in DAMON_PADDR. Patches 4 and 5 fix wrong folios walking in DAMON_VADDR. Patches 6-8 handle extreme and unlikely memory situations that can cause divide by zero and underflow. All bugs are discovered by Sashiko. This patch (of 8): damos_apply_scheme() splits a region and apply the action to the subregion if it is needed for not violating the quota. The split operation (damon_split_region_at()) could fail for allocation failure. In the case, the quota could be violated. From the user's perspective, DAMOS becomes more aggressive than expected under the extreme situation. Handle the failure. The user impact is not critical. The failure of damon_split_region_at() is unlikely since it is arguably too small to fail. Also DAMOS being aggressive is limited to the single region. Users can set min_nr_regions to set the maximum size of each region. If it is reasonably set, the transient overhead shouldn't be critical. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-1-sj@kernel.org Link: https://lore.kernel.org/20260901131850.98037-2-sj@kernel.org Link: https://lore.kernel.org/20260718171523.87547-1-sj@kernel.org [1] Fixes: 2b8a248d5873 ("mm/damon/schemes: implement size quota for schemes application speed control") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 5.16.x --- mm/damon/core.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index d63d4c6fd3ffdd..53952d16ebd882 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2607,7 +2607,8 @@ static void damos_apply_scheme(struct damon_ctx *c, struct damon_target *t, c->min_region_sz); if (!sz) goto update_stat; - damon_split_region_at(t, r, sz); + if (damon_split_region_at(t, r, sz)) + goto update_stat; } if (damos_core_filter_out(c, t, r, s)) return; From aff2abfca2cca38547040f2c827db34d19cfd8d1 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:43 -0700 Subject: [PATCH 0369/1012] mm/damon/paddr: respect folio end for DAMOS_STAT The function for applying DAMOS_STAT in DAMON physical address space operation set (paddr), namely damon_pa_stat(), applies DAMOS filters to folios of the given region. For that, it gets folios of addresses in the region. It starts from the region start address and advances the address by the size of the folio of the address until it goes out of the region. If the start address is in the middle of a large folio, and if the next folios are small, some of the next folios could be skipped. Fix the issue by advancing the address to exactly the start address of the next folio. The user impact is that the DAMOS_STAT-based page level monitoring results become inaccurate. Since the page level monitoring is supposed to provide relatively high precision, this is definitely a problem. It is arguably not critical since it is only monitoring quality degradation. Link: https://lore.kernel.org/20260901131850.98037-3-sj@kernel.org Fixes: bdbe1d7bc325 ("mm/damon/paddr: increment pa_stat damon address range by folio size") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 6.14.x --- mm/damon/paddr.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 5c6c3a597fd0bf..2ab7b3842701ed 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -379,7 +379,7 @@ static unsigned long damon_pa_stat(struct damon_region *r, if (!damos_pa_filter_out(s, folio)) *sz_filter_passed += folio_size(folio) / addr_unit; - addr += folio_size(folio); + addr = PFN_PHYS(folio_pfn(folio)) + folio_size(folio); folio_put(folio); } s->last_applied = folio; From 1809a457ce31e94df5f773ec317e91cf6080af18 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:44 -0700 Subject: [PATCH 0370/1012] mm/damon/paddr: respect folio end for DAMOS actions except STAT A few functions for applying DAMOS actions including pageout, lru_[de]prio and migrate_{hot,cold} in DAMON physical address space operation set (paddr) collect folios of the given region by getting the folios of region-internal addresses. Then, those functions apply the action to the collected folios at once. The collection starts from the region start address and advances the address by the size of the folio of the address until it goes out of the region. If the start address is in the middle of a large folio, and if the next folios are small, some of the next folios could be skipped. Fix the issue by advancing the address to exactly the start address of the next folio. The user impact is that DAMOS action is applied to less than expected amount of memory. Given the best effort nature of DAMON, it is no big problem, but it is clearly a bug that is better to be fixed. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-4-sj@kernel.org Link: https://lore.kernel.org/20260517234112.89245-1-sj@kernel.org [1] Fixes: 3a06696305e7 ("mm/damon/ops: have damon_get_folio return folio even for tail pages") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 6.15.x --- mm/damon/paddr.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 2ab7b3842701ed..9ddd1ec8202b7f 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -264,7 +264,7 @@ static unsigned long damon_pa_pageout(struct damon_region *r, else list_add(&folio->lru, &folio_list); put_folio: - addr += folio_size(folio); + addr = PFN_PHYS(folio_pfn(folio)) + folio_size(folio); folio_put(folio); } if (install_young_filter) @@ -302,7 +302,7 @@ static inline unsigned long damon_pa_de_activate( folio_deactivate(folio); applied += folio_nr_pages(folio); put_folio: - addr += folio_size(folio); + addr = PFN_PHYS(folio_pfn(folio)) + folio_size(folio); folio_put(folio); } s->last_applied = folio; @@ -350,7 +350,7 @@ static unsigned long damon_pa_migrate(struct damon_region *r, folio_is_file_lru(folio)); list_add(&folio->lru, &folio_list); put_folio: - addr += folio_size(folio); + addr = PFN_PHYS(folio_pfn(folio)) + folio_size(folio); folio_put(folio); } applied = damon_migrate_pages(&folio_list, s->target_nid); From 95a06680295a0af4a0ee5a2a240fccfa6a290c7a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:45 -0700 Subject: [PATCH 0371/1012] mm/damon/vaddr: respect folio end for DAMOS_STAT For applying DAMOS_STAT action to a region, DAMON virtual address space operation set (vaddr) calls walk_page_range[_vma]() for the region. The pmd walk entry function, namely damon_va_stat_pmd_entry(), applies DAMOS filters to folios of addresses of the region in the pmd. It starts from the walking address and advances the address by the size of the folio of the address until it goes out of the pmd or the region. Let's suppose it is for the first pmd of the region, and the region start address is in the middle of a large folio. Also, the next folios are small. Then, some of the next folios could be skipped. Fix the issue by advancing the address to exactly the start address of the next folio. The user impact is that the DAMOS_STAT-based page level monitoring results become inaccurate. Since the page level monitoring is supposed to provide relatively high precision, this is definitely a problem. It is arguably not critical since it is only monitoring quality degradation. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-5-sj@kernel.org Link: https://lore.kernel.org/20260514015053.149396-1-sj@kernel.org [1] Fixes: 63f39737d1e3 ("mm/damon/vaddr: support stat-purpose DAMOS filters") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 6.18.x --- mm/damon/vaddr.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 04ee2a2c6a4d65..b1bed5d19a34b8 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -838,6 +838,8 @@ static int damos_va_stat_pmd_entry(pmd_t *pmd, unsigned long addr, return 0; for (; addr < next; pte += nr, addr += nr * PAGE_SIZE) { + unsigned long page_idx; + nr = 1; ptent = ptep_get(pte); @@ -851,7 +853,8 @@ static int damos_va_stat_pmd_entry(pmd_t *pmd, unsigned long addr, if (!damos_va_filter_out(s, folio, vma, addr, pte, NULL)) *sz_filter_passed += folio_size(folio); - nr = folio_nr_pages(folio); + page_idx = folio_page_idx(folio, pte_page(ptent)); + nr = folio_nr_pages(folio) - page_idx; s->last_applied = folio; } pte_unmap_unlock(start_pte, ptl); From 91fff08721353f53bad5ea4df885923133234f3a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:46 -0700 Subject: [PATCH 0372/1012] mm/damon/vaddr: respect folio end for DAMOS_MIGRATE_{HOT,COLD} For applying DAMOS_MIGRATE_{HOT,COLD} actions to a region, DAMON virtual address space operation set (vaddr) calls walk_page_range[_vma]() for the region. The pmd walk entry function, namely damon_va_migrate_pmd_entry(), collects folios of addresses of the region in the pmd. It starts from the walking address and advances the address by the size of the folio of the address until it goes out of the pmd or the region. Let's suppose it is for the first pmd of the region, and the region start address is in the middle of a large folio. Also, the next folios are small. Then, some of the next folios could be skipped. Fix the issue by advancing the address to exactly the start address of the next folio. The user impact is that DAMOS action is applied to less than expected amount of memory. Given the best effort nature of DAMON, it is no big problem, but it is clearly a bug that is better to be fixed. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-6-sj@kernel.org Link: https://lore.kernel.org/20260514015053.149396-1-sj@kernel.org [1] Fixes: 09efc56a3b1c ("mm/damon/vaddr: consistently use only pmd_entry for damos_migrate") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 6.19.x --- mm/damon/vaddr.c | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index b1bed5d19a34b8..5b4d16c8db6285 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -676,6 +676,8 @@ static int damos_va_migrate_pmd_entry(pmd_t *pmd, unsigned long addr, return 0; for (; addr < next; pte += nr, addr += nr * PAGE_SIZE) { + unsigned long page_idx; + nr = 1; ptent = ptep_get(pte); @@ -688,7 +690,8 @@ static int damos_va_migrate_pmd_entry(pmd_t *pmd, unsigned long addr, continue; damos_va_migrate_dests_add(folio, walk->vma, addr, dests, migration_lists); - nr = folio_nr_pages(folio); + page_idx = folio_page_idx(folio, pte_page(ptent)); + nr = folio_nr_pages(folio) - page_idx; } pte_unmap_unlock(start_pte, ptl); return 0; From 5359eed187a1a6d030cd58d0f2f7be3a6f87f3a1 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:47 -0700 Subject: [PATCH 0373/1012] mm/damon/core: handle extreme memory state in damon_get_node_mem_bp() In an extreme and unlikely situation, si_meminfo_node() might let the caller show zero total ram. That could cause a divide by zero in damon_get_node_mem_bp(). It could also show free memory larger than the total memory. This could cause underflow and make DAMOS temporarily make unexpected behavior. Thanks to safety guards in the auto-tuning feedback loop, that should not be a real problem, though. Fix the problems by respectively returning 100% and 0% for used and free memory queries in the corner cases. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-7-sj@kernel.org Link: https://lore.kernel.org/20260328133216.9697-1-sj@kernel.org [1] Fixes: 0e1c773b501f ("mm/damon/core: introduce damos quota goal metrics for memory node utilization") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 6.16.x --- mm/damon/core.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 53952d16ebd882..2942f23a240b95 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2810,6 +2810,13 @@ static __kernel_ulong_t damos_get_node_mem_bp( } si_meminfo_node(&i, goal->nid); + if (!i.totalram || i.totalram < i.freeram) { + if (goal->metric == DAMOS_QUOTA_NODE_MEM_USED_BP) + return 10000; + else /* DAMOS_QUOTA_NODE_MEM_FREE_BP */ + return 0; + } + if (goal->metric == DAMOS_QUOTA_NODE_MEM_USED_BP) numerator = i.totalram - i.freeram; else /* DAMOS_QUOTA_NODE_MEM_FREE_BP */ From d02cb2fe7e47807a8ce371861ae23861f8abb2fb Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:48 -0700 Subject: [PATCH 0374/1012] mm/damon/core: handle extreme memory state in get_node_memcg_used_bp() In extreme unlikely situations, total memory might be zero. In less extreme but still very unlikely situations, lruvec_page_state() calls might let the caller show used memory larger than total memory. In the two cases, damos_get_node_memcg_used_bp() could cause division by zero, or return underflowed value, respectively. Handle the cases by respectively returning 100% and 0% for used and free memory queries in the corner cases. This issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-8-sj@kernel.org Link: https://lore.kernel.org/20260329154813.47382-1-sj@kernel.org [1] Fixes: b74a120bcf50 ("mm/damon/core: implement DAMOS_QUOTA_NODE_MEMCG_USED_BP") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 6.19.x --- mm/damon/core.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 2942f23a240b95..d00cf1fae23909 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2857,6 +2857,12 @@ static unsigned long damos_get_node_memcg_used_bp( mem_cgroup_put(memcg); si_meminfo_node(&i, goal->nid); + if (!i.totalram || i.totalram < used_pages) { + if (goal->metric == DAMOS_QUOTA_NODE_MEMCG_USED_BP) + return 10000; + else /* DAMOS_QUOTA_NODE_MEMCG_FREE_BP */ + return 0; + } if (goal->metric == DAMOS_QUOTA_NODE_MEMCG_USED_BP) numerator = used_pages; else /* DAMOS_QUOTA_NODE_MEMCG_FREE_BP */ From 5e2b987777e6b82e1f535f55713a3cd5780b672b Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:18:49 -0700 Subject: [PATCH 0375/1012] mm/damon/core: handle extreme memory state in get_in_active_mem_bp() damos_get_in_active_mem_bp() uses the sum of the active and inactive memory amount as a denominator. In an extreme and unlikely environment, active and inactive memory might be zero. In this case, hence, it results in a divide by zero problem. Avoid it by changing the denominator to one if it is zero, before it is being used. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260901131850.98037-9-sj@kernel.org Link: https://lore.kernel.org/20260721034756.147011-1-sj@kernel.org [1] Fixes: 4835e2871321 ("mm/damon/core: introduce [in]active memory ratio damos quota goal metric") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Usama Arif Cc: # 7.0.x --- mm/damon/core.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index d00cf1fae23909..f6a9d2da0cd721 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3009,7 +3009,7 @@ static unsigned int damos_get_in_active_mem_bp(bool active_ratio) global_node_page_state(NR_LRU_BASE + LRU_ACTIVE_FILE); inactive = global_node_page_state(NR_LRU_BASE + LRU_INACTIVE_ANON) + global_node_page_state(NR_LRU_BASE + LRU_INACTIVE_FILE); - total = active + inactive; + total = max(active + inactive, 1); if (active_ratio) return mult_frac(active, 10000, total); return mult_frac(inactive, 10000, total); From f145fbdcb92cd1ce46ebbfe995b8e02069c1d5fb Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:48 -0700 Subject: [PATCH 0376/1012] mm/damon/core: introduce DAMON_FILTER_TYPE_PGIDLE_UNSET Patch series "mm/damon: introduce data access-as-a-data attribute", v1.1. TL;DR: extend DAMON's data attributes monitoring system to support page table accessed bit and PG_idle based access monitoring. DAMON was initially introduced as a data access monitor. Users found data access pattern becomes more useful when it is combined with other data attributes such as belonging cgroups and backing page types. For such cases, DAMON has extended to support such data attributes monitoring in addition to the original data access monitoring. The DAMON probes system was introduced for this purpose. Users set probes for filtering data attributes of their interest. For use cases where the primary interests are the attributes but the access pattern, probe weights system has been introduced. When it is used, DAMON applies its adaptive regions adjustment based on the monitored data attributes. However, DAMON stops access monitoring when the probe weights are used. DAMON cannot optimally help users who have interests in both data attributes and access patterns. Data access can also be thought of as another data attribute, though. Extend the probe system to support data access as a data attribute. Introduce a new probe filter type, pgidle_unset. It shows if the page is not set as idle. Specifically, it shows the page table accessed bit and the PG_idle flag. It can inform if the region is ever accessed. But it cannot say when it is accessed. To answer the second question, introduce a new probe feature, prep actions. Using the features, Users can specify what preparation actions should be made to each region for each probe. DAMON executes the preparation actions for each sampling interval, like it is doing the preparation for access check in the access monitoring mode. To help 'pgidle_unset' probe action use case, 'set_pgidle' preparation action is introduced together. The action does exactly what the access monitoring was doing: clearing the page table accessed bits and setting the PG_idle flags. Tests ===== I compared the access pattern monitoring results from the classic way and the probe based way. As expected, the probe based way shows the results similar to that of the classic way. More detailed test methods and results are below. First, do the access monitoring using the DAMON user-space tool [1] in the classic way. The system is idle. It shows no access as expected. $ sudo ./damo/damo start $ sudo ./damo/damo report access heatmap: 00000000000000000000000000000000000000008999999811111110000000000000000000000000 # min/max temperatures: -640,000,000, -100,000,000, column size: 99.800 MiB intervals: sample 5 ms aggr 100 ms (max access hz 200) 0 addr 4.000 KiB size 3.898 GiB access 0 hz age 6.400 s 1 addr 3.898 GiB size 787.301 MiB access 0 hz age 1 s 2 addr 4.667 GiB size 773.457 MiB access 0 hz age 5.800 s 3 addr 5.423 GiB size 2.374 GiB access 0 hz age 6.400 s memory bw estimate: 0 B per second total size: 7.797 GiB record DAMON intervals: sample 5 ms, aggr 100 ms Start an artificial memory access generator (masim) [2] in another window. $ ./masim/masim.py run --config_file ./masim/configs/zigzag.cfg Show the monitoring results. As expected, it captures accesses. $ sudo ./damo/damo report access heatmap: 00000000000000000000000000000000000000011111111111111118888888833378988888888888 # min/max temperatures: -1,470,000,000, -12,482,536, column size: 99.800 MiB intervals: sample 5 ms aggr 100 ms (max access hz 200) 0 addr 4.000 KiB size 1.542 GiB access 0 hz age 14.700 s 1 addr 1.542 GiB size 788.324 MiB access 0 hz age 14.600 s 2 addr 2.311 GiB size 772.844 MiB access 0 hz age 14.200 s [...] 50 addr 6.839 GiB size 1.742 MiB access 0 hz age 1.300 s 51 addr 6.840 GiB size 8.000 KiB access 100 hz age 0 ns 52 addr 6.840 GiB size 1.496 MiB access 160 hz age 0 ns [...] 97 addr 7.162 GiB size 1.199 MiB access 40 hz age 400 ms 98 addr 7.163 GiB size 1.199 MiB access 20 hz age 400 ms 99 addr 7.164 GiB size 2.004 MiB access 160 hz age 0 ns 100 addr 7.166 GiB size 646.004 MiB access 0 hz age 400 ms memory bw estimate: 26.148 GiB per second total size: 7.797 GiB record DAMON intervals: sample 5 ms, aggr 100 ms After the artificial memory access generator (masim) is terminated, restart DAMON with the probe-based access monitoring. As expected, it shows no access since the system is idle again. $ sudo ./damo/damo stop $ sudo ./damo/damo start --probe_prep set_pgidle \ --probe_filter allow pgidle_unset --probe_weight 1 $ sudo ./damo/damo report attrs heatmap: 00000000000000000000000000000000000000008999999711111100000000000000000000000000 # min/max temperatures: -600,000,000, -430,000,000, column size: 99.800 MiB probe prep: set_pgidle, filter: allow pgidle_unset (weight: 1) intervals: sample 5 ms aggr 100 ms (max probe hits 20) # addr size age probe_hits 0 4.000 KiB 3.898 GiB 6 s 0 1 5.285 GiB 2.512 GiB 6 s 0 2 4.659 GiB 641.816 MiB 5.700 s 0 3 3.898 GiB 778.375 MiB 4.300 s 0 memory bw estimate: 0 B per second total size: 7.797 GiB record DAMON intervals: sample 5 ms, aggr 100 ms Start the artificial memory access generator [2] again. $ ./masim/masim.py run --config_file ./masim/configs/zigzag.cfg Show the monitoring results. As expected, it captures accesses similar to the classic monitoring mode. $ sudo ./damo/damo report attrs heatmap: 00000000000000000000000000000000000000011111110000000177777777777777878798887777 # min/max temperatures: -1,330,000,000, 84,847,514, column size: 99.800 MiB probe prep: set_pgidle, filter: allow pgidle_unset (weight: 1) intervals: sample 5 ms aggr 100 ms (max probe hits 20) # addr size age probe_hits 0 4.000 KiB 1.508 GiB 13.300 s 0 1 1.508 GiB 794.684 MiB 13.200 s 0 2 2.284 GiB 764.555 MiB 13 s 0 [...] 50 6.711 GiB 4.625 MiB 0 ns 17 51 6.716 GiB 2.504 MiB 0 ns 17 52 6.732 GiB 796.000 KiB 0 ns 17 [...] 90 7.162 GiB 512.000 KiB 2.200 s 2 91 7.061 GiB 700.000 KiB 2.300 s 1 92 6.935 GiB 316.000 KiB 2.300 s 4 memory bw estimate: 0 B per second total size: 7.797 GiB record DAMON intervals: sample 5 ms, aggr 100 ms Patches Sequence ================ First four patches (patches 1-4) introduce the new probe filter type for knowing if a region is accessed. Patch 1 defines the new type in the core. Patch 2 implements the execution of the new filter in the physical address space DAMON operation set. Patch 3 implements a user interface on DAMON sysfs interface. Patch 4 updates the documentation for the new filter type. Following 13 patches (patches 5-17) introduce the probe preparation actions feature. Patch 5 defines the data structure for specifying the preparation actions. Patch 6 completes setup of the API parameter for the prep. Patch 7 extends the DAMON operation set callback list to connect the parameter with the underlying operation set. Patch 8 implements the execution of the prep in the physical address space DAMON operation set. Following five patches (patches 9-13) extends DAMON sysfs interface for the new prep feature. Patch 14 adds simple selftest for basic file operations of the new sysfs files. Final three patches (patches 15-17) respectively update design, usage and ABI documents for the new feature and its interface. This patch (of 17): Introduce a new DAMON filter type, pgidle_unset. It will match pages that have their PG_Idle flag unset, or the page table accessed bit set. In other words, it says if the page is accessed. Link: https://lore.kernel.org/20260901132506.99243-1-sj@kernel.org Link: https://lore.kernel.org/20260901132506.99243-2-sj@kernel.org Link: https://github.com/damonitor/damo [1] Link: https://github.com/sjp38/masim [2] Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 7b1b6050a8286f..7a42cbe791845d 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -745,12 +745,14 @@ struct damon_intervals_goal { /** * enum damon_filter_type - Type of &struct damon_filter * - * @DAMON_FILTER_TYPE_ANON: Anonymous pages. - * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages. + * @DAMON_FILTER_TYPE_ANON: Anonymous pages. + * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages. + * @DAMON_FILTER_TYPE_PGIDLE_UNSET: Pgidle is unset. */ enum damon_filter_type { DAMON_FILTER_TYPE_ANON, DAMON_FILTER_TYPE_MEMCG, + DAMON_FILTER_TYPE_PGIDLE_UNSET, }; /** From 7ee1cca08a3502b73defedf149d5a8684f600d90 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:49 -0700 Subject: [PATCH 0377/1012] mm/damon/paddr: support PGIDLE_UNSET probe filter type Implement support of DAMON_FILTER_TYPE_PGIDLE_UNSET in the physical address space DAMON operations set. It reuses damon_folio_young(), which was being used for access monitoring. Link: https://lore.kernel.org/20260901132506.99243-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/paddr.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 9ddd1ec8202b7f..6f756f84938948 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -132,6 +132,12 @@ static bool damon_pa_filter_match(struct damon_filter *filter, matched = filter->memcg_id == mem_cgroup_id(memcg); rcu_read_unlock(); break; + case DAMON_FILTER_TYPE_PGIDLE_UNSET: + if (!folio) + matched = false; + else + matched = damon_folio_young(folio); + break; default: break; } From 49662a5688ea42df64db3b0b643532a6faeb212d Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:50 -0700 Subject: [PATCH 0378/1012] mm/damon/sysfs: support pgidle_unset probe filter type Extend DAMON sysfs interface to allow users to set DAMON_FILTER_TYPE_PGIDLE_UNSET by writing 'pgidle_unset' to the probe filter type file. Link: https://lore.kernel.org/20260901132506.99243-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 3c81b4c91ac0dd..c1ff739dab27a8 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -781,6 +781,10 @@ damon_sysfs_filter_type_names[] = { .type = DAMON_FILTER_TYPE_MEMCG, .name = "memcg", }, + { + .type = DAMON_FILTER_TYPE_PGIDLE_UNSET, + .name = "pgidle_unset", + }, }; static ssize_t type_show(struct kobject *kobj, From 820e07c5e23e24d8a0886f30e0e9f5183fa1bd69 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:51 -0700 Subject: [PATCH 0379/1012] Docs/mm/damon/design: document pgidle_unset probe filter type Update DAMON design document for the newly added pgidle_unset probe filter type. Also use a list for the types, as it becomes not very easy to read the whole types in a simple sentence. Link: https://lore.kernel.org/20260901132506.99243-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 63cbb7b536da20..947d91ae24a392 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -293,8 +293,12 @@ registration is made by specifying a probe per attribute. Each of the probe specifies a rule to determine if a given memory region has the related attribute. The rule is constructed with multiple filters. The filters work same to :ref:`DAMOS filters ` except the supported -filter types. Currently only ``anon`` and ``memcg`` filter types are supported -for data attributes monitoring. +filter types. Currently below filter types are supported. + +- ``anon``: Same to that for DAMOS filters. +- ``memcg``: Same to that for DAMOS filters. +- ``pgidle_unset``: Matches if the page for the memory is marked as not + access-idle. If such probes are registered, DAMON executes the probes for each region's sampling memory when it does the access :ref:`sampling From 471bfdd0c977876ede698b1182b2bd4575989765 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:52 -0700 Subject: [PATCH 0380/1012] mm/damon/core: introduce damon_prep struct Some DAMON probe filter types require preparatory actions. For example, pgilde_unset probe filter can say if the page was accessed but when. To answer the second question, the PG_Idle flag should be set at a specific time. It can make life much easier if DAMON can do such preparatory actions. Introduce a new data type called damon_prep. It specifies each of the preparation actions for each probe. DAMON will execute the action for each region per sampling interval, like it clears page table accessed bits and unsets PG_Idle flag for access monitoring. Also introduce DAMON_PREP_SET_PGIDLE as the initial prep action. As the name says, it will do exactly what DAMON was doing as the preparation action for the access monitoring. Link: https://lore.kernel.org/20260901132506.99243-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 32 ++++++++++++++++++++++++++++++++ mm/damon/core.c | 26 ++++++++++++++++++++++++++ 2 files changed, 58 insertions(+) diff --git a/include/linux/damon.h b/include/linux/damon.h index 7a42cbe791845d..1780c14942e634 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -742,6 +742,27 @@ struct damon_intervals_goal { unsigned long max_sample_us; }; +/** + * enum damon_prep_action - DAMON probing preparation action. + * + * @DAMON_PREP_SET_PGIDLE: Set the probing memory as idle page. + */ +enum damon_prep_action { + DAMON_PREP_SET_PGIDLE, +}; + +/** + * struct damon_prep - DAMON probing preparation request. + * + * @action: Action to do to the probing memory for the preparation. + */ +struct damon_prep { + enum damon_prep_action action; +/* private: */ + /* siblings list. */ + struct list_head list; +}; + /** * enum damon_filter_type - Type of &struct damon_filter * @@ -783,6 +804,8 @@ struct damon_filter { struct damon_probe { unsigned int weight; /* private: */ + /* Preparation actions to apply to each probing memory. */ + struct list_head preps; /* Filters for assessing if a given region is for this probe. */ struct list_head filters; /* Siblings list. */ @@ -962,6 +985,12 @@ static inline unsigned long damon_sz_region(struct damon_region *r) return r->ar.end - r->ar.start; } +#define damon_for_each_prep(p, probe) \ + list_for_each_entry(p, &(probe)->preps, list) + +#define damon_for_each_prep_safe(p, next, probe) \ + list_for_each_entry_safe(p, next, &(probe)->preps, list) + #define damon_for_each_filter(f, p) \ list_for_each_entry(f, &(p)->filters, list) @@ -1015,6 +1044,9 @@ static inline unsigned long damon_sz_region(struct damon_region *r) #ifdef CONFIG_DAMON +struct damon_prep *damon_new_prep(enum damon_prep_action action); +void damon_add_prep(struct damon_probe *p, struct damon_prep *prep); + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow); void damon_add_filter(struct damon_probe *probe, struct damon_filter *f); diff --git a/mm/damon/core.c b/mm/damon/core.c index f6a9d2da0cd721..a0a61ebdf5bb50 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -112,6 +112,28 @@ int damon_select_ops(struct damon_ctx *ctx, enum damon_ops_id id) return err; } +struct damon_prep *damon_new_prep(enum damon_prep_action action) +{ + struct damon_prep *prep; + + prep = kmalloc_obj(*prep); + if (!prep) + return NULL; + prep->action = action; + INIT_LIST_HEAD(&prep->list); + return prep; +} + +void damon_add_prep(struct damon_probe *p, struct damon_prep *prep) +{ + list_add_tail(&prep->list, &p->preps); +} + +static void damon_free_prep(struct damon_prep *p) +{ + kfree(p); +} + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow) { @@ -168,6 +190,7 @@ struct damon_probe *damon_new_probe(void) if (!p) return NULL; p->weight = 0; + INIT_LIST_HEAD(&p->preps); INIT_LIST_HEAD(&p->filters); INIT_LIST_HEAD(&p->list); return p; @@ -185,8 +208,11 @@ static void damon_del_probe(struct damon_probe *p) static void damon_free_probe(struct damon_probe *p) { + struct damon_prep *prep, *prep_next; struct damon_filter *f, *next; + damon_for_each_prep_safe(prep, prep_next, p) + damon_free_prep(prep); damon_for_each_filter_safe(f, next, p) damon_free_filter(f); kfree(p); From 3fdc73d2488b15f7d750de06294b57a5fbd88203 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:53 -0700 Subject: [PATCH 0381/1012] mm/damon/core: commit preps damon_commit_probes() is ignoring damon_prep. Commit the prep actions, too. Link: https://lore.kernel.org/20260901132506.99243-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 59 +++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 59 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index a0a61ebdf5bb50..c631b532437777 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -129,11 +129,34 @@ void damon_add_prep(struct damon_probe *p, struct damon_prep *prep) list_add_tail(&prep->list, &p->preps); } +static void damon_del_prep(struct damon_prep *p) +{ + list_del(&p->list); +} + static void damon_free_prep(struct damon_prep *p) { kfree(p); } +static void damon_destroy_prep(struct damon_prep *p) +{ + damon_del_prep(p); + damon_free_prep(p); +} + +static struct damon_prep *damon_nth_prep(int n, struct damon_probe *p) +{ + struct damon_prep *prep; + int i = 0; + + damon_for_each_prep(prep, p) { + if (i++ == n) + return prep; + } + return NULL; +} + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow) { @@ -1699,6 +1722,36 @@ static int damon_commit_targets( return err; } +static void damon_commit_prep(struct damon_prep *dst, struct damon_prep *src) +{ + dst->action = src->action; +} + +static int damon_commit_preps(struct damon_probe *dst, struct damon_probe *src) +{ + struct damon_prep *dst_prep, *next, *src_prep, *new_prep; + int i = 0, j = 0; + + damon_for_each_prep_safe(dst_prep, next, dst) { + src_prep = damon_nth_prep(i++, src); + if (src_prep) + damon_commit_prep(dst_prep, src_prep); + else + damon_destroy_prep(dst_prep); + } + + damon_for_each_prep_safe(src_prep, next, src) { + if (j++ < i) + continue; + + new_prep = damon_new_prep(src_prep->action); + if (!new_prep) + return -ENOMEM; + damon_add_prep(dst, new_prep); + } + return 0; +} + static void damon_commit_filter(struct damon_filter *dst, struct damon_filter *src) { @@ -1757,6 +1810,9 @@ static int damon_commit_probes(struct damon_ctx *dst, struct damon_ctx *src) src_probe = damon_nth_probe(i++, src); if (src_probe) { dst_probe->weight = src_probe->weight; + err = damon_commit_preps(dst_probe, src_probe); + if (err) + return err; err = damon_commit_filters(dst_probe, src_probe); if (err) return err; @@ -1774,6 +1830,9 @@ static int damon_commit_probes(struct damon_ctx *dst, struct damon_ctx *src) return -ENOMEM; damon_add_probe(dst, new_probe); new_probe->weight = src_probe->weight; + err = damon_commit_preps(new_probe, src_probe); + if (err) + return err; err = damon_commit_filters(new_probe, src_probe); if (err) return err; From 4612c22119c3367131b8e47d8a54d0f8c33bea1e Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:54 -0700 Subject: [PATCH 0382/1012] mm/damon/core: introduce damon_operations->prep_probes() damon_prep needs to be executed by the underlying DAMON operation set. Extend the operation set callback list for the execution of damon_prep actions. If the underlying operation set implements the callback, DAMON core executes it in the monitoring preparation time. Link: https://lore.kernel.org/20260901132506.99243-8-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 5 +++++ mm/damon/core.c | 20 +++++++++++++++++++- 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 1780c14942e634..871d26adf6ae57 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -630,6 +630,7 @@ enum damon_ops_id { * @update: Update operations-related data structures. * @prepare_access_checks: Prepare next access check of target regions. * @check_accesses: Check the accesses to target regions. + * @prep_probes: Prepare applying probes for each region. * @apply_probes: Apply probes for each region. * @get_scheme_score: Get the score of a region for a scheme. * @apply_scheme: Apply a DAMON-based operation scheme. @@ -657,6 +658,9 @@ enum damon_ops_id { * last preparation and update the number of observed accesses of each region. * It should also return max number of observed accesses that made as a result * of its update. The value will be used for regions adjustment threshold. + * @prep_probes should execute required &struct damon_prep for next &struct + * damon_probe applications to each region. It should also set + * &damon_region->sampling_addr of each region if ``set_samples`` is true. * @apply_probes should apply the data attribute probes to each region and * accordingly update the probe hits counter of the region. It should also * set &damon_region->sampling_addr of each region if ``set_samples`` is true. @@ -679,6 +683,7 @@ struct damon_operations { void (*update)(struct damon_ctx *context); void (*prepare_access_checks)(struct damon_ctx *context); unsigned int (*check_accesses)(struct damon_ctx *context); + void (*prep_probes)(struct damon_ctx *context, bool set_samples); unsigned int (*apply_probes)(struct damon_ctx *context, bool set_samples, bool return_max_wsum); int (*get_scheme_score)(struct damon_ctx *context, diff --git a/mm/damon/core.c b/mm/damon/core.c index c631b532437777..5ee2d0448a8091 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -157,6 +157,18 @@ static struct damon_prep *damon_nth_prep(int n, struct damon_probe *p) return NULL; } +static bool damon_has_prep(struct damon_ctx *c) +{ + struct damon_prep *prep; + struct damon_probe *probe; + + damon_for_each_probe(probe, c) { + damon_for_each_prep(prep, probe) + return true; + } + return false; +} + struct damon_filter *damon_new_filter(enum damon_filter_type type, bool matching, bool allow) { @@ -3905,14 +3917,19 @@ static int kdamond_fn(void *data) unsigned long next_ops_update_sis = ctx->next_ops_update_sis; unsigned long sample_interval = ctx->attrs.sample_interval; bool access_check_disabled = damon_has_probe_weights(ctx); + bool do_prep; unsigned int max_merge_score = 0, max_wsum; bool get_max_wsum; if (kdamond_wait_activation(ctx)) break; + do_prep = ctx->ops.prep_probes && damon_has_prep(ctx); + if (!access_check_disabled && ctx->ops.prepare_access_checks) ctx->ops.prepare_access_checks(ctx); + if (do_prep) + ctx->ops.prep_probes(ctx, access_check_disabled); kdamond_usleep(sample_interval); ctx->passed_sample_intervals++; @@ -3927,7 +3944,8 @@ static int kdamond_fn(void *data) else get_max_wsum = false; max_wsum = ctx->ops.apply_probes(ctx, - access_check_disabled, get_max_wsum); + access_check_disabled && !do_prep, + get_max_wsum); if (get_max_wsum) max_merge_score = max_wsum; } From 9dab68c753eb974aacc21415c0b6d49bdc7b6bef Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:55 -0700 Subject: [PATCH 0383/1012] mm/damon/paddr: support damon_prep Implement damon_operations->prep_probes() callback. Support the only existing prep action, DAMON_PREP_SET_PGIDLE in a way similar to what it was doing for the access check preparation: unset page table accessed bits and set PG_Idle flag. Reuse the function for the access check preparation. Link: https://lore.kernel.org/20260901132506.99243-9-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/paddr.c | 35 +++++++++++++++++++++++++++++++++++ 1 file changed, 35 insertions(+) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 6f756f84938948..c1e7d7a4f40df3 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -105,6 +105,40 @@ static unsigned int damon_pa_check_accesses(struct damon_ctx *ctx) return max_nr_accesses; } +static void damon_pa_prep_probes_region(struct damon_region *r, + struct damon_probe *probe, struct damon_ctx *ctx) +{ + struct damon_prep *p; + + damon_for_each_prep(p, probe) { + switch (p->action) { + case DAMON_PREP_SET_PGIDLE: + damon_pa_mkold(damon_pa_phys_addr(r->sampling_addr, + ctx->addr_unit)); + break; + default: + break; + } + } +} + +static void damon_pa_prep_probes(struct damon_ctx *ctx, bool set_samples) +{ + struct damon_target *t; + struct damon_region *r; + struct damon_probe *p; + + damon_for_each_target(t, ctx) { + damon_for_each_region(r, t) { + if (set_samples) + r->sampling_addr = damon_rand(ctx, r->ar.start, + r->ar.end); + damon_for_each_probe(p, ctx) + damon_pa_prep_probes_region(r, p, ctx); + } + } +} + static bool damon_pa_filter_match(struct damon_filter *filter, struct folio *folio) { @@ -448,6 +482,7 @@ static int __init damon_pa_initcall(void) .update = NULL, .prepare_access_checks = damon_pa_prepare_access_checks, .check_accesses = damon_pa_check_accesses, + .prep_probes = damon_pa_prep_probes, .apply_probes = damon_pa_apply_probes, .target_valid = NULL, .apply_scheme = damon_pa_apply_scheme, From 34a8c283a2f965983828163a7b248c535e411996 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:56 -0700 Subject: [PATCH 0384/1012] mm/damon/sysfs: implement preps directory Implement a sysfs directory named 'preps' under the probe directory. It will be evolved to be used for specifying probe preps. Link: https://lore.kernel.org/20260901132506.99243-10-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 65 +++++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 59 insertions(+), 6 deletions(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index c1ff739dab27a8..35fb308039b612 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -749,6 +749,35 @@ static const struct kobj_type damon_sysfs_intervals_ktype = { .default_groups = damon_sysfs_intervals_groups, }; +/* + * preps directory + */ + +struct damon_sysfs_preps { + struct kobject kobj; +}; + +static struct damon_sysfs_preps *damon_sysfs_preps_alloc(void) +{ + return kzalloc_obj(struct damon_sysfs_preps); +} + +static void damon_sysfs_preps_release(struct kobject *kobj) +{ + kfree(container_of(kobj, struct damon_sysfs_preps, kobj)); +} + +static struct attribute *damon_sysfs_preps_attrs[] = { + NULL, +}; +ATTRIBUTE_GROUPS(damon_sysfs_preps); + +static const struct kobj_type damon_sysfs_preps_ktype = { + .release = damon_sysfs_preps_release, + .sysfs_ops = &kobj_sysfs_ops, + .default_groups = damon_sysfs_preps_groups, +}; + /* * filter directory */ @@ -1069,6 +1098,7 @@ static const struct kobj_type damon_sysfs_filters_ktype = { struct damon_sysfs_probe { struct kobject kobj; unsigned int weight; + struct damon_sysfs_preps *preps; struct damon_sysfs_filters *filters; }; @@ -1079,25 +1109,48 @@ static struct damon_sysfs_probe *damon_sysfs_probe_alloc(void) static int damon_sysfs_probe_add_dirs(struct damon_sysfs_probe *probe) { + struct damon_sysfs_preps *preps; struct damon_sysfs_filters *filters; int err; - filters = damon_sysfs_filters_alloc(); - if (!filters) + preps = damon_sysfs_preps_alloc(); + if (!preps) return -ENOMEM; + probe->preps = preps; + + err = kobject_init_and_add(&preps->kobj, &damon_sysfs_preps_ktype, + &probe->kobj, "preps"); + if (err) + goto put_preps_out; + + filters = damon_sysfs_filters_alloc(); + if (!filters) { + err = -ENOMEM; + goto del_preps_out; + } probe->filters = filters; err = kobject_init_and_add(&filters->kobj, &damon_sysfs_filters_ktype, &probe->kobj, "filters"); - if (err) { - kobject_put(&filters->kobj); - probe->filters = NULL; - } + if (err) + goto put_filters_out; + return err; + +put_filters_out: + kobject_put(&filters->kobj); + probe->filters = NULL; +del_preps_out: + kobject_del(&preps->kobj); +put_preps_out: + kobject_put(&preps->kobj); + probe->preps = NULL; return err; } static void damon_sysfs_probe_rm_dirs(struct damon_sysfs_probe *probe) { + if (probe->preps) + kobject_put(&probe->preps->kobj); if (probe->filters) { damon_sysfs_filters_rm_dirs(probe->filters); kobject_put(&probe->filters->kobj); From 58cc07b8c464f42984347563b86c6884f2ce5d1b Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:57 -0700 Subject: [PATCH 0385/1012] mm/damon/sysfs: implement preps/nr_preps file Implement nr_preps file under the preps directory. It will be evolved to be used for generating sub directories that will represent each probe preparation action. Link: https://lore.kernel.org/20260901132506.99243-11-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 53 +++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 52 insertions(+), 1 deletion(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 35fb308039b612..b9722ebffd6f17 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -755,6 +755,7 @@ static const struct kobj_type damon_sysfs_intervals_ktype = { struct damon_sysfs_preps { struct kobject kobj; + int nr; }; static struct damon_sysfs_preps *damon_sysfs_preps_alloc(void) @@ -762,12 +763,60 @@ static struct damon_sysfs_preps *damon_sysfs_preps_alloc(void) return kzalloc_obj(struct damon_sysfs_preps); } +static void damon_sysfs_preps_rm_dirs(struct damon_sysfs_preps *preps) +{ + preps->nr = 0; +} + +static int damon_sysfs_preps_add_dirs(struct damon_sysfs_preps *preps, + int nr_preps) +{ + preps->nr = nr_preps; + return 0; +} + +static ssize_t nr_preps_show(struct kobject *kobj, struct kobj_attribute *attr, + char *buf) +{ + struct damon_sysfs_preps *preps = container_of(kobj, + struct damon_sysfs_preps, kobj); + + return sysfs_emit(buf, "%d\n", preps->nr); +} + +static ssize_t nr_preps_store(struct kobject *kobj, + struct kobj_attribute *attr, const char *buf, size_t count) +{ + struct damon_sysfs_preps *preps; + int nr, err = kstrtoint(buf, 0, &nr); + + if (err) + return err; + if (nr < 0) + return -EINVAL; + + preps = container_of(kobj, struct damon_sysfs_preps, kobj); + + if (!mutex_trylock(&damon_sysfs_lock)) + return -EBUSY; + err = damon_sysfs_preps_add_dirs(preps, nr); + mutex_unlock(&damon_sysfs_lock); + if (err) + return err; + + return count; +} + static void damon_sysfs_preps_release(struct kobject *kobj) { kfree(container_of(kobj, struct damon_sysfs_preps, kobj)); } +static struct kobj_attribute damon_sysfs_preps_nr_attr = + __ATTR_RW_MODE(nr_preps, 0600); + static struct attribute *damon_sysfs_preps_attrs[] = { + &damon_sysfs_preps_nr_attr.attr, NULL, }; ATTRIBUTE_GROUPS(damon_sysfs_preps); @@ -1149,8 +1198,10 @@ static int damon_sysfs_probe_add_dirs(struct damon_sysfs_probe *probe) static void damon_sysfs_probe_rm_dirs(struct damon_sysfs_probe *probe) { - if (probe->preps) + if (probe->preps) { + damon_sysfs_preps_rm_dirs(probe->preps); kobject_put(&probe->preps->kobj); + } if (probe->filters) { damon_sysfs_filters_rm_dirs(probe->filters); kobject_put(&probe->filters->kobj); From f55712890fd843a29df17edac0ec83144dd28799 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:58 -0700 Subject: [PATCH 0386/1012] mm/damon/sysfs: create directories for nr_preps writes Implement nr_preps write action to actually create subdirectories of the number. Each of the directory will be evolved to represent each probe preparation action. Link: https://lore.kernel.org/20260901132506.99243-12-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 80 +++++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 79 insertions(+), 1 deletion(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index b9722ebffd6f17..4ff473b6895355 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -749,12 +749,50 @@ static const struct kobj_type damon_sysfs_intervals_ktype = { .default_groups = damon_sysfs_intervals_groups, }; +/* + * prep directory + */ + +struct damon_sysfs_prep { + struct kobject kobj; +}; + +static struct damon_sysfs_prep *damon_sysfs_prep_alloc(void) +{ + struct damon_sysfs_prep *prep; + + prep = kzalloc_obj(struct damon_sysfs_prep); + if (!prep) + return prep; + return prep; +} + +static void damon_sysfs_prep_release(struct kobject *kobj) +{ + struct damon_sysfs_prep *prep = container_of(kobj, + struct damon_sysfs_prep, kobj); + + kfree(prep); +} + +static struct attribute *damon_sysfs_prep_attrs[] = { + NULL, +}; +ATTRIBUTE_GROUPS(damon_sysfs_prep); + +static const struct kobj_type damon_sysfs_prep_ktype = { + .release = damon_sysfs_prep_release, + .sysfs_ops = &kobj_sysfs_ops, + .default_groups = damon_sysfs_prep_groups, +}; + /* * preps directory */ struct damon_sysfs_preps { struct kobject kobj; + struct damon_sysfs_prep **preps_arr; int nr; }; @@ -765,13 +803,53 @@ static struct damon_sysfs_preps *damon_sysfs_preps_alloc(void) static void damon_sysfs_preps_rm_dirs(struct damon_sysfs_preps *preps) { + struct damon_sysfs_prep **preps_arr = preps->preps_arr; + int i; + + for (i = 0; i < preps->nr; i++) { + kobject_del(&preps_arr[i]->kobj); + kobject_put(&preps_arr[i]->kobj); + } preps->nr = 0; + kfree(preps_arr); + preps->preps_arr = NULL; } static int damon_sysfs_preps_add_dirs(struct damon_sysfs_preps *preps, int nr_preps) { - preps->nr = nr_preps; + struct damon_sysfs_prep **preps_arr, *prep; + int err, i; + + damon_sysfs_preps_rm_dirs(preps); + if (!nr_preps) + return 0; + + preps_arr = kmalloc_objs(*preps_arr, nr_preps, + GFP_KERNEL | __GFP_NOWARN); + if (!preps_arr) + return -ENOMEM; + preps->preps_arr = preps_arr; + + for (i = 0; i < nr_preps; i++) { + prep = damon_sysfs_prep_alloc(); + if (!prep) { + damon_sysfs_preps_rm_dirs(preps); + return -ENOMEM; + } + + err = kobject_init_and_add(&prep->kobj, + &damon_sysfs_prep_ktype, &preps->kobj, "%d", + i); + if (err) { + kobject_put(&prep->kobj); + damon_sysfs_preps_rm_dirs(preps); + return err; + } + + preps_arr[i] = prep; + preps->nr++; + } return 0; } From 4ac7c56df7b3bcdcd787d5565e3648a37c38ef73 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:24:59 -0700 Subject: [PATCH 0387/1012] mm/damon/sysfs: implement prep_action file Add a file named prep_action under the prep directory. It represents the corresponding probe preparation action. Link: https://lore.kernel.org/20260901132506.99243-13-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 57 ++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 57 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 4ff473b6895355..2118089dd9d8dd 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -755,6 +755,7 @@ static const struct kobj_type damon_sysfs_intervals_ktype = { struct damon_sysfs_prep { struct kobject kobj; + enum damon_prep_action action; }; static struct damon_sysfs_prep *damon_sysfs_prep_alloc(void) @@ -764,9 +765,61 @@ static struct damon_sysfs_prep *damon_sysfs_prep_alloc(void) prep = kzalloc_obj(struct damon_sysfs_prep); if (!prep) return prep; + prep->action = DAMON_PREP_SET_PGIDLE; return prep; } +struct damon_sysfs_prep_action_name { + const enum damon_prep_action action; + const char *name; +}; + +static const struct damon_sysfs_prep_action_name +damon_sysfs_prep_action_names[] = { + { + .action = DAMON_PREP_SET_PGIDLE, + .name = "set_pgidle", + }, +}; + +static ssize_t prep_action_show(struct kobject *kobj, + struct kobj_attribute *attr, char *buf) +{ + struct damon_sysfs_prep *prep = container_of(kobj, + struct damon_sysfs_prep, kobj); + int i; + + for (i = 0; i < ARRAY_SIZE(damon_sysfs_prep_action_names); i++) { + const struct damon_sysfs_prep_action_name *action_name; + + action_name = &damon_sysfs_prep_action_names[i]; + if (action_name->action == prep->action) + return sysfs_emit(buf, "%s\n", action_name->name); + } + return -EINVAL; +} + +static ssize_t prep_action_store(struct kobject *kobj, + struct kobj_attribute *attr, const char *buf, size_t count) +{ + struct damon_sysfs_prep *prep = container_of(kobj, + struct damon_sysfs_prep, kobj); + ssize_t ret = -EINVAL; + int i; + + for (i = 0; i < ARRAY_SIZE(damon_sysfs_prep_action_names); i++) { + const struct damon_sysfs_prep_action_name *action_name; + + action_name = &damon_sysfs_prep_action_names[i]; + if (sysfs_streq(buf, action_name->name)) { + prep->action = action_name->action; + ret = count; + break; + } + } + return ret; +} + static void damon_sysfs_prep_release(struct kobject *kobj) { struct damon_sysfs_prep *prep = container_of(kobj, @@ -775,7 +828,11 @@ static void damon_sysfs_prep_release(struct kobject *kobj) kfree(prep); } +static struct kobj_attribute damon_sysfs_prep_prep_action_attr = + __ATTR_RW_MODE(prep_action, 0600); + static struct attribute *damon_sysfs_prep_attrs[] = { + &damon_sysfs_prep_prep_action_attr.attr, NULL, }; ATTRIBUTE_GROUPS(damon_sysfs_prep); From eb40b6f28b3b49322c7ac42263a1e1d3bc78aae4 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:00 -0700 Subject: [PATCH 0388/1012] mm/damon/sysfs: pass preps to DAMON core DAMON sysfs interface provides the files for setting DAMON probe preps. But the underlying code is not really passing the user-set values to DAMON core. Pass those. Link: https://lore.kernel.org/20260901132506.99243-14-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 28 ++++++++++++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 2118089dd9d8dd..7f340b6f1921b9 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -812,7 +812,10 @@ static ssize_t prep_action_store(struct kobject *kobj, action_name = &damon_sysfs_prep_action_names[i]; if (sysfs_streq(buf, action_name->name)) { + if (!mutex_trylock(&damon_sysfs_lock)) + return -EBUSY; prep->action = action_name->action; + mutex_unlock(&damon_sysfs_lock); ret = count; break; } @@ -2175,6 +2178,23 @@ static int damon_sysfs_set_attrs(struct damon_ctx *ctx, return damon_set_attrs(ctx, &attrs); } +static int damon_sysfs_set_preps(struct damon_probe *probe, + struct damon_sysfs_preps *sys_preps) +{ + int i; + + for (i = 0; i < sys_preps->nr; i++) { + struct damon_sysfs_prep *sys_prep = sys_preps->preps_arr[i]; + struct damon_prep *prep; + + prep = damon_new_prep(sys_prep->action); + if (!prep) + return -ENOMEM; + damon_add_prep(probe, prep); + } + return 0; +} + static int damon_sysfs_set_filters(struct damon_probe *probe, struct damon_sysfs_filters *sys_filters) { @@ -2210,7 +2230,15 @@ static int damon_sysfs_set_probe(struct damon_probe *probe, struct damon_sysfs_probe *sys_probe) { struct damon_sysfs_filters *sys_filters; + struct damon_sysfs_preps *sys_preps; + int err; + sys_preps = sys_probe->preps; + if (sys_preps) { + err = damon_sysfs_set_preps(probe, sys_preps); + if (err) + return err; + } sys_filters = sys_probe->filters; if (!sys_filters) return 0; From d02f4c29b651b51a5e91f04331bf4594b56290fe Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:01 -0700 Subject: [PATCH 0389/1012] selftests/damon/sysfs.sh: test probe prep sysfs files Add basic file operations test for newly introduced DAMON probe prep sysfs directories and files. Link: https://lore.kernel.org/20260901132506.99243-15-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/damon/sysfs.sh | 27 ++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/tools/testing/selftests/damon/sysfs.sh b/tools/testing/selftests/damon/sysfs.sh index f7fb94b84e716d..ddebde6edabe4e 100755 --- a/tools/testing/selftests/damon/sysfs.sh +++ b/tools/testing/selftests/damon/sysfs.sh @@ -361,6 +361,32 @@ test_intervals() test_intervals_goal "$intervals_dir/intervals_goal" } +test_damon_prep() +{ + damon_prep_dir=$1 + ensure_file "$damon_prep_dir/prep_action" "exist" "600" + ensure_write_succ "$damon_prep_dir/prep_action" "set_pgidle" \ + "valid input" + ensure_write_fail "$damon_prep_dir/prep_action" "foo" "invalid input" +} + +test_damon_preps() +{ + preps_dir=$1 + ensure_dir "$preps_dir" "exist" + ensure_file "$preps_dir/nr_preps" "exist" "600" + ensure_write_succ "$preps_dir/nr_preps" "1" "valid input" + test_damon_prep "$preps_dir/0" + + ensure_write_succ "$preps_dir/nr_preps" "2" "valid input" + test_damon_prep "$preps_dir/0" + test_damon_prep "$preps_dir/1" + + ensure_write_succ "$preps_dir/nr_preps" "0" "valid input" + ensure_dir "$preps_dir/0" "not_exist" + ensure_dir "$preps_dir/1" "not_exist" +} + test_damon_filter() { damon_filter_dir=$1 @@ -392,6 +418,7 @@ test_probe() { probe_dir=$1 ensure_dir "$probe_dir" "exist" + test_damon_preps "$probe_dir/preps" test_damon_filters "$probe_dir/filters" } From 6b3f3e5e1fb57a3854f8c53c0919d786a7029375 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:02 -0700 Subject: [PATCH 0390/1012] Docs/mm/damon/design: document probe preps Update DAMON design document for the newly added DAMON probe preps feature. Link: https://lore.kernel.org/20260901132506.99243-16-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 947d91ae24a392..d036340dae8afb 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -309,6 +309,13 @@ Users can therefore know how much of a given DAMON region has a specific data attribute by reading the per-region per-probe probe hits counter after each aggregation interval. +Users can optionally register probing preparation actions per probe. If such +actions are registered, DAMON applies the actions to each region's sampling +memory before starting the next sampling interval. Currently only one action, +``set_pgidle`` is supported. The action marks the page for the probing target +memory as access-idle. This can be useful to be used together with +``pgidle_unset`` probe filter. + This is a sampling based mechanism. Hence, it is lightweight but the output may include some measurement errors. The output should be used with good understanding of statistics. From 3a501c7b19ce980c04cfc757ab052f5e36615d22 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:03 -0700 Subject: [PATCH 0391/1012] Docs/admin-guide/mm/damon/usage: document probe preps sysfs files Update DAMON usage document for the newly added DAMON probe preps sysfs files. Link: https://lore.kernel.org/20260901132506.99243-17-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/admin-guide/mm/damon/usage.rst | 18 +++++++++++++++--- 1 file changed, 15 insertions(+), 3 deletions(-) diff --git a/Documentation/admin-guide/mm/damon/usage.rst b/Documentation/admin-guide/mm/damon/usage.rst index da5f9afd08aef8..023c6334024f8e 100644 --- a/Documentation/admin-guide/mm/damon/usage.rst +++ b/Documentation/admin-guide/mm/damon/usage.rst @@ -74,6 +74,9 @@ comma (","). │ │ │ │ │ │ nr_regions/min,max │ │ │ │ │ │ :ref:`probes `/nr_probes │ │ │ │ │ │ │ 0/weight + │ │ │ │ │ │ │ │ preps/nr_preps + │ │ │ │ │ │ │ │ │ 0/prep_action + │ │ │ │ │ │ │ │ │ ... │ │ │ │ │ │ │ │ filters/nr_filters │ │ │ │ │ │ │ │ │ 0/type,matching,allow,path │ │ │ │ │ │ │ │ │ ... @@ -283,9 +286,18 @@ In the beginning, this directory has only one file, ``nr_probes``. Writing a number (``N``) to the file creates the number of child directories named ``0`` to ``N-1``. Each directory represents each monitoring probe. -In each probe directory, one directory, ``filters`` exists. The directory -contains files for installing filters for the probe, that is used to determine -the data attribute for the probe. +In each probe directory, two directories, ``preps`` and ``filters`` exist. The +directories contain files for installing probing preparation actions and +filters for the probe, that are used to determine the data attribute for the +probe. + +In the beginning, ``preps`` directory has only one file, ``nr_preps``. +Writing a number (``N``) to the file creates the number of child directories +named ``0`` to ``N-1``. Each directory represents each preparation action. +Each directory has one file, ``prep_action``. The preparation action can be +selected by writing the name of the action to the ``prep_action`` file. Refer +to the :ref:`design doc ` for the list of +supported actions. Each probe directory also contains ``weight`` file. Reading from and writing to the file gets and sets the :ref:`attributes-only monitoring From d3838cf97e5b2381dacd5a71ed8e20d9352d3923 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 06:25:04 -0700 Subject: [PATCH 0392/1012] Docs/ABI/damon: document probe prep sysfs files Update DAMON ABI document for the newly added DAMON probe prep sysfs files. Link: https://lore.kernel.org/20260901132506.99243-18-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/ABI/testing/sysfs-kernel-mm-damon | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/Documentation/ABI/testing/sysfs-kernel-mm-damon b/Documentation/ABI/testing/sysfs-kernel-mm-damon index e675a57145e36d..f8d2601e829047 100644 --- a/Documentation/ABI/testing/sysfs-kernel-mm-damon +++ b/Documentation/ABI/testing/sysfs-kernel-mm-damon @@ -173,6 +173,19 @@ Contact: SJ Park Description: Writing to and reading from this file sets and gets the per-probe attribute weight. +What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/preps/nr_preps +Date: Jun 2026 +Contact: SJ Park +Description: Writing a number 'N' to this file creates the number of + directories for each DAMON probing preparation action named '0' + to 'N-1' under the preps/ directory. + +What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/preps//prep_action +Date: Jun 2026 +Contact: SJ Park +Description: Writing to and reading from this file sets and gets the probing + preparation action. + What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/filters/nr_filters Date: May 2026 Contact: SJ Park From 7cc4120ffa69b71a493efdd436764facc650a386 Mon Sep 17 00:00:00 2001 From: Qinyun Tan Date: Tue, 1 Sep 2026 19:51:04 +0800 Subject: [PATCH 0393/1012] mm/list_lru: don't copy stale shrinker id from non-memcg-aware shrinkers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit With cgroup.memory=nokmem, shrinker_memcg_alloc() fails with -ENOSYS for shrinkers without SHRINKER_NONSLAB, and shrinker_alloc() falls back to a non-memcg-aware shrinker. On this fallback path, shrinker->id is never assigned and keeps 0 from kzalloc(), which is a valid id belonging to whichever memcg-aware shrinker registers first. __list_lru_init() copies shrinker->id unconditionally, so every list_lru backed by such a fallback shrinker (thp-deferred_split, zswap-shrinker, workingset shadow nodes, superblock lrus, ...) ends up with lru->shrinker_id == 0 instead of -1. Under nokmem the list_lru collapses to the shared per-node lists, but __list_lru_add() still calls set_shrinker_bit() against the memcg of the added object. Most list_lru users are unaffected because their objects resolve to a NULL memcg without kmem accounting, but the THP deferred split queue holds user folios, which are charged regardless of nokmem. On a system where no SHRINKER_NONSLAB shrinker registers, shrinker_nr_max stays 0 and every memcg's shrinker_info has map_nr_max == 0, so the first folio added by khugepaged triggers on every boot: WARNING: mm/shrinker.c:212 at set_shrinker_bit+0x99/0xa0 On systems where a SHRINKER_NONSLAB shrinker (btrfs, xfs) did register and expand the maps, there is no warning; instead bit 0 is set spuriously for an unrelated shrinker. shrinker->id is only meaningful while SHRINKER_MEMCG_AWARE is set, and all readers inside mm/shrinker.c already check the flag before using the id. Make __list_lru_init() do the same and fall back to -1, so set_shrinker_bit() is never reached with a bogus id. The stale shrinker->id itself is left as is; cleaning that up is a separate topic. Verified on a machine booting with cgroup.memory=nokmem and CONFIG_TRANSPARENT_HUGEPAGE=y: the warning fires once per boot from khugepaged, disappears when nokmem is removed from the command line, and no longer triggers with this fix applied and nokmem set. Link: https://lore.kernel.org/20260901115104.2944996-1-qinyuntan@linux.alibaba.com Fixes: fafaeceb89a5e ("mm: switch deferred split shrinker to list_lru") Signed-off-by: Qinyun Tan Signed-off-by: Andrew Morton Acked-by: Muchun Song Reviewed-by: Baolin Wang Reported-by: Wentao Guan Closes: https://lore.kernel.org/linux-mm/20260904190028.21542-1-guanwentao@uniontech.com/ Tested-by: Wentao Guan Cc: Michal Hocko Cc: Roman Gushchin Cc: Johannes Weiner Cc: Shakeel Butt Cc: Dave Chinner Cc: David Hildenbrand Cc: Lance Yang Cc: Michal Koutný Cc: Xunlei Pang Cc: --- mm/list_lru.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/mm/list_lru.c b/mm/list_lru.c index a4522ca93ebcb9..8a6dd0a489e12c 100644 --- a/mm/list_lru.c +++ b/mm/list_lru.c @@ -666,7 +666,12 @@ int __list_lru_init(struct list_lru *lru, bool memcg_aware, struct shrinker *shr int i; #ifdef CONFIG_MEMCG - if (shrinker) + /* + * If the shrinker fell back to being non-memcg-aware (e.g. with + * cgroup.memory=nokmem), its id was never assigned and holds a + * stale 0. Don't let set_shrinker_bit() act on it. + */ + if (shrinker && (shrinker->flags & SHRINKER_MEMCG_AWARE)) lru->shrinker_id = shrinker->id; else lru->shrinker_id = -1; From 07170f0f060abc09550142f81b07f3a3d1e76bb7 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 1 Sep 2026 15:38:04 -0400 Subject: [PATCH 0394/1012] mm: remove unused mark_page_reserved() Patch series "mm: remove three unused helpers from mm.h", v2. I happened to notice these were unused. Two of them are relatively recently unused, and one has been unused for a few years. Remove them. This patch (of 2): mark_page_reserved() lost its last caller in commit 6215d9f4470f ("arch, mm: consolidate empty_zero_page"). Remove it. Link: https://lore.kernel.org/20260901-mm-remove-unused-helpers-v2-0-f6474e169c23@columbia.edu Link: https://lore.kernel.org/20260901-mm-remove-unused-helpers-v2-1-f6474e169c23@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mm.h | 6 ------ 1 file changed, 6 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 1b28e6fc8d5dd1..e3736c42c4dbc0 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4080,12 +4080,6 @@ static inline void free_reserved_page(struct page *page) free_reserved_pages(page, 0); } -static inline void mark_page_reserved(struct page *page) -{ - SetPageReserved(page); - adjust_managed_page_count(page, -1); -} - static inline void free_reserved_ptdesc(struct ptdesc *pt) { free_reserved_page(ptdesc_page(pt)); From edbb27cc4a0c05402ea48b547be0e5597d2cb594 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 1 Sep 2026 15:38:05 -0400 Subject: [PATCH 0395/1012] mm: remove unused totalram_pages_inc() and totalram_pages_dec() totalram_pages_inc() and totalram_pages_dec() have had no callers since commit 7fbc5e26123e ("memblock: extract page freeing from free_reserved_area() into a helper") and commit 287b89773d81 ("powerpc/pseries/cmm: Use adjust_managed_page_count() insted of totalram_pages_*"), respectively. Remove them. Drop the totalram_pages_inc() stub from tools mm.h too. Link: https://lore.kernel.org/20260901-mm-remove-unused-helpers-v2-2-f6474e169c23@columbia.edu Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/mm.h | 10 ---------- tools/include/linux/mm.h | 4 ---- 2 files changed, 14 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index e3736c42c4dbc0..c105a3758915b2 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -57,16 +57,6 @@ static inline unsigned long totalram_pages(void) return (unsigned long)atomic_long_read(&_totalram_pages); } -static inline void totalram_pages_inc(void) -{ - atomic_long_inc(&_totalram_pages); -} - -static inline void totalram_pages_dec(void) -{ - atomic_long_dec(&_totalram_pages); -} - static inline void totalram_pages_add(long count) { atomic_long_add(count, &_totalram_pages); diff --git a/tools/include/linux/mm.h b/tools/include/linux/mm.h index 84b5954f66c3d5..d586a510e6e18b 100644 --- a/tools/include/linux/mm.h +++ b/tools/include/linux/mm.h @@ -33,10 +33,6 @@ static inline phys_addr_t virt_to_phys(volatile void *address) return (phys_addr_t)address; } -static inline void totalram_pages_inc(void) -{ -} - static inline void totalram_pages_add(long count) { } From 62c3f32df941c11e7bee0ca4b13d357e53da59c6 Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 2 Sep 2026 01:53:01 +0900 Subject: [PATCH 0396/1012] percpu: remove redundant assignments to bits Patch series "percpu: remove code with no effect". While reading the percpu initialization path, I found some minor cleanups for unnecessary initializations and an obsolete return statement. No functional change. This patch (of 4): pcpu_chunk_refresh_hint() and pcpu_find_block_fit() set bits to 0 and later call pcpu_next_md_free_region() or pcpu_next_fit_region(), which unconditionally set *bits to 0. Nothing uses it in between, so the assignments have no effect. Remove redundant assignments. No functional change. Link: https://lore.kernel.org/20260901165307.1026248-1-ekffu200098@gmail.com Link: https://lore.kernel.org/20260901165307.1026248-2-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Cc: Dennis Zhou Cc: Tejun Heo Cc: Christoph Lameter --- mm/percpu.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/percpu.c b/mm/percpu.c index 47a903fe3b5124..b094c617147cb5 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -758,7 +758,6 @@ static void pcpu_chunk_refresh_hint(struct pcpu_chunk *chunk, bool full_scan) chunk_md->contig_hint = 0; } - bits = 0; pcpu_for_each_md_free_region(chunk, bit_off, bits) pcpu_block_update(chunk_md, bit_off, bit_off + bits); } @@ -1122,7 +1121,6 @@ static int pcpu_find_block_fit(struct pcpu_chunk *chunk, int alloc_bits, return -1; bit_off = pcpu_next_hint(chunk_md, alloc_bits); - bits = 0; pcpu_for_each_fit_region(chunk, alloc_bits, align, bit_off, bits) { if (!pop_only || pcpu_is_populated(chunk, bit_off, bits, &next_off)) From da105c9837380eee9034621301ccb734808d4c7b Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 2 Sep 2026 01:53:02 +0900 Subject: [PATCH 0397/1012] percpu: remove unnecessary initialization in pcpu_build_alloc_info() pcpu_build_alloc_info() initializes nr_groups to 1 and unconditionally sets it to the number of groups. Nothing uses it in between, so the initialization has no effect. Remove unnecessary initialization. No functional change. Link: https://lore.kernel.org/20260901165307.1026248-3-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Cc: Christoph Lameter Cc: Dennis Zhou Cc: Tejun Heo --- mm/percpu.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/percpu.c b/mm/percpu.c index b094c617147cb5..0ec9b2adfc208e 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -2817,7 +2817,7 @@ static struct pcpu_alloc_info * __init __flatten pcpu_build_alloc_info( static int group_cnt[NR_CPUS] __initdata; static struct cpumask mask __initdata; const size_t static_size = __per_cpu_end - __per_cpu_start; - int nr_groups = 1, nr_units = 0; + int nr_groups, nr_units = 0; size_t size_sum, min_unit_size, alloc_size; int upa, max_upa, best_upa; /* units_per_alloc */ int last_allocs, group, unit; From c8981817d51f20975fc8b9c38df59688e039bd6b Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 2 Sep 2026 01:53:03 +0900 Subject: [PATCH 0398/1012] percpu: remove unnecessary cpumask_clear() in pcpu_build_alloc_info() pcpu_build_alloc_info() clears mask and unconditionally sets it to cpu_possible_mask. Nothing uses it in between, so cpumask_clear() has no effect. Remove unnecessary cpumask_clear(). No functional change. Link: https://lore.kernel.org/20260901165307.1026248-4-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Cc: Christoph Lameter Cc: Dennis Zhou Cc: Tejun Heo --- mm/percpu.c | 1 - 1 file changed, 1 deletion(-) diff --git a/mm/percpu.c b/mm/percpu.c index 0ec9b2adfc208e..fe9fe5a0c3d597 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -2828,7 +2828,6 @@ static struct pcpu_alloc_info * __init __flatten pcpu_build_alloc_info( /* this function may be called multiple times */ memset(group_map, 0, sizeof(group_map)); memset(group_cnt, 0, sizeof(group_cnt)); - cpumask_clear(&mask); /* calculate size_sum and ensure dyn_size is enough for early alloc */ size_sum = PFN_ALIGN(static_size + reserved_size + From a5b9f9fcc97850cce693a6667977868d76bb3db2 Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Wed, 2 Sep 2026 01:53:04 +0900 Subject: [PATCH 0399/1012] percpu: remove unnecessary return in pcpu_populate_pte() Since commit c6f239796b55 ("mm/memblock: add memblock_alloc_or_panic interface"), pcpu_populate_pte() no longer has an error label after the return statement, so the return statement has no effect. Remove unnecessary return. No functional change. Link: https://lore.kernel.org/20260901165307.1026248-5-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Cc: Christoph Lameter Cc: Dennis Zhou Cc: Tejun Heo --- mm/percpu.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/percpu.c b/mm/percpu.c index fe9fe5a0c3d597..3eff382e565ad5 100644 --- a/mm/percpu.c +++ b/mm/percpu.c @@ -3178,8 +3178,6 @@ void __init __weak pcpu_populate_pte(unsigned long addr) new = memblock_alloc_or_panic(PTE_TABLE_SIZE, PTE_TABLE_SIZE); pmd_populate_kernel(&init_mm, pmd, new); } - - return; } /** From 609f0695dff8be787ee766205b0d9d6b14429ee2 Mon Sep 17 00:00:00 2001 From: Sergey Senozhatsky Date: Tue, 1 Sep 2026 14:13:05 +0900 Subject: [PATCH 0400/1012] zram: remove unreachable kernel_read_file_from_path() return check Sashiko reported that: kernel_read_file_from_path() returns negative error for zero-sized files, so we cannot have "sz == 0" return, remove it and use a generic error message instead. Link: https://lore.kernel.org/20260901051335.2202390-1-senozhatsky@chromium.org Signed-off-by: Sergey Senozhatsky Signed-off-by: Andrew Morton Cc: Haoqin Huang --- drivers/block/zram/zram_drv.c | 5 ----- 1 file changed, 5 deletions(-) diff --git a/drivers/block/zram/zram_drv.c b/drivers/block/zram/zram_drv.c index 4ba0f77b2abd80..2359eaa6f53184 100644 --- a/drivers/block/zram/zram_drv.c +++ b/drivers/block/zram/zram_drv.c @@ -1722,11 +1722,6 @@ static int comp_params_store(struct zram *zram, u32 prio, s32 level, dict_path, sz); return sz; } - if (sz == 0) { - pr_err("failed to load dictionary %s (empty file)\n", - dict_path); - return -EINVAL; - } } zram->params[prio].dict_sz = sz; From 8df5eee58d33382c2593d5c3d81d43550bd87d32 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 17:27:20 -0700 Subject: [PATCH 0401/1012] mm/damon/tests/core-kunit: test committing psi goal to psi goal Patch series "mm/damon: fix misc bugs in kunit, quota goals and sysfs refresh_ms". DAMOS quota goals commit unit test is mistakenly not testing a test case that was designed to test. DAMOS quota goals and DAMON sysfs refresh_ms file have bugs that can produce non critical but still unexpected behaviors. Fix the bugs. Patch 1 fixes the DAMOS quota goals commit unit test to cover a mistakenly uncovered case. Patches 2 and 3 fix the bugs in DAMOS PSI goal initialization and eligible_mem_bp online commit, respectively. Patch 4 fixes the bug in DAMON sysfs refresh_ms file handling. This patch (of 4): damon_test_commit_quota_goal_for() is set to test committing a new psi goal on an existing psi goal. However, damon_test_commit_quota_goal() is mistakenly not covering the test case. Add the test case. Link: https://lore.kernel.org/20260902002725.108635-1-sj@kernel.org Link: https://lore.kernel.org/20260902002725.108635-2-sj@kernel.org Fixes: 99f89debafc5 ("mm/damon/tests/core-kunit: add damos_commit_quota_goal() test") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Lian Wang Tested-by: Lian Wang Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: Quanmin Yan Cc: # 6.19.x --- mm/damon/tests/core-kunit.h | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 7071ec277b0072..f1e11548c771b7 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -793,6 +793,13 @@ static void damos_test_commit_quota_goal(struct kunit *test) .last_psi_total = 456, }; + damos_test_commit_quota_goal_for(test, &dst, + &(struct damos_quota_goal) { + .metric = DAMOS_QUOTA_SOME_MEM_PSI_US, + .target_value = 234, + .current_value = 345, + .last_psi_total = 567, + }); damos_test_commit_quota_goal_for(test, &dst, &(struct damos_quota_goal){ .metric = DAMOS_QUOTA_USER_INPUT, From 19b0f24a1dd36049cecc6bb179459666594f575e Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 17:27:21 -0700 Subject: [PATCH 0402/1012] mm/damon/core: handle uninitialized damos_quota_goal->last_psi_total When DAMOS_QUOTA_SOME_MEM_PSI_US metric damos quota goal is set, the PSI delta for the feedback loop is calculated using damos_quota_goal->last_psi_total. However, it is initialized only after the first feedback loop. The first iteration of the loop uses the uninitialized value. As a result, the feedback loop can change the effective quota in an unexpected way at the first iteration. The user impact of the issue is not big, because the issue impacts only the first iteration of the feedback loop. The feedback loop also has an internal cap of the quota adjustment. The wrong adjustment will soon be corrected over a few iterations. For this reason, doing no initialization at commit time was intentional. It is also explicitly commented. That said, nobody likes behaviors that are unexpected or difficult to be expected. Check last_psi_total initialization and skip the tuning round when it is not initialized. For this, initialize last_psi_total with U64_MAX in the goal creation and the goal commit time. U64_MAX means the field is not initialized. The tuning round shows the value and adjusts it to guarantee the current quota is maintained for the round, and last_psi_total is correctly initialized on the next round. Before this change, committing a new PSI goal on an existing PSI goal with goal-only DAMON sysfs command (commit_schemes_quota_goals) just worked. After this change, the tuning round right after the commit will be unnecessarily skipped, because last_psi_total is unconditionally marked as not initialized in the damos_commit_quota_goal_union(). This is an intended tradeoff for simplicity. Skipping just one round of tuning is no problem. Meanwhile it makes both the code and the behavior simple to understand. Also update the quota goal commit unit test for changed last_psi_total setup behavior. The issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260902002725.108635-3-sj@kernel.org Link: https://lore.kernel.org/20260718005316.89585-1-sj@kernel.org [1] Fixes: 2dbb60f789cb ("mm/damon/core: implement PSI metric DAMOS quota goal") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Lian Wang Tested-by: Lian Wang Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: Quanmin Yan Cc: # 6.9.x --- mm/damon/core.c | 13 +++++++++++-- mm/damon/tests/core-kunit.h | 9 +++------ 2 files changed, 14 insertions(+), 8 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 5ee2d0448a8091..f950a3b9fcb99f 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -697,6 +697,8 @@ struct damos_quota_goal *damos_new_quota_goal( return NULL; goal->metric = metric; goal->target_value = target_value; + if (metric == DAMOS_QUOTA_SOME_MEM_PSI_US) + goal->last_psi_total = U64_MAX; INIT_LIST_HEAD(&goal->list); return goal; } @@ -1190,6 +1192,9 @@ static void damos_commit_quota_goal_union( struct damos_quota_goal *dst, struct damos_quota_goal *src) { switch (dst->metric) { + case DAMOS_QUOTA_SOME_MEM_PSI_US: + dst->last_psi_total = U64_MAX; + break; case DAMOS_QUOTA_NODE_MEM_USED_BP: case DAMOS_QUOTA_NODE_MEM_FREE_BP: dst->nid = src->nid; @@ -1211,7 +1216,6 @@ static void damos_commit_quota_goal( dst->target_value = src->target_value; if (dst->metric == DAMOS_QUOTA_USER_INPUT) dst->current_value = src->current_value; - /* keep last_psi_total as is, since it will be updated in next cycle */ damos_commit_quota_goal_union(dst, src); } @@ -3139,7 +3143,12 @@ static void damos_set_quota_goal_current_value(struct damon_ctx *c, break; case DAMOS_QUOTA_SOME_MEM_PSI_US: now_psi_total = damos_get_some_mem_psi_total(); - goal->current_value = now_psi_total - goal->last_psi_total; + /* uninitialized last_psi_total; make no effect this round */ + if (goal->last_psi_total == U64_MAX) + goal->current_value = goal->target_value; + else + goal->current_value = now_psi_total - + goal->last_psi_total; goal->last_psi_total = now_psi_total; break; case DAMOS_QUOTA_NODE_MEM_USED_BP: diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index f1e11548c771b7..af26b3d60957b5 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -757,19 +757,16 @@ static void damos_test_commit_quota_goal_for(struct kunit *test, struct damos_quota_goal *dst, struct damos_quota_goal *src) { - u64 dst_last_psi_total = 0; - - if (dst->metric == DAMOS_QUOTA_SOME_MEM_PSI_US) - dst_last_psi_total = dst->last_psi_total; damos_commit_quota_goal(dst, src); KUNIT_EXPECT_EQ(test, dst->metric, src->metric); KUNIT_EXPECT_EQ(test, dst->target_value, src->target_value); if (src->metric == DAMOS_QUOTA_USER_INPUT) KUNIT_EXPECT_EQ(test, dst->current_value, src->current_value); - if (dst_last_psi_total && src->metric == DAMOS_QUOTA_SOME_MEM_PSI_US) - KUNIT_EXPECT_EQ(test, dst->last_psi_total, dst_last_psi_total); switch (dst->metric) { + case DAMOS_QUOTA_SOME_MEM_PSI_US: + KUNIT_EXPECT_EQ(test, dst->last_psi_total, U64_MAX); + break; case DAMOS_QUOTA_NODE_MEM_USED_BP: case DAMOS_QUOTA_NODE_MEM_FREE_BP: KUNIT_EXPECT_EQ(test, dst->nid, src->nid); From be171e617d203c2269eda2c31a88250d2ca707c6 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Tue, 22 Sep 2026 22:51:37 -0700 Subject: [PATCH 0403/1012] mm/damon/core: keep the temporal tuner quota over an unmeasured PSI round The patch in mm-unstable of subject "mm/damon/core: handle uninitialized damos_quota_goal->last_psi_total" scores a PSI quota goal without a previous sample as achieved. This leaves the consist tuner's input unchanged, but the temporal tuner sets its quota to zero for an achieved goal. A running scheme with a nonzero temporal quota therefore loses a charge window after a quota-goal commit. Use the effective quota to preserve the temporal tuner's previous goal-achievement state during an unmeasured round, as SJ suggested [1]. Score the goal as achieved when the effective quota is zero and as not achieved otherwise. A new scheme's initially zero quota stays zero, and the consist tuner's behavior is unchanged. Move the PSI current-value calculation and last_psi_total update into a helper that takes the current PSI total. This lets a unit test cover the unmeasured and measured rounds without depending on system memory pressure. Link: https://lore.kernel.org/20260923055139.2982-1-sj@kernel.org Link: https://lore.kernel.org/damon/20260916001311.101024-1-sj@kernel.org/ [1] Fixes: 2dbb60f789cb ("mm/damon/core: implement PSI metric DAMOS quota goal") Signed-off-by: Karl Mehltretter Signed-off-by: SJ Park Signed-off-by: Andrew Morton Suggested-by: SJ Park Reviewed-by: SJ Park Reviewed-by: Kunwu Chan Assisted-by: LLM Cc: Lian Wang Cc: # 6.9.x --- mm/damon/core.c | 30 +++++++++++++++++++++++------- 1 file changed, 23 insertions(+), 7 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index f950a3b9fcb99f..71e8cedbc64399 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2891,6 +2891,28 @@ static inline u64 damos_get_some_mem_psi_total(void) #endif /* CONFIG_PSI */ +static void damos_set_psi_current_val(u64 now_psi_total, + struct damos_quota_goal *goal, struct damos *s) +{ + u64 last_psi_total = goal->last_psi_total; + + goal->last_psi_total = now_psi_total; + if (last_psi_total != U64_MAX) { + goal->current_value = now_psi_total - last_psi_total; + return; + } + /* uninitialized last_psi_total; make no effect this round */ + if (s->quota.goal_tuner == DAMOS_QUOTA_GOAL_TUNER_CONSIST) { + goal->current_value = goal->target_value; + return; + } + /* let temporal tuner show the same achievement as in the last round */ + if (!s->quota.esz) + goal->current_value = goal->target_value; + else + goal->current_value = 0; +} + #ifdef CONFIG_NUMA static bool invalid_mem_node(int nid) { @@ -3143,13 +3165,7 @@ static void damos_set_quota_goal_current_value(struct damon_ctx *c, break; case DAMOS_QUOTA_SOME_MEM_PSI_US: now_psi_total = damos_get_some_mem_psi_total(); - /* uninitialized last_psi_total; make no effect this round */ - if (goal->last_psi_total == U64_MAX) - goal->current_value = goal->target_value; - else - goal->current_value = now_psi_total - - goal->last_psi_total; - goal->last_psi_total = now_psi_total; + damos_set_psi_current_val(now_psi_total, goal, s); break; case DAMOS_QUOTA_NODE_MEM_USED_BP: case DAMOS_QUOTA_NODE_MEM_FREE_BP: From 45391a2e7fca92c389703add0dd71b67b1f87d14 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 17:27:22 -0700 Subject: [PATCH 0404/1012] mm/damon/core: copy nid for eligible_mem_bp damos quota goal commit damos_commit_quota_goal_union() is not updating the ->nid union field when the goal metric is DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP. Hence, if a DAMOS quota goal of the type is online committed in a way that it will reuse other quota goal's memory space, the new goal will work with a garbage nid value. As a result, the DAMOS scheme can show unexpected aggressiveness. Do the update. The user impact is not catastrophic. No leak or crash happens. Doing the quota goal online commit that can reproduce the issue is expected to be not common. This issue was not found by real users but the AI review. That said, the issue can reliably be reproduced. This issue was discovered [1] by Sashiko. Link: https://lore.kernel.org/20260902002725.108635-4-sj@kernel.org Link: https://lore.kernel.org/20260827045035.94611-1-sj@kernel.org [1] Fixes: 9138e27a3bc3 ("mm/damon: add node_eligible_mem_bp goal metric") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Lian Wang Tested-by: Lian Wang Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: Quanmin Yan Cc: # 7.2.x --- mm/damon/core.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 71e8cedbc64399..cf4ec122a7f99d 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1204,6 +1204,9 @@ static void damos_commit_quota_goal_union( dst->nid = src->nid; dst->memcg_id = src->memcg_id; break; + case DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP: + dst->nid = src->nid; + break; default: break; } From 296cfada6065f1cded520370d49aef953acd9d93 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 17:27:23 -0700 Subject: [PATCH 0405/1012] mm/damon/sysfs: set next refresh jiffies per sysfs context When 'refresh_ms' is set, DAMON sysfs interface periodically updates auto-tuned parameters and DAMOS stats. The timestamp for the next refresh is initialized when a DAMON context starts, and updated in its damon_call() callback function. That is, each DAMON context updates it. However, the timestamp is a global variable that is shared with all the contexts. When there are multiple DAMON contexts having different refresh_ms, the update frequency will be changed, depending on the order of the contexts. When there are multiple kdamonds, it will be even more chaotic. Fix the problem by having the timestamp per each context. The user impact is not very critical. It does not leak, corrupt or crash. The update will not be faster or slower than the lowest and largest refresh_ms values of the contexts, respectively. The user can also manually ask the updates on demand using kdamond state commands. That said, clearly this is a bug and can easily be reproduced. Link: https://lore.kernel.org/20260902002725.108635-5-sj@kernel.org Fixes: 9fd7bb5083d1 ("mm/damon/sysfs: change next_update_jiffies to a global variable") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Lian Wang Tested-by: Lian Wang Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: Quanmin Yan Cc: # 6.18.x --- mm/damon/sysfs.c | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 7f340b6f1921b9..7ec14f48d157a4 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -2034,6 +2034,7 @@ struct damon_sysfs_kdamond { struct damon_sysfs_contexts *contexts; struct damon_ctx *damon_ctx; unsigned int refresh_ms; + unsigned long next_refresh_jiffies; }; static struct damon_sysfs_kdamond *damon_sysfs_kdamond_alloc(void) @@ -2484,17 +2485,15 @@ static struct damon_ctx *damon_sysfs_build_ctx( return ctx; } -static unsigned long damon_sysfs_next_update_jiffies; - static int damon_sysfs_repeat_call_fn(void *data) { struct damon_sysfs_kdamond *sysfs_kdamond = data; if (!sysfs_kdamond->refresh_ms) return 0; - if (time_before(jiffies, damon_sysfs_next_update_jiffies)) + if (time_before(jiffies, sysfs_kdamond->next_refresh_jiffies)) return 0; - damon_sysfs_next_update_jiffies = jiffies + + sysfs_kdamond->next_refresh_jiffies = jiffies + msecs_to_jiffies(sysfs_kdamond->refresh_ms); if (!mutex_trylock(&damon_sysfs_lock)) @@ -2542,8 +2541,8 @@ static int damon_sysfs_turn_damon_on(struct damon_sysfs_kdamond *kdamond) } kdamond->damon_ctx = ctx; - damon_sysfs_next_update_jiffies = - jiffies + msecs_to_jiffies(kdamond->refresh_ms); + kdamond->next_refresh_jiffies = jiffies + + msecs_to_jiffies(kdamond->refresh_ms); repeat_call_control->fn = damon_sysfs_repeat_call_fn; repeat_call_control->data = kdamond; From 866c67522ac86a882753492f9787b82a8056a8ea Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:15 +0800 Subject: [PATCH 0406/1012] mm/mglru: separate folio generation update from LRU accounting MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "mm/mglru: speed up inc_min_seq() and fix cold/hot inversions", v3. This is an aging speedup series split out from the MGLRU swappiness series [1], with the inc_min_seq changes separated to make them easier to review. Currently, inc_min_seq() has two primary issues that affect both performance and folio hotness assessment: 1. It processes each folio one by one, while many operations can be batched or skipped. For example, a batch of folios can be moved together from the oldest generation to the second-oldest generation, and the associated counting can also be done in batches. 2. It may cause potential cold/hot inversion by placing promoted folios (which have been scanned and found to have young PTEs) behind non-promoted folios. A similar inversion can also occur among non-promoted folios, as tail folios from the oldest generation are placed before head folios when moving them to the second-oldest generation. This series tries to batch operations as much as possible and fix the potential cold/hot inversion by keeping promoted folios ahead of non-promoted folios, while also preserving the order of non-promoted folios when moving them from the oldest generation to the second-oldest generation. Minor issue: inc_min_seq() also counts protected folios improperly, as promoted folios should be skipped, as in sort_folio(). We need a stable workload with a stable number of folios to measure aging and evaluate the speedup in inc_min_seq(). So I asked ChatGPT to generate the microbenchmark below. It ages an LRU vec containing 512 MB of memory 100 times: #define _GNU_SOURCE #include #include #include #include #include #include #include #include #include #define SIZE (512UL * 1024 * 1024) #define LRU_GEN "/sys/kernel/debug/lru_gen" #define TARGET_CGROUP "/system.slice/agetest.scope" #define START_GEN 3 #define END_GEN 103 static long long nsec_diff(const struct timespec *start, const struct timespec *end) { return (end->tv_sec - start->tv_sec) * 1000000000LL + (end->tv_nsec - start->tv_nsec); } static int find_memcg_id(void) { FILE *fp; char line[4096]; int memcg_id; fp = fopen(LRU_GEN, "r"); if (!fp) { perror("fopen lru_gen"); return -1; } while (fgets(line, sizeof(line), fp)) { char *p; if (strncmp(line, "memcg ", 6)) continue; p = line + 6; if (sscanf(p, "%d", &memcg_id) != 1) continue; /* * The memcg path follows the numeric ID. */ p = strchr(p, ' '); if (!p) continue; if (strstr(p, TARGET_CGROUP)) { fclose(fp); return memcg_id; } } fclose(fp); fprintf(stderr, "Cannot find %s\n", TARGET_CGROUP); return -1; } int main(void) { void *addr; int memcg_id; int fd; long long total_ns = 0; /* * mmap 512 MB and touch every page. */ addr = mmap(NULL, SIZE, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0); if (addr == MAP_FAILED) { perror("mmap"); return 1; } memset(addr, 0x55, SIZE); printf("mmap: %p, size: %lu MB\n", addr, SIZE / 1024 / 1024); /* * Find the memcg ID automatically. */ memcg_id = find_memcg_id(); if (memcg_id < 0) return 1; printf("memcg: %d (%s)\n", memcg_id, TARGET_CGROUP); printf("aging generation %d -> %d\n", START_GEN, END_GEN); fd = open(LRU_GEN, O_WRONLY); if (fd < 0) { perror("open lru_gen"); return 1; } for (int gen = START_GEN; gen <= END_GEN; gen++) { char buf[128]; int len; struct timespec start, end; long long ns; len = snprintf(buf, sizeof(buf), "+ %d 0 %d\n", memcg_id, gen); clock_gettime(CLOCK_MONOTONIC, &start); if (write(fd, buf, len) != len) { perror("write lru_gen"); close(fd); return 1; } clock_gettime(CLOCK_MONOTONIC, &end); ns = nsec_diff(&start, &end); total_ns += ns; printf("gen %3d: %8.3f ms\n", gen, ns / 1000000.0); fflush(stdout); } close(fd); printf("\nTotal: %.3f ms\n", total_ns / 1000000.0); printf("Average: %.3f ms\n", total_ns / (double)(END_GEN - START_GEN + 1) / 1000000.0); while (1) sleep(1); return 0; } Run the above microbenchmark with: systemd-run --scope --unit=agetest -p MemoryMax=1024M ./agetest I’m seeing a significant speedup in inc_min_seq() on my x86 PC: W/o patch: Running scope as unit: agetest.scope mmap: 0x72c1b5a00000, size: 512 MB memcg: 12673 (/system.slice/agetest.scope) aging generation 3 -> 103 gen 3: 7.433 ms gen 4: 0.949 ms gen 5: 2.535 ms gen 6: 5.043 ms gen 7: 5.041 ms gen 8: 5.027 ms ... gen 100: 5.035 ms gen 101: 5.011 ms gen 102: 5.029 ms gen 103: 5.056 ms Total: 503.946 ms Average: 4.990 ms W/ patch: Running scope as unit: agetest.scope mmap: 0x775ec5200000, size: 512 MB memcg: 12717 (/system.slice/agetest.scope) aging generation 3 -> 103 gen 3: 7.558 ms gen 4: 0.916 ms gen 5: 2.277 ms gen 6: 2.328 ms … gen 100: 2.303 ms gen 101: 2.297 ms gen 102: 2.305 ms gen 103: 2.312 ms Total: 236.635 ms Average: 2.343 ms The average aging time drops from 4.990 ms to 2.343 ms! Thanks, Xueyuan, for testing this on ARM[2]. It actually shows an even larger improvement. Xueyuan tested this series on his arm64 machine (24 cores, 4K base pages) and reproduced the improvement: THP=never (PTE): baseline: 7.644 ms patched: 2.964 ms (-61.2%) THP=always (PMD): baseline: 0.0373 ms patched: 0.0292 ms (-21.8%) This patch (of 7): folio_inc_gen() currently updates both the folio's generation and the LRU size accounting. This makes it difficult to batch the LRU size updates when moving multiple folios. Extract the generation update into __folio_inc_gen(), which only updates the folio's generation and reports whether the generation was actually increased. Keep folio_inc_gen() as the wrapper that performs the LRU size accounting when needed. This separates the per-folio generation update from LRU accounting and allows the latter to be batched by subsequent changes. Link: https://lore.kernel.org/20260901232421.40157-1-baohua@kernel.org Link: https://lore.kernel.org/20260901232421.40157-2-baohua@kernel.org Link: https://lore.kernel.org/linux-mm/20260812121658.69965-1-baohua@kernel.org/ [1] Link: https://lore.kernel.org/linux-mm/20260827035416.3012015-1-xueyuan.chen21@gmail.com/ [2] Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Tested-by: Xueyuan Chen Reviewed-by: Lian Wang Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie Cc: Kunwu Chan Cc: Ridong Chen --- mm/vmscan.c | 28 +++++++++++++++++++++------- 1 file changed, 21 insertions(+), 7 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index ce3bab78af3cd1..d1a051e7db1eb6 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3300,21 +3300,21 @@ static int folio_update_gen(struct folio *folio, int gen, const vma_flags_t *vma return ((old_flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; } -/* protect pages accessed multiple times through file descriptors */ -static int folio_inc_gen(struct lruvec *lruvec, struct folio *folio) +static int __folio_inc_gen(struct folio *folio, int old_gen, bool *increased) { - int type = folio_is_file_lru(folio); - struct lru_gen_folio *lrugen = &lruvec->lrugen; - int new_gen, old_gen = lru_gen_from_seq(lrugen->min_seq[type]); unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f); + int new_gen; VM_WARN_ON_ONCE_FOLIO(!(old_flags & LRU_GEN_MASK), folio); do { new_gen = ((old_flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; /* folio_update_gen() has promoted this page? */ - if (new_gen >= 0 && new_gen != old_gen) + if (new_gen >= 0 && new_gen != old_gen) { + if (increased) + *increased = false; return new_gen; + } new_gen = (old_gen + 1) % MAX_NR_GENS; @@ -3322,8 +3322,22 @@ static int folio_inc_gen(struct lruvec *lruvec, struct folio *folio) new_flags |= (new_gen + 1UL) << LRU_GEN_PGOFF; } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags)); - lru_gen_update_size(lruvec, folio, old_gen, new_gen); + if (increased) + *increased = true; + return new_gen; +} + +/* protect pages accessed multiple times through file descriptors */ +static int folio_inc_gen(struct lruvec *lruvec, struct folio *folio) +{ + int type = folio_is_file_lru(folio); + struct lru_gen_folio *lrugen = &lruvec->lrugen; + int new_gen, old_gen = lru_gen_from_seq(lrugen->min_seq[type]); + bool gen_increased; + new_gen = __folio_inc_gen(folio, old_gen, &gen_increased); + if (gen_increased) + lru_gen_update_size(lruvec, folio, old_gen, new_gen); return new_gen; } From 87f50fea5a53650384a4064a49492f6d6c69dc38 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:16 +0800 Subject: [PATCH 0407/1012] mm/mglru: batch update lrugen->nr_pages in inc_min_seq() Currently, folio_inc_gen() updates lrugen->nr_pages for every folio as it advances generations. Instead, accumulate the size changes and update lrugen->nr_pages in a batch after scanning the entire oldest generation, or when the scan stops because remaining reaches zero. Since we only move folios from the oldest generation to the second oldest generation, the active/inactive state cannot change. We can therefore skip __lru_update_size(). Link: https://lore.kernel.org/20260901232421.40157-3-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Tested-by: Xueyuan Chen Reviewed-by: Lian Wang Reviewed-by: Kunwu Chan Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 23 ++++++++++++++++++----- 1 file changed, 18 insertions(+), 5 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index d1a051e7db1eb6..801345ca4da3c6 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3923,6 +3923,7 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) struct lru_gen_folio *lrugen = &lruvec->lrugen; int hist = lru_hist_from_seq(lrugen->min_seq[type]); int new_gen, old_gen = lru_gen_from_seq(lrugen->min_seq[type]); + int target_gen = (old_gen + 1) % MAX_NR_GENS; /* For file type, skip the check if swappiness is anon only */ if (type && (swappiness == SWAPPINESS_ANON_ONLY)) @@ -3932,35 +3933,47 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) if (!type && !swappiness) goto done; + VM_WARN_ON_ONCE(get_nr_gens(lruvec, type) != MAX_NR_GENS); + VM_WARN_ON_ONCE(lru_gen_is_active(lruvec, old_gen) != + lru_gen_is_active(lruvec, target_gen)); /* prevent cold/hot inversion if the type is evictable */ for (zone = 0; zone < MAX_NR_ZONES; zone++) { struct list_head *head = &lrugen->folios[old_gen][type][zone]; + long delta = 0; while (!list_empty(head)) { struct folio *folio = lru_to_folio(head); + long nr_pages = folio_nr_pages(folio); int refs = folio_lru_refs(folio); bool workingset = folio_test_workingset(folio); + bool gen_increased; VM_WARN_ON_ONCE_FOLIO(folio_test_unevictable(folio), folio); VM_WARN_ON_ONCE_FOLIO(folio_test_active(folio), folio); VM_WARN_ON_ONCE_FOLIO(folio_is_file_lru(folio) != type, folio); VM_WARN_ON_ONCE_FOLIO(folio_zonenum(folio) != zone, folio); - new_gen = folio_inc_gen(lruvec, folio); + new_gen = __folio_inc_gen(folio, old_gen, &gen_increased); list_move_tail(&folio->lru, &lrugen->folios[new_gen][type][zone]); - + if (gen_increased) + delta += nr_pages; /* don't count the workingset being lazily promoted */ if (refs + workingset != BIT(LRU_REFS_WIDTH) + 1) { int tier = lru_tier_from_refs(refs, workingset); - int delta = folio_nr_pages(folio); WRITE_ONCE(lrugen->protected[hist][type][tier], - lrugen->protected[hist][type][tier] + delta); + lrugen->protected[hist][type][tier] + nr_pages); } if (!--remaining) - return false; + break; } + WRITE_ONCE(lrugen->nr_pages[old_gen][type][zone], + lrugen->nr_pages[old_gen][type][zone] - delta); + WRITE_ONCE(lrugen->nr_pages[target_gen][type][zone], + lrugen->nr_pages[target_gen][type][zone] + delta); + if (!remaining) + return false; } done: reset_ctrl_pos(lruvec, type, true); From d056a8bf1240504f90709cd8df039f60f12662a8 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:17 +0800 Subject: [PATCH 0408/1012] mm/mglru: enhance cold/hot inversion handling in inc_min_seq() During aging, a folio's generation may already have been updated by folio_update_gen(), even though it has not yet been moved to the corresponding generation list. Such folios are hotter than those already in that generation. It makes sense for inc_min_seq() to increment the generation of folios that were never promoted during aging and move them to the tail of the new oldest generation. However, folios that were already promoted should instead be moved to the head of their updated generation, just as sort_folio() does in scan_folios(). Otherwise, promoted folios could end up behind folios that were never promoted, effectively inverting their hot/cold ordering. Link: https://lore.kernel.org/20260901232421.40157-4-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reviewed-by: Kairui Song Tested-by: Xueyuan Chen Reviewed-by: Ridong Chen Reviewed-by: Lian Wang Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kunwu Chan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 801345ca4da3c6..a608483ff9c7f9 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3954,9 +3954,17 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) VM_WARN_ON_ONCE_FOLIO(folio_zonenum(folio) != zone, folio); new_gen = __folio_inc_gen(folio, old_gen, &gen_increased); - list_move_tail(&folio->lru, &lrugen->folios[new_gen][type][zone]); - if (gen_increased) + /* + * If gen_increased is false, this is a promotion. Put folios + * at the head of the promoted gen. Otherwise, put them at + * the tail of the second-oldest gen. + */ + if (gen_increased) { delta += nr_pages; + list_move_tail(&folio->lru, &lrugen->folios[new_gen][type][zone]); + } else { + list_move(&folio->lru, &lrugen->folios[new_gen][type][zone]); + } /* don't count the workingset being lazily promoted */ if (refs + workingset != BIT(LRU_REFS_WIDTH) + 1) { int tier = lru_tier_from_refs(refs, workingset); From 2a011e97b995ca39b488982fde0c8bd18f939269 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:18 +0800 Subject: [PATCH 0409/1012] mm/mglru: exclude folios promoted by aging from protected in inc_min_seq() Some folios may have been promoted during aging, so don't count them as protected, similar to sort_folio(). Link: https://lore.kernel.org/20260901232421.40157-5-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Tested-by: Xueyuan Chen Reviewed-by: Ridong Chen Reviewed-by: Lian Wang Cc: Axel Rasmussen Cc: Baolin Wang Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Kunwu Chan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index a608483ff9c7f9..79311f62d3e615 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3962,17 +3962,17 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) if (gen_increased) { delta += nr_pages; list_move_tail(&folio->lru, &lrugen->folios[new_gen][type][zone]); + + /* don't count the workingset being lazily promoted */ + if (refs + workingset != BIT(LRU_REFS_WIDTH) + 1) { + int tier = lru_tier_from_refs(refs, workingset); + + WRITE_ONCE(lrugen->protected[hist][type][tier], + lrugen->protected[hist][type][tier] + nr_pages); + } } else { list_move(&folio->lru, &lrugen->folios[new_gen][type][zone]); } - /* don't count the workingset being lazily promoted */ - if (refs + workingset != BIT(LRU_REFS_WIDTH) + 1) { - int tier = lru_tier_from_refs(refs, workingset); - - WRITE_ONCE(lrugen->protected[hist][type][tier], - lrugen->protected[hist][type][tier] + nr_pages); - } - if (!--remaining) break; } From 1ada96d6f22c0b00c4fa6938a14ab198eb4043b0 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:19 +0800 Subject: [PATCH 0410/1012] mm/mglru: make LRU folio prefetch helper an inline function `prefetchw_prev_lru_folio()` is currently implemented as a macro with a potentially unused argument. This makes the helper harder to read and can also trigger checkpatch warnings about unused macro arguments. Make it a `static inline` function and remove the unnecessary `_field` argument. The helper always prefetches the previous folio's `flags`, so there is no need to make the field configurable. Link: https://lore.kernel.org/20260901232421.40157-6-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Tested-by: Xueyuan Chen Reviewed-by: Lian Wang Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Kunwu Chan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 26 +++++++++++++++----------- 1 file changed, 15 insertions(+), 11 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 79311f62d3e615..4b0ed3c9aab295 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -182,17 +182,21 @@ struct scan_control { }; #ifdef ARCH_HAS_PREFETCHW -#define prefetchw_prev_lru_folio(_folio, _base, _field) \ - do { \ - if ((_folio)->lru.prev != _base) { \ - struct folio *prev; \ - \ - prev = lru_to_folio(&(_folio->lru)); \ - prefetchw(&prev->_field); \ - } \ - } while (0) +static inline void prefetchw_prev_lru_folio(struct folio *folio, + struct list_head *base) +{ + if (folio->lru.prev != base) { + struct folio *prev; + + prev = lru_to_folio(&folio->lru); + prefetchw(&prev->flags); + } +} #else -#define prefetchw_prev_lru_folio(_folio, _base, _field) do { } while (0) +static inline void prefetchw_prev_lru_folio(struct folio *folio, + struct list_head *base) +{ +} #endif /* @@ -1700,7 +1704,7 @@ static unsigned long isolate_lru_folios(unsigned long nr_to_scan, struct folio *folio; folio = lru_to_folio(src); - prefetchw_prev_lru_folio(folio, src, flags); + prefetchw_prev_lru_folio(folio, src); nr_pages = folio_nr_pages(folio); total_scan += nr_pages; From 5e4c4c0fe279cfd23deea7175a7be41b34a28284 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:20 +0800 Subject: [PATCH 0411/1012] mm/mglru: move folios from oldest gen to second-oldest gen from head to tail For reclamation, it makes sense to reclaim folios from tail to head, as folios near the head are relatively hot. However, when moving folios from the oldest generation to the second-oldest generation, using the tail-to-head order would effectively cause a cold/hot inversion. The impact of the added prefetching might be arch-dependent. Some architectures could benefit more from prefetching, while others might see little to no impact. In my x86 test, it shows a very slight improvement: ***** no-prefetch: agetest: ... gen 100: 2.421 ms gen 101: 2.424 ms gen 102: 2.420 ms gen 103: 2.413 ms Total: 248.418 ms Average: 2.460 ms agetest: ... gen 100: 2.393 ms gen 101: 2.396 ms gen 102: 2.395 ms gen 103: 2.392 ms Total: 245.627 ms Average: 2.432 ms agetest: ... gen 100: 2.433 ms gen 101: 2.427 ms gen 102: 2.432 ms gen 103: 2.450 ms Total: 249.186 ms Average: 2.467 ms **** has-prefetch: agetest: .... gen 100: 2.314 ms gen 101: 2.310 ms gen 102: 2.321 ms gen 103: 2.303 ms Total: 237.619 ms Average: 2.353 ms agetest: gen 100: 2.342 ms gen 101: 2.343 ms gen 102: 2.339 ms gen 103: 2.335 ms Total: 239.929 ms Average: 2.376 ms agetest: gen 100: 2.344 ms gen 101: 2.347 ms gen 102: 2.352 ms gen 103: 2.348 ms Total: 241.188 ms Average: 2.388 ms Basically, its 2.3xx vs. 2.4xx, lower is better. Link: https://lore.kernel.org/20260901232421.40157-7-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Tested-by: Xueyuan Chen Reviewed-by: Lian Wang Cc: Axel Rasmussen Cc: Baolin Wang Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Kunwu Chan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 23 +++++++++++++++++++++-- 1 file changed, 21 insertions(+), 2 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 4b0ed3c9aab295..49fab93470ad87 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -192,11 +192,27 @@ static inline void prefetchw_prev_lru_folio(struct folio *folio, prefetchw(&prev->flags); } } + +static inline void prefetchw_next_lru_folio(struct folio *folio, + struct list_head *base) +{ + if (folio->lru.next != base) { + struct folio *next; + + next = list_entry(folio->lru.next, struct folio, lru); + prefetchw(&next->flags); + } +} #else static inline void prefetchw_prev_lru_folio(struct folio *folio, struct list_head *base) { } + +static inline void prefetchw_next_lru_folio(struct folio *folio, + struct list_head *base) +{ +} #endif /* @@ -3943,10 +3959,11 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) /* prevent cold/hot inversion if the type is evictable */ for (zone = 0; zone < MAX_NR_ZONES; zone++) { struct list_head *head = &lrugen->folios[old_gen][type][zone]; + struct list_head *pos = head->next; long delta = 0; - while (!list_empty(head)) { - struct folio *folio = lru_to_folio(head); + while (pos != head) { + struct folio *folio = list_entry(pos, struct folio, lru); long nr_pages = folio_nr_pages(folio); int refs = folio_lru_refs(folio); bool workingset = folio_test_workingset(folio); @@ -3957,6 +3974,8 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) VM_WARN_ON_ONCE_FOLIO(folio_is_file_lru(folio) != type, folio); VM_WARN_ON_ONCE_FOLIO(folio_zonenum(folio) != zone, folio); + prefetchw_next_lru_folio(folio, head); + pos = pos->next; new_gen = __folio_inc_gen(folio, old_gen, &gen_increased); /* * If gen_increased is false, this is a promotion. Put folios From b47bbc3c74238316a16b007a066b927d9bef72f0 Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Wed, 2 Sep 2026 07:24:21 +0800 Subject: [PATCH 0412/1012] mm/mglru: batch move folios to the second-oldest gen's LRU Detect folios that need to move from the oldest generation to the second-oldest generation, and batch-move them together. This can significantly reduce the sys time of inc_min_seq(), especially when the other type is significantly behind the preferred type. Link: https://lore.kernel.org/20260901232421.40157-8-baohua@kernel.org Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Tested-by: Xueyuan Chen Reviewed-by: Lian Wang Reviewed-by: Baolin Wang Assisted-by: gemini:gemini-3.6-flash Cc: Axel Rasmussen Cc: David Hildenbrand Cc: David Stevens Cc: Johannes Weiner Cc: Kairui Song Cc: Kunwu Chan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Ridong Chen Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 20 +++++++++++++++++++- 1 file changed, 19 insertions(+), 1 deletion(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 49fab93470ad87..ba7adf36e69f7b 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3936,6 +3936,19 @@ static void clear_mm_walk(void) kfree(walk); } +static inline void flush_lru_batch(struct list_head *head, struct list_head **batch_end, + struct list_head *dst) +{ + LIST_HEAD(movable); + + if (!*batch_end) + return; + + list_cut_position(&movable, head, *batch_end); + list_splice_tail_init(&movable, dst); + *batch_end = NULL; +} + static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) { int zone; @@ -3958,8 +3971,10 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) lru_gen_is_active(lruvec, target_gen)); /* prevent cold/hot inversion if the type is evictable */ for (zone = 0; zone < MAX_NR_ZONES; zone++) { + struct list_head *target_list = &lrugen->folios[target_gen][type][zone]; struct list_head *head = &lrugen->folios[old_gen][type][zone]; struct list_head *pos = head->next; + struct list_head *batch_end = NULL; long delta = 0; while (pos != head) { @@ -3984,7 +3999,7 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) */ if (gen_increased) { delta += nr_pages; - list_move_tail(&folio->lru, &lrugen->folios[new_gen][type][zone]); + batch_end = &folio->lru; /* don't count the workingset being lazily promoted */ if (refs + workingset != BIT(LRU_REFS_WIDTH) + 1) { @@ -3994,11 +4009,14 @@ static bool inc_min_seq(struct lruvec *lruvec, int type, int swappiness) lrugen->protected[hist][type][tier] + nr_pages); } } else { + flush_lru_batch(head, &batch_end, target_list); list_move(&folio->lru, &lrugen->folios[new_gen][type][zone]); } if (!--remaining) break; } + flush_lru_batch(head, &batch_end, target_list); + WRITE_ONCE(lrugen->nr_pages[old_gen][type][zone], lrugen->nr_pages[old_gen][type][zone] - delta); WRITE_ONCE(lrugen->nr_pages[target_gen][type][zone], From 2f49b679ea349f988fe692ad90c74345811d0ce9 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 07:03:07 -0700 Subject: [PATCH 0413/1012] mm/damon/tests/core-kunit: test damon_commit_filter() Patch series "mm/damon: add kunit and selftests for probes and probe weights". DAMON recently introduced probes and probe weights. Add kunit and selftests for ensuring the parameters for features can be set using the core API and the sysfs ABI, respectively. This patch (of 6): Add kunit test to ensure damon_commit_filter() updates destination filter as expected for valid inputs. Link: https://lore.kernel.org/20260902140313.85983-1-sj@kernel.org Link: https://lore.kernel.org/20260902140313.85983-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Shuah Khan --- mm/damon/tests/core-kunit.h | 40 +++++++++++++++++++++++++++++++++++++ 1 file changed, 40 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index af26b3d60957b5..3fc5b631b45c3a 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1316,6 +1316,45 @@ static void damon_test_commit_target_regions(struct kunit *test) (unsigned long[][2]) {{3, 8}, {8, 10}}, 2); } +static void damon_test_commit_filter_for(struct kunit *test, + struct damon_filter *dst, struct damon_filter *src) +{ + damon_commit_filter(dst, src); + KUNIT_EXPECT_EQ(test, dst->type, src->type); + KUNIT_EXPECT_EQ(test, dst->matching, src->matching); + KUNIT_EXPECT_EQ(test, dst->allow, src->allow); + switch (src->type) { + case DAMON_FILTER_TYPE_MEMCG: + KUNIT_EXPECT_EQ(test, dst->memcg_id, src->memcg_id); + break; + default: + break; + } +} + +static void damon_test_commit_filter(struct kunit *test) +{ + struct damon_filter dst = { + .type = DAMON_FILTER_TYPE_ANON, + .matching = false, + .allow = false, + }; + + damon_test_commit_filter_for(test, &dst, + &(struct damon_filter){ + .type = DAMON_FILTER_TYPE_ANON, + .matching = true, + .allow = true, + }); + damon_test_commit_filter_for(test, &dst, + &(struct damon_filter){ + .type = DAMON_FILTER_TYPE_MEMCG, + .matching = false, + .allow = false, + .memcg_id = 123, + }); +} + static void damon_test_commit_ctx(struct kunit *test) { struct damon_ctx *src, *dst; @@ -1722,6 +1761,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damos_test_commit_pageout), KUNIT_CASE(damos_test_commit_migrate_hot), KUNIT_CASE(damon_test_commit_target_regions), + KUNIT_CASE(damon_test_commit_filter), KUNIT_CASE(damon_test_commit_ctx), KUNIT_CASE(damon_test_valid_probe_params), KUNIT_CASE(damos_test_filter_out), From c520313f8762a05cb5caa62f716a4117fb6a363a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 07:03:08 -0700 Subject: [PATCH 0414/1012] mm/damon/tests/core-kunit: add damon_commit_probes() test Add kunit test to ensure damon_commit_probes() updates destination DAMON context with source probes as expected. Link: https://lore.kernel.org/20260902140313.85983-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Shuah Khan --- mm/damon/tests/core-kunit.h | 84 +++++++++++++++++++++++++++++++++++++ 1 file changed, 84 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 3fc5b631b45c3a..f2568fba552ec4 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1355,6 +1355,89 @@ static void damon_test_commit_filter(struct kunit *test) }); } +static struct damon_ctx *damon_test_help_setup_probes(unsigned int weights[], + int nr_weights) +{ + struct damon_ctx *ctx; + struct damon_probe *probe; + int i; + + ctx = damon_new_ctx(); + if (!ctx) + return NULL; + for (i = 0; i < nr_weights; i++) { + probe = damon_new_probe(); + if (!probe) { + damon_destroy_ctx(ctx); + return NULL; + } + probe->weight = weights[i]; + damon_add_probe(ctx, probe); + } + return ctx; +} + +static void damon_test_commit_probes_for(struct kunit *test, + unsigned int dst_weights[], int nr_dst_probes, + unsigned int src_weights[], int nr_src_probes) +{ + struct damon_ctx *dst, *src; + int err; + struct damon_probe *dst_probe, *src_probe; + + dst = damon_test_help_setup_probes(dst_weights, nr_dst_probes); + if (!dst) + kunit_skip(test, "dst alloc fail"); + src = damon_test_help_setup_probes(src_weights, nr_src_probes); + if (!src) { + damon_destroy_ctx(dst); + kunit_skip(test, "src alloc fail"); + } + + err = damon_commit_probes(dst, src); + KUNIT_EXPECT_EQ(test, err, 0); + if (err) + goto out; + nr_dst_probes = 0; + damon_for_each_probe(dst_probe, dst) + nr_dst_probes++; + nr_src_probes = 0; + damon_for_each_probe(src_probe, src) + nr_src_probes++; + KUNIT_EXPECT_EQ(test, nr_dst_probes, nr_src_probes); + if (nr_dst_probes != nr_src_probes) + goto out; + nr_dst_probes = 0; + damon_for_each_probe(dst_probe, dst) { + src_probe = damon_nth_probe(nr_dst_probes, src); + KUNIT_EXPECT_EQ(test, src_probe->weight, dst_probe->weight); + nr_dst_probes++; + } +out: + damon_destroy_ctx(dst); + damon_destroy_ctx(src); +} + +static void damon_test_commit_probes(struct kunit *test) +{ + damon_test_commit_probes_for(test, + (unsigned int[]){}, 0, (unsigned int[]){}, 0); + damon_test_commit_probes_for(test, + (unsigned int[]){}, 0, (unsigned int[]){1}, 1); + damon_test_commit_probes_for(test, + (unsigned int[]){}, 0, (unsigned int[]){1, 2}, 2); + damon_test_commit_probes_for(test, + (unsigned int[]){1}, 1, (unsigned int[]){2}, 1); + damon_test_commit_probes_for(test, + (unsigned int[]){1}, 1, (unsigned int[]){2, 3}, 2); + damon_test_commit_probes_for(test, + (unsigned int[]){2, 3}, 2, (unsigned int[]){1}, 1); + damon_test_commit_probes_for(test, + (unsigned int[]){2, 3}, 2, (unsigned int[]){}, 0); + damon_test_commit_probes_for(test, + (unsigned int[]){2}, 1, (unsigned int[]){}, 0); +} + static void damon_test_commit_ctx(struct kunit *test) { struct damon_ctx *src, *dst; @@ -1762,6 +1845,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damos_test_commit_migrate_hot), KUNIT_CASE(damon_test_commit_target_regions), KUNIT_CASE(damon_test_commit_filter), + KUNIT_CASE(damon_test_commit_probes), KUNIT_CASE(damon_test_commit_ctx), KUNIT_CASE(damon_test_valid_probe_params), KUNIT_CASE(damos_test_filter_out), From 94f99ce9df24bc0d3c6bbf54def797be1d9feff5 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 07:03:09 -0700 Subject: [PATCH 0415/1012] selftests/damon/_damon_sysfs: implement DamonProbes Extend _damon_sysfs.py to support staging and committing DAMON probes. It will be used for setting DAMON probes via sysfs changes for testing purposes. Link: https://lore.kernel.org/20260902140313.85983-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Shuah Khan --- tools/testing/selftests/damon/_damon_sysfs.py | 125 +++++++++++++++++- 1 file changed, 124 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/damon/_damon_sysfs.py b/tools/testing/selftests/damon/_damon_sysfs.py index f604b7d6530b3c..e7095c3365245a 100644 --- a/tools/testing/selftests/damon/_damon_sysfs.py +++ b/tools/testing/selftests/damon/_damon_sysfs.py @@ -570,6 +570,117 @@ def stage(self): return err return None +class DamonFilter: + type_ = None + matching = None + allow = None + filters = None + path = None + idx = None + + def __init__(self, type_='anon', matching=False, allow=False, path=None): + self.type_ = type_ + self.matching = matching + self.allow = allow + self.path = path + + def sysfs_dir(self): + return os.path.join(self.filters.sysfs_dir(), '%d' % self.idx) + + def stage(self): + err = write_file(os.path.join(self.sysfs_dir(), 'type'), self.type_) + if err is not None: + return err + err = write_file(os.path.join(self.sysfs_dir(), 'matching'), + 'Y' if self.matching else 'N') + if err is not None: + return err + err = write_file(os.path.join(self.sysfs_dir(), 'allow'), + 'Y' if self.allow else 'N') + if err is not None: + return err + if self.type_ == 'memcg': + err = write_file(os.path.join(self.sysfs_dir(), 'path'), self.path) + if err is not None: + return err + return None + +class DamonFilters: + filters = None + probe = None + + def __init__(self, filters=None): + if filters is None: + filters = [] + self.filters = filters + for idx, filter in enumerate(self.filters): + filter.filters = self + filter.idx = idx + + def sysfs_dir(self): + return os.path.join(self.probe.sysfs_dir(), 'filters') + + def stage(self): + err = write_file( + os.path.join(self.sysfs_dir(), 'nr_filters'), + len(self.filters)) + if err is not None: + return err + for filter in self.filters: + err = filter.stage() + if err is not None: + return err + return None + +class DamonProbe: + weight = None + filters = None + probes = None + idx = None + + def __init__(self, weight=0, filters=None): + self.weight = weight + if filters is None: + filters = DamonFilters() + self.filters = filters + self.filters.probe = self + + def sysfs_dir(self): + return os.path.join(self.probes.sysfs_dir(), '%d' % self.idx) + + def stage(self): + err = write_file( + os.path.join(self.sysfs_dir(), 'weight'), '%d' % self.weight) + if err is not None: + return err + return self.filters.stage() + +class DamonProbes: + probes = None + attrs = None + + def __init__(self, probes=None): + if probes is None: + probes = [] + self.probes = probes + for idx, probe in enumerate(self.probes): + probe.probes = self + probe.idx = idx + + def sysfs_dir(self): + return os.path.join(self.attrs.sysfs_dir(), 'probes') + + def stage(self): + err = write_file(os.path.join(self.sysfs_dir(), 'nr_probes'), + len(self.probes)) + if err is not None: + return err + for probe in self.probes: + err = probe.stage() + if err is not None: + return err + return None + class DamonAttrs: sample_us = None aggr_us = None @@ -577,11 +688,12 @@ class DamonAttrs: update_us = None min_nr_regions = None max_nr_regions = None + probes = None context = None def __init__(self, sample_us=5000, aggr_us=100000, intervals_goal=None, update_us=1000000, min_nr_regions=10, - max_nr_regions=1000): + max_nr_regions=1000, probes=None): self.sample_us = sample_us self.aggr_us = aggr_us if intervals_goal is None: @@ -591,6 +703,10 @@ def __init__(self, sample_us=5000, aggr_us=100000, self.update_us = update_us self.min_nr_regions = min_nr_regions self.max_nr_regions = max_nr_regions + if probes is None: + probes = DamonProbes() + self.probes = probes + self.probes.attrs = self def interval_sysfs_dir(self): return os.path.join(self.context.sysfs_dir(), 'monitoring_attrs', @@ -600,6 +716,9 @@ def nr_regions_range_sysfs_dir(self): return os.path.join(self.context.sysfs_dir(), 'monitoring_attrs', 'nr_regions') + def sysfs_dir(self): + return os.path.join(self.context.sysfs_dir(), 'monitoring_attrs') + def stage(self): err = write_file(os.path.join(self.interval_sysfs_dir(), 'sample_us'), self.sample_us) @@ -629,6 +748,10 @@ def stage(self): if err is not None: return err + err = self.probes.stage() + if err is not None: + return err + class DamonCtx: ops = None monitoring_attrs = None From 76f55ec0ccffa4ef644be1999374a3a524a2a6f1 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 07:03:10 -0700 Subject: [PATCH 0416/1012] selftests/damon/drgn_dump_damon_status: dump probes Extend drgn_dump_damon_status.py to dump damon_ctx->probes. It will be used to see if in-kernel DAMON status are changed as the user sets the probes via sysfs. Link: https://lore.kernel.org/20260902140313.85983-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Shuah Khan --- .../selftests/damon/drgn_dump_damon_status.py | 32 +++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/tools/testing/selftests/damon/drgn_dump_damon_status.py b/tools/testing/selftests/damon/drgn_dump_damon_status.py index 09552e91bc7820..4622046fd01180 100755 --- a/tools/testing/selftests/damon/drgn_dump_damon_status.py +++ b/tools/testing/selftests/damon/drgn_dump_damon_status.py @@ -48,6 +48,37 @@ def attrs_to_dict(attrs): ['max_nr_regions', int], ]) +def filter_to_dict(damon_filter): + filter_type_keyword = { + 0: 'anon', + 1: 'memcg', + } + dict_ = { + 'type': filter_type_keyword[int(damon_filter.type)], + 'matching': bool(damon_filter.matching), + 'allow': bool(damon_filter.allow), + } + type_ = dict_['type'] + if type_ == 'memcg': + dict_['memcg_id'] = int(damon_filter.memcg_id) + return dict_ + +def filters_to_list(filters): + return [filter_to_dict(f) + for f in list_for_each_entry( + 'struct damon_filter', filters.address_of_(), 'list')] + +def probe_to_dict(probe): + return to_dict(probe, [ + ['weight', int], + ['filters', filters_to_list], + ]) + +def probes_to_list(probes): + return [probe_to_dict(p) + for p in list_for_each_entry( + 'struct damon_probe', probes.address_of_(), 'list')] + def addr_range_to_dict(addr_range): return to_dict(addr_range, [ ['start', int], @@ -199,6 +230,7 @@ def damon_ctx_to_dict(ctx): return to_dict(ctx, [ ['ops', ops_to_dict], ['attrs', attrs_to_dict], + ['probes', probes_to_list], ['adaptive_targets', targets_to_list], ['schemes', schemes_to_list], ['pause', bool], From df59d2a4cad510e61fa0c138bf5b925612fcc0f4 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 07:03:11 -0700 Subject: [PATCH 0417/1012] selftests/damon/sysfs.py: extend commit assertion function for probes Extend DAMON sysfs testing commit assertion helper function to check probes too. Link: https://lore.kernel.org/20260902140313.85983-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Shuah Khan --- tools/testing/selftests/damon/sysfs.py | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/tools/testing/selftests/damon/sysfs.py b/tools/testing/selftests/damon/sysfs.py index 88a26422ff44cf..159cbeac067f89 100755 --- a/tools/testing/selftests/damon/sysfs.py +++ b/tools/testing/selftests/damon/sysfs.py @@ -178,6 +178,24 @@ def assert_monitoring_attrs_committed(attrs, dump): assert_true(dump['max_nr_regions'] == attrs.max_nr_regions, 'max_nr_regions', dump) +def assert_damon_filters_committed(filters, dump): + assert_true(len(dump) == len(filters.filters), 'probe filters', dump) + for idx, damon_filter in enumerate(filters.filters): + filter_dump = dump[idx] + assert_true(filter_dump['type'] == damon_filter.type_, 'type', + filter_dump) + assert_true(filter_dump['matching'] == damon_filter.matching, + 'matching', filter_dump) + assert_true(filter_dump['allow'] == damon_filter.allow, 'allow', + filter_dump) + +def assert_probes_committed(probes, dump): + assert_true(len(dump) == len(probes.probes), 'probes length', dump) + for idx, probe in enumerate(probes.probes): + probe_dump = dump[idx] + assert_true(probe.weight == probe_dump['weight'], 'weight', probe_dump) + assert_damon_filters_committed(probe.filters, probe_dump['filters']) + def assert_monitoring_target_committed(target, dump): # target.pid is the pid "number", while dump['pid'] is 'struct pid' # pointer, and hence cannot be compared. @@ -196,6 +214,7 @@ def assert_ctx_committed(ctx, dump): } assert_true(dump['ops']['id'] == ops_val[ctx.ops], 'ops_id', dump) assert_monitoring_attrs_committed(ctx.monitoring_attrs, dump['attrs']) + assert_probes_committed(ctx.monitoring_attrs.probes, dump['probes']) assert_monitoring_targets_committed(ctx.targets, dump['adaptive_targets']) assert_schemes_committed(ctx.schemes, dump['schemes']) assert_true(dump['pause'] == ctx.pause, 'pause', dump) From 23caccd9988c509ecfe72a9db5008c879f55e447 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 07:03:12 -0700 Subject: [PATCH 0418/1012] selftests/damon/sysfs.py: test damon probes Extend sysfs.py to commit DAMON probes via sysfs, and see if it changed in-kernel DAMON status as expected using drgn. Link: https://lore.kernel.org/20260902140313.85983-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Shuah Khan --- tools/testing/selftests/damon/sysfs.py | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/damon/sysfs.py b/tools/testing/selftests/damon/sysfs.py index 159cbeac067f89..66c826189320d7 100755 --- a/tools/testing/selftests/damon/sysfs.py +++ b/tools/testing/selftests/damon/sysfs.py @@ -319,7 +319,21 @@ def main(): intervals_goal=_damon_sysfs.IntervalsGoal( access_bp=400, aggrs=3, min_sample_us=5000, max_sample_us=10000000), - update_us=2000000), + update_us=2000000, + probes=_damon_sysfs.DamonProbes( + probes=[_damon_sysfs.DamonProbe( + weight=42, + filters=_damon_sysfs.DamonFilters( + filters=[ + _damon_sysfs.DamonFilter( + type_='anon', + matching=True, + allow=True, + ), + ]), + ), + ]), + ), schemes=[_damon_sysfs.Damos( action='pageout', access_pattern=_damon_sysfs.DamosAccessPattern( From 3cfffab519de23c340058b0c7ebd1f0546cd3a93 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sat, 26 Sep 2026 11:41:06 +0100 Subject: [PATCH 0419/1012] mm: move drivers/char/mem.c to mm/char-mem.c The memory character driver implements several mm-specific features and is always compiled into the kernel, so move it to mm/ where it belongs. Among other things the driver implements /dev/mem which provides raw access to physical memory, and /dev/zero which either allows mapping of a shmem region (if mapped with MAP_SHARED) or, uniquely, anonymous memory (if mapped MAP_PRIVATE). This change lays the foundations to allow MAP_PRIVATE-/dev/zero to be mapped precisely the same as anonymous memory is mapped as currently it is an edge case within mm. Also update a couple of comments that reference 'drivers/char/mem.c' to reference 'mm/char-mem.c'. Link: https://lore.kernel.org/20260926-map-private-dev-zero-v3-1-d4781e84ccfc@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Jan Kara --- MAINTAINERS | 4 ++-- drivers/char/Makefile | 2 +- mm/Makefile | 3 ++- drivers/char/mem.c => mm/char-mem.c | 2 +- mm/shmem.c | 2 +- 5 files changed, 7 insertions(+), 6 deletions(-) rename drivers/char/mem.c => mm/char-mem.c (99%) diff --git a/MAINTAINERS b/MAINTAINERS index 360977678f707e..3ff1a8f07e526f 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17268,7 +17268,7 @@ F: Documentation/ABI/testing/sysfs-kernel-mm-cma F: Documentation/ABI/testing/sysfs-kernel-mm-numa F: Documentation/admin-guide/mm/ F: Documentation/mm/ -F: drivers/char/mem.c +F: mm/char-mem.c F: include/linux/cma.h F: include/linux/dmapool.h F: include/linux/ioremap.h @@ -17492,7 +17492,7 @@ L: linux-mm@kvack.org S: Maintained W: http://www.linux-mm.org T: git git://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm -F: drivers/char/mem.c +F: mm/char-mem.c F: include/trace/events/mmap.h F: fs/proc/task_mmu.c F: fs/proc/task_nommu.c diff --git a/drivers/char/Makefile b/drivers/char/Makefile index a46d7bf7c4c828..bb3bf66937b37b 100644 --- a/drivers/char/Makefile +++ b/drivers/char/Makefile @@ -3,7 +3,7 @@ # Makefile for the kernel character device drivers. # -obj-y += mem.o random.o +obj-y += random.o obj-$(CONFIG_TTY_PRINTK) += ttyprintk.o obj-y += misc.o obj-$(CONFIG_TEST_MISC_MINOR) += misc_minor_kunit.o diff --git a/mm/Makefile b/mm/Makefile index e7245cb88c6651..2a3ec53d62eef6 100644 --- a/mm/Makefile +++ b/mm/Makefile @@ -55,7 +55,8 @@ obj-y := filemap.o mempool.o oom_kill.o fadvise.o \ mm_init.o percpu.o slab_common.o \ compaction.o show_mem.o \ interval_tree.o list_lru.o workingset.o \ - debug.o gup.o mmap_lock.o vma_init.o $(mmu-y) + debug.o gup.o mmap_lock.o vma_init.o char-mem.o \ + $(mmu-y) # Give 'page_alloc' its own module-parameter namespace page-alloc-y := page_alloc.o diff --git a/drivers/char/mem.c b/mm/char-mem.c similarity index 99% rename from drivers/char/mem.c rename to mm/char-mem.c index 5b93c92c2cf194..3cc48054a66e71 100644 --- a/drivers/char/mem.c +++ b/mm/char-mem.c @@ -1,6 +1,6 @@ // SPDX-License-Identifier: GPL-2.0 /* - * linux/drivers/char/mem.c + * mm/char-mem.c * * Copyright (C) 1991, 1992 Linus Torvalds * diff --git a/mm/shmem.c b/mm/shmem.c index 84f0a2eb85fecd..05bc7c3aa52548 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -2984,7 +2984,7 @@ unsigned long shmem_get_unmapped_area(struct file *file, sb = file_inode(file)->i_sb; } else { /* - * Called directly from mm/mmap.c, or drivers/char/mem.c + * Called directly from mm/mmap.c, or mm/char-mem.c * for "/dev/zero", to create a shared anonymous object. */ if (IS_ERR(shm_mnt)) From 2ea88d511d8248c2f053dd095edc44bfdc42776f Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sat, 26 Sep 2026 11:41:07 +0100 Subject: [PATCH 0420/1012] mm: implement file_is_dev_zero() to uniquely identify /dev/zero To lay the foundation for a future change that converts MAP_PRIVATE-/dev/zero mappings to be truly anonymous, add the ability to uniquely identify these mappings. With the memory character device now part of mm/ this is trivially achievable through a file_is_dev_zero() predicate that simply tests that the file operation hooks are zero_fops. Also update userland VMA tests to expose file_is_dev_zero() and provide a stub zero_fops for testing. Link: https://lore.kernel.org/20260926-map-private-dev-zero-v3-2-d4781e84ccfc@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Jan Kara --- mm/char-mem.c | 13 +++++++++++++ mm/internal.h | 3 +++ tools/testing/vma/include/dup.h | 7 +++++++ tools/testing/vma/shared.c | 9 +++++++++ 4 files changed, 32 insertions(+) diff --git a/mm/char-mem.c b/mm/char-mem.c index 3cc48054a66e71..87fb71a011919f 100644 --- a/mm/char-mem.c +++ b/mm/char-mem.c @@ -31,6 +31,8 @@ #include #include +#include "internal.h" + #define DEVMEM_MINOR 1 #define DEVPORT_MINOR 4 @@ -707,6 +709,17 @@ static const struct memdev { #endif }; +/** + * file_is_dev_zero() - is the specified @file associated with the /dev/zero + * driver? + * @file: File to test. + * Returns: true if it is, false otherwise. + */ +bool file_is_dev_zero(const struct file *file) +{ + return file && file->f_op == &zero_fops; +} + static int memory_open(struct inode *inode, struct file *filp) { int minor; diff --git a/mm/internal.h b/mm/internal.h index e16f1250b25c80..5d474e5f77092a 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -1638,4 +1638,7 @@ static inline bool can_spin_trylock(void) return true; } +/* char-mem.c */ +bool file_is_dev_zero(const struct file *file); + #endif /* __MM_INTERNAL_H */ diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 57046d8ac81d80..0d1a2ac8892261 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1641,3 +1641,10 @@ static inline pgoff_t linear_anon_page_index(const struct vm_area_struct *vma, return pgoff; } + +extern const struct file_operations zero_fops; + +static inline bool file_is_dev_zero(const struct file *file) +{ + return file && file->f_op == &zero_fops; +} diff --git a/tools/testing/vma/shared.c b/tools/testing/vma/shared.c index 4a39c9d5048964..8c4826499f4054 100644 --- a/tools/testing/vma/shared.c +++ b/tools/testing/vma/shared.c @@ -12,6 +12,15 @@ const struct vm_operations_struct vma_dummy_vm_ops; struct anon_vma dummy_anon_vma; struct task_struct __current; +static int mmap_zero_prepare(struct vm_area_desc *desc) +{ + return 0; +} + +const struct file_operations zero_fops = { + .mmap_prepare = mmap_zero_prepare, +}; + struct vm_area_struct *alloc_vma(struct mm_struct *mm, unsigned long start, unsigned long end, pgoff_t pgoff, vma_flags_t vma_flags) From ebc5bba855c1f2351257eb731b87191489b2c37c Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sat, 26 Sep 2026 11:41:08 +0100 Subject: [PATCH 0421/1012] mm/vma: only permit MAP_PRIVATE /dev/zero to be mapped anonymous In order to use mmap_prepare() with MAP_PRIVATE mappings of /dev/zero without the success_hook hack we explicitly permitted mmap_prepare handlers to set NULL vm_ops. However this is dangerous and we really only want to allow this for MAP_PRIVATE-mapped /dev/zero. Therefore use the newly introduced file_is_dev_zero() to uniquely identify MAP_PRIVATE-/dev/zero mappings and only permit this behaviour for them. Then, remove all ability for mmap_prepare or mmap hooks to set a VMA anonymous and update mmap_zero_prepare() to leave it to the core mmap code to do so. Note that this disallows nested MAP_PRIVATE-mappings of /dev/zero regions. Doing this would be broken in any case. We therefore do not need to update the mmap_prepare() compatibility layer to reflect these changes, as the mmap hook check suffices to disallow this behaviour. Now we're setting vma->vm_ops to NULL for an mmap_prepare-initialised MAP_PRIVATE-/dev/zero mapping, we have to avoid a subtle issue when updating user-defined fields via set_vma_user_defined_fields(). The default for vma->vm_ops for all mmap_prepare-initialised mappings is vma_dummy_vm_ops, so map->vm_ops will be set to this and setting vma->vm_ops to this will render the VMA mistakenly non-anon. In general, we should never be setting user-defined fields for an anonymous VMA, so explicitly check for this to avoid doing so for the one case where a mapping can be both mmap_prepare and anonymous. In the case of legacy ->mmap hooks some drivers may set vma->vm_ops NULL believing this is the equivalent of setting no VMA operations. Therefore update mmap_file() to correct this by setting dummy VMA operations if this occurs. An example of this is drm_gem_shmem_mmap() which deliberately clears vma->vm_ops before handing the VMA to dma-buf. Cases such as this will be updated when they are converted to mmap_prepare. Also, in order to avoid a single commit bisection hazard, add a temporary workaround to set the VMA anonymous only after vma->vm_file is assigned in __mmap_new_file_vma(). This is because vma_set_range() calls vma_set_pgoff() and assert_sane_pgoff() in turn, prior to the vma->vm_file being assigned. If we set the VMA anonymous early then this assert will fail. This is removed in the subsequent commit. Link: https://lore.kernel.org/20260926-map-private-dev-zero-v3-3-d4781e84ccfc@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Jan Kara --- mm/char-mem.c | 6 +----- mm/internal.h | 17 ++++++++++------- mm/vma.c | 33 +++++++++++++++++++++++++-------- 3 files changed, 36 insertions(+), 20 deletions(-) diff --git a/mm/char-mem.c b/mm/char-mem.c index 87fb71a011919f..e9f94d2b5133ef 100644 --- a/mm/char-mem.c +++ b/mm/char-mem.c @@ -508,11 +508,7 @@ static int mmap_zero_prepare(struct vm_area_desc *desc) if (vma_desc_test(desc, VMA_MAYSHARE_BIT)) return shmem_zero_setup_desc(desc); - /* - * This is a highly unique situation where we mark a MAP_PRIVATE mapping - * of /dev/zero anonymous, despite it not being. - */ - vma_desc_set_anonymous(desc); + /* MAP_PRIVATE semantics are taken care of for us by core mm. */ return 0; } diff --git a/mm/internal.h b/mm/internal.h index 5d474e5f77092a..da14c56fb24e11 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -226,15 +226,18 @@ static inline int mmap_file(struct file *file, struct vm_area_struct *vma) { int err = vfs_mmap(file, vma); - if (likely(!err)) - return 0; - /* - * OK, we tried to call the file hook for mmap(), but an error - * arose. The mapping is in an inconsistent state and we must not invoke - * any further hooks on it. + * Either we tried to call the file hook for mmap() and an error arose + * or a driver set vma->vm_ops = NULL intending there to be no VMA + * operations. + * + * In the former case the VMA is in an inconsistent state and we mustn't + * invoke any further hooks on it, in the latter case the hook actually + * wanted no further hooks to be invoked, so fix both by setting dummy + * VMA ops. */ - vma->vm_ops = &vma_dummy_vm_ops; + if (unlikely(err || !vma->vm_ops)) + vma->vm_ops = &vma_dummy_vm_ops; return err; } diff --git a/mm/vma.c b/mm/vma.c index f3f230cbd4595d..500728613f640d 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2644,6 +2644,19 @@ static int __mmap_new_file_vma(struct mmap_state *map, return 0; } +static bool map_is_private(const struct mmap_state *map) +{ + return !vma_flags_test(&map->vma_flags, VMA_SHARED_BIT); +} + +static bool map_is_anon(const struct mmap_state *map) +{ + if (!map_is_private(map)) + return false; + + return !map->file || file_is_dev_zero(map->file); +} + /* * __mmap_new_vma() - Allocate a new VMA for the region, as merging was not * possible. @@ -2657,8 +2670,7 @@ static int __mmap_new_file_vma(struct mmap_state *map, static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, struct mmap_action *action) { - const bool is_anon = !map->file && - !vma_flags_test(&map->vma_flags, VMA_SHARED_BIT); + const bool is_anon = map_is_anon(map); struct vma_iterator *vmi = map->vmi; int error = 0; struct vm_area_struct *vma; @@ -2674,7 +2686,7 @@ static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, vma_iter_config(vmi, map->addr, map->end); - if (is_anon) + if (is_anon && !map->file) vma_set_anonymous(vma); vma_set_range(vma, map->addr, map->end, map->pgoff, map->anon_pgoff); @@ -2692,6 +2704,10 @@ static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, else if (!is_anon) error = shmem_zero_setup(vma); + /* Temporary MAP_PRIVATE-/dev/zero workaround. */ + if (is_anon && map->file) + vma_set_anonymous(vma); + if (error) goto free_iter_vma; @@ -2800,6 +2816,10 @@ static int call_mmap_prepare(struct mmap_state *map, if (err) return err; + /* It's invalid for mmap_prepare hooks to clear vm_ops. */ + if (!desc->vm_ops) + return -EINVAL; + err = call_action_prepare(map, desc); if (err) return err; @@ -2822,10 +2842,7 @@ static int call_mmap_prepare(struct mmap_state *map, static void set_vma_user_defined_fields(struct vm_area_struct *vma, struct mmap_state *map) { - if (map->vm_ops) - vma->vm_ops = map->vm_ops; - else /* Only /dev/zero should do this. */ - vma_set_anonymous(vma); + vma->vm_ops = map->vm_ops; vma->vm_private_data = map->vm_private_data; } @@ -2907,7 +2924,7 @@ static unsigned long __mmap_region(struct file *file, unsigned long addr, allocated_new = true; } - if (have_mmap_prepare && allocated_new) + if (have_mmap_prepare && !map_is_anon(&map) && allocated_new) set_vma_user_defined_fields(vma, &map); __mmap_complete(&map, vma); From b8ba1bdf99072138cc3afd915830401432b6cc06 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sat, 26 Sep 2026 11:41:09 +0100 Subject: [PATCH 0422/1012] mm/vma: make MAP_PRIVATE-mapped /dev/zero mappings truly anonymous When mapping /dev/zero with MAP_PRIVATE, one ends up with strange VMAs originating from Linux's distant past. These have vma->vm_file set but NULL vma->vm_ops, meaning they satisfy vma_is_anonymous() but otherwise resemble a file-backed VMA. The introduction of anonymous page offsets and their subsequent use as indexes for MAP_PRIVATE-file-backed mappings mean the rmap does the right thing with these but we are left with inconsistencies. The vma_start_pgoff(vma) == vma_start_anon_pgoff(vma) invariant is true for all other anonymous VMAs, but not these. These VMAs are also observable as files in /proc//[maps, smaps, map_files] but otherwise behave like anonymous mappings. Therefore let's make these VMAs actually anonymous at mapping time which will activate the anonymous code path for mappings. This means we no longer have to account for this discrepancy anywhere and no longer have to think about these at all. This is user-observable, as MAP_PRIVATE-/dev/zero will no longer appear in procfs as a file-backed mapping, but the impact of this change should be low as likely nobody is relying upon this. However in any case, in using MAP_PRIVATE-/dev/zero they are explicitly asking anonymous memory, so no longer seeing these as file mappings is in fact correct. A previous commit gave us file_is_dev_zero() to positively identify these mappings, so we expressly only do so for these alone. Update assert_sane_pgoff(), the comment for vma_start_pgoff() and linear_anon_page_index() to reflect the change. We make this change in call_mmap_prepare() alone as /dev/zero has been converted to an mmap_prepare hook and we do not permit nested MAP_PRIVATE mapping of /dev/zero. We also remove the now defunct vma_desc_set_anonymous() and eliminate the temporary bisection hazard fix from the previous commit. Also update the VMA userland tests to reflect the change. Finally, update the procfs self tests proc-self-map-files-001 and proc-self-map-files-002 which both intend to map an arbitrary file MAP_PRIVATE then assert procfs state, but happen to choose /dev/zero. Fix them by updating these to /proc/self/exe which is guaranteed to be present if procfs is mounted. Link: https://lore.kernel.org/20260926-map-private-dev-zero-v3-4-d4781e84ccfc@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Jan Kara --- include/linux/mm.h | 10 ++----- include/linux/pagemap.h | 3 +-- mm/vma.c | 26 ++++++++++++------- mm/vma.h | 3 --- .../selftests/proc/proc-self-map-files-001.c | 2 +- .../selftests/proc/proc-self-map-files-002.c | 2 +- tools/testing/vma/include/dup.h | 3 +-- 7 files changed, 23 insertions(+), 26 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index c105a3758915b2..c49ef99b4413b4 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1529,11 +1529,6 @@ static inline void vma_set_anonymous(struct vm_area_struct *vma) vma->vm_ops = NULL; } -static inline void vma_desc_set_anonymous(struct vm_area_desc *desc) -{ - desc->vm_ops = NULL; -} - static inline bool vma_is_anonymous(const struct vm_area_struct *vma) { return !vma->vm_ops; @@ -4389,9 +4384,8 @@ static inline unsigned long vma_pages(const struct vm_area_struct *vma) * If @vma is a MAP_PRIVATE file-backed mapping, then this returns the * page offset within the file. * - * Edge cases: nommu does not abide by these, MAP_PRIVATE-/dev/zero satisfies - * vma_is_anonymous() but has file-backed page offset, and MAP_PRIVATE-pfnmap - * regions have their page offset set to the first PFN in the range. + * Edge cases: nommu does not abide by these and CoW MAP_PRIVATE-pfnmap regions + * have their page offset set to the first PFN in the range. * * Returns: The page offset of the start of @vma. */ diff --git a/include/linux/pagemap.h b/include/linux/pagemap.h index 0adfa6605653db..939f3a5e973f6b 100644 --- a/include/linux/pagemap.h +++ b/include/linux/pagemap.h @@ -1128,8 +1128,7 @@ static inline pgoff_t linear_anon_page_index(const struct vm_area_struct *vma, const pgoff_t pgoff = __linear_anon_page_index(vma, address); VM_WARN_ON_ONCE(!vma_is_cow_mapping(vma)); - /* Account for MAP_PRIVATE-/dev/zero which is only semi-anonymous. */ - if (vma_is_anonymous(vma) && !vma->vm_file) + if (vma_is_anonymous(vma)) VM_WARN_ON_ONCE(pgoff != linear_page_index(vma, address)); return pgoff; diff --git a/mm/vma.c b/mm/vma.c index 500728613f640d..c24ee55b7ffff5 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2644,6 +2644,13 @@ static int __mmap_new_file_vma(struct mmap_state *map, return 0; } +static void map_set_anon(struct mmap_state *map) +{ + map->file = NULL; + map->vm_ops = NULL; + map->pgoff = map->addr >> PAGE_SHIFT; +} + static bool map_is_private(const struct mmap_state *map) { return !vma_flags_test(&map->vma_flags, VMA_SHARED_BIT); @@ -2651,10 +2658,7 @@ static bool map_is_private(const struct mmap_state *map) static bool map_is_anon(const struct mmap_state *map) { - if (!map_is_private(map)) - return false; - - return !map->file || file_is_dev_zero(map->file); + return map_is_private(map) && !map->file; } /* @@ -2686,7 +2690,7 @@ static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, vma_iter_config(vmi, map->addr, map->end); - if (is_anon && !map->file) + if (is_anon) vma_set_anonymous(vma); vma_set_range(vma, map->addr, map->end, map->pgoff, map->anon_pgoff); @@ -2704,10 +2708,6 @@ static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, else if (!is_anon) error = shmem_zero_setup(vma); - /* Temporary MAP_PRIVATE-/dev/zero workaround. */ - if (is_anon && map->file) - vma_set_anonymous(vma); - if (error) goto free_iter_vma; @@ -2836,6 +2836,14 @@ static int call_mmap_prepare(struct mmap_state *map, map->vm_ops = desc->vm_ops; map->vm_private_data = desc->private_data; + /* + * MAP_PRIVATE-/dev/zero mappings are an ancient way of getting + * anonymous mappings. Rather than allowing these mappings to be odd + * outliers, simply make them truly anonymous. + */ + if (map_is_private(map) && file_is_dev_zero(map->file)) + map_set_anon(map); + return 0; } diff --git a/mm/vma.h b/mm/vma.h index 36973abaa015a8..f856d9ace3a6a6 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -267,9 +267,6 @@ static inline void assert_sane_pgoff(struct vm_area_struct *vma, pgoff_t pgoff) */ if (!vma_is_anonymous(vma)) return; - /* MAP_PRIVATE-/dev/zero is anon, non-NULL vm_file, but has file pgoff. */ - if (vma->vm_file) - return; /* If faulted in, could have been remapped. */ if (vma->anon_vma) return; diff --git a/tools/testing/selftests/proc/proc-self-map-files-001.c b/tools/testing/selftests/proc/proc-self-map-files-001.c index 4209c64283d6f6..bbca9f9e27439e 100644 --- a/tools/testing/selftests/proc/proc-self-map-files-001.c +++ b/tools/testing/selftests/proc/proc-self-map-files-001.c @@ -51,7 +51,7 @@ int main(void) int fd; unsigned long a, b; - fd = open("/dev/zero", O_RDONLY); + fd = open("/proc/self/exe", O_RDONLY); if (fd == -1) return 1; diff --git a/tools/testing/selftests/proc/proc-self-map-files-002.c b/tools/testing/selftests/proc/proc-self-map-files-002.c index e6aa00a183bcd9..5786cdffbbf636 100644 --- a/tools/testing/selftests/proc/proc-self-map-files-002.c +++ b/tools/testing/selftests/proc/proc-self-map-files-002.c @@ -57,7 +57,7 @@ int main(void) int fd; unsigned long a, b; - fd = open("/dev/zero", O_RDONLY); + fd = open("/proc/self/exe", O_RDONLY); if (fd == -1) return 1; diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 0d1a2ac8892261..16c09dac59d9b4 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1635,8 +1635,7 @@ static inline pgoff_t linear_anon_page_index(const struct vm_area_struct *vma, const pgoff_t pgoff = __linear_anon_page_index(vma, address); VM_WARN_ON_ONCE(!vma_is_cow_mapping(vma)); - /* Account for MAP_PRIVATE-/dev/zero which is only semi-anonymous. */ - if (vma_is_anonymous(vma) && !vma->vm_file) + if (vma_is_anonymous(vma)) VM_WARN_ON_ONCE(pgoff != linear_page_index(vma, address)); return pgoff; From 5e6d2a1bb3eb64f524e3d2dfd997de670a2c76eb Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sat, 26 Sep 2026 11:41:10 +0100 Subject: [PATCH 0423/1012] tools/testing/vma: add test to assert MAP_PRIVATE-/dev/zero is anon Now we've made MAP_PRIVATE-mapped /dev/zero mappings truly anonymous, add a VMA userland test to assert that this is the case and everything is as we would expect for an anonymous mapping. Link: https://lore.kernel.org/20260926-map-private-dev-zero-v3-5-d4781e84ccfc@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Jan Kara --- tools/testing/vma/tests/mmap.c | 37 ++++++++++++++++++++++++++++++++++ 1 file changed, 37 insertions(+) diff --git a/tools/testing/vma/tests/mmap.c b/tools/testing/vma/tests/mmap.c index c85bc000d1cb7a..fa73faff226263 100644 --- a/tools/testing/vma/tests/mmap.c +++ b/tools/testing/vma/tests/mmap.c @@ -45,7 +45,44 @@ static bool test_mmap_region_basic(void) return true; } +static bool test_pure_anon_dev_zero(void) +{ + const vma_flags_t vma_flags = mk_vma_flags(VMA_READ_BIT, VMA_WRITE_BIT, + VMA_MAYREAD_BIT, VMA_MAYWRITE_BIT); + struct file file = { + .f_op = &zero_fops, + }; + struct mm_struct mm = {}; + struct vm_area_struct *vma; + unsigned long addr; + VMA_ITERATOR(vmi, &mm, 0); + + current->mm = &mm; + + /* + * Map a MAP_PRIVATE-/dev/zero mapping at address 0x300000 with a page + * offset of 0x10, which we expect to be reset to the anonymous page + * offset. + */ + addr = __mmap_region(&file, 0x300000, 0x3000, vma_flags, 0x10, NULL); + ASSERT_EQ(addr, 0x300000); + + /* Assert that it truly is an anonymous mapping. */ + vma = vma_lookup(&mm, addr); + ASSERT_NE(vma, NULL); + ASSERT_TRUE(vma_is_anonymous(vma)); + ASSERT_EQ(vma->vm_file, NULL); + ASSERT_EQ(vma->vm_private_data, NULL); + /* Expect anonymous page offsets. */ + ASSERT_EQ(vma->vm_pgoff, 0x300); + ASSERT_EQ(vma_start_anon_pgoff(vma), 0x300); + + cleanup_mm(&mm, &vmi); + return true; +} + static void run_mmap_tests(int *num_tests, int *num_fail) { TEST(mmap_region_basic); + TEST(pure_anon_dev_zero); } From b0668cbc1e51609e190e942ccb8ba3ea9be379c3 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Sat, 26 Sep 2026 11:41:11 +0100 Subject: [PATCH 0424/1012] tools/testing/selftests/mm: add MAP_PRIVATE-/dev/zero merge tests Assert that MAP_PRIVATE-mapped /dev/zero mappings behave like they are anonymous. Test both unfaulted and faulted/unfaulted merges with page offset 0 which would not merge if the mappings were treated as if they were file-backed. With the recent change that makes them behave as pure anonymous mappings, the merges should succeed as their page offsets are equal to their anonymous page offsets. Link: https://lore.kernel.org/20260926-map-private-dev-zero-v3-6-d4781e84ccfc@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Arnd Bergmann Cc: Greg Kroah-Hartman Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Pedro Falcato Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Jan Kara --- tools/testing/selftests/mm/merge.c | 95 ++++++++++++++++++++++++++++++ 1 file changed, 95 insertions(+) diff --git a/tools/testing/selftests/mm/merge.c b/tools/testing/selftests/mm/merge.c index 52b8727b6628e0..7dd2933417c1d6 100644 --- a/tools/testing/selftests/mm/merge.c +++ b/tools/testing/selftests/mm/merge.c @@ -1362,6 +1362,101 @@ TEST_F(merge, anon_and_page_offset_mismatch_memfd) ASSERT_EQ(procmap->query.vma_end, (unsigned long)ptr + 5 * page_size); } +TEST_F(merge, merge_map_private_dev_zero_unfaulted) +{ + struct procmap_fd *procmap = &self->procmap; + unsigned int page_size = self->page_size; + char *carveout = self->carveout; + char *ptr, *ptr2; + int fd_zero; + + if (access("/dev/zero", F_OK)) + SKIP(return, "No /dev/zero."); + fd_zero = open("/dev/zero", O_RDWR); + ASSERT_NE(fd_zero, -1); + + /* + * Map two MAP_PRIVATE-/dev/zero VMAs next to one another with offset 0 + * each. + * + * With these being made truly anonymous upon mapping, they will + * merge. If they were file-backed VMAs the page offsets would prevent + * the merge: + * + * |-----||------| |-------------| + * | ptr || ptr2 | -> | ptr | + * |-----||------| |-------------| + */ + ptr = mmap(carveout, 5 * page_size, PROT_READ | PROT_WRITE, + MAP_FIXED | MAP_PRIVATE, fd_zero, 0); + ptr2 = mmap(&carveout[5 * page_size], 5 * page_size, + PROT_READ | PROT_WRITE, MAP_FIXED | MAP_PRIVATE, fd_zero, 0); + close(fd_zero); + ASSERT_NE(ptr, MAP_FAILED); + ASSERT_NE(ptr2, MAP_FAILED); + + /* Assert that they merged. */ + ASSERT_TRUE(find_vma_procmap(procmap, ptr)); + ASSERT_EQ(procmap->query.vma_start, (unsigned long)ptr); + ASSERT_EQ(procmap->query.vma_end, (unsigned long)ptr + 10 * page_size); +} + +TEST_F(merge, merge_map_private_dev_zero_faulted_unfaulted) +{ + struct procmap_fd *procmap = &self->procmap; + unsigned int page_size = self->page_size; + char *carveout = self->carveout; + char *ptr, *ptr2; + int fd_zero; + + if (access("/dev/zero", F_OK)) + SKIP(return, "No /dev/zero."); + fd_zero = open("/dev/zero", O_RDWR); + ASSERT_NE(fd_zero, -1); + + /* + * Map a MAP_PRIVATE mapping of /dev/zero with page offset 0, then fault + * it in: + * + * |-------------------------------| + * | faulted | + * |-------------------------------| + */ + ptr = mmap(carveout, 15 * page_size, PROT_READ | PROT_WRITE, + MAP_FIXED | MAP_PRIVATE, fd_zero, 0); + ASSERT_NE(ptr, MAP_FAILED); + memset(ptr, 'x', 15 * page_size); + + /* + * Unmap the middle: + * + * |---------| |---------| + * | faulted | | faulted | + * |---------| |---------| + */ + ASSERT_EQ(munmap(&ptr[5 * page_size], 5 * page_size), 0); + + /* + * Map in a new unfaulted mapping in the middle with page offset 0 - + * this should merge and would not if it were treated as a file rather + * than pure anon: + * + * |---------|-----------|---------| + * | faulted | unfaulted | faulted | + * |---------|-----------|---------| + */ + ptr2 = mmap(&carveout[5 * page_size], 5 * page_size, + PROT_READ | PROT_WRITE, MAP_FIXED | MAP_PRIVATE, + fd_zero, 0); + close(fd_zero); + ASSERT_NE(ptr2, MAP_FAILED); + + /* Assert that they merged. */ + ASSERT_TRUE(find_vma_procmap(procmap, ptr)); + ASSERT_EQ(procmap->query.vma_start, (unsigned long)ptr); + ASSERT_EQ(procmap->query.vma_end, (unsigned long)ptr + 15 * page_size); +} + TEST_F(merge_with_fork, mremap_faulted_to_unfaulted_prev) { struct procmap_fd *procmap = &self->procmap; From f2b1152811f046ea83df285c67cafbe36d19eafa Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Wed, 2 Sep 2026 21:16:50 +0800 Subject: [PATCH 0425/1012] xfs: remove dead kswapd flag inheritance from btree split worker Patch series "mm: replace PF_KCOMPACTD/PF_KSWAPD with kthread_func()". The task_struct->flags field is a limited 32-bit resource. Two flag bits, PF_KCOMPACTD and PF_KSWAPD, are used to identify whether a kernel thread is kcompactd or kswapd. Converting these to use kthread_func() frees up two valuable flag bits for future use. This patch (of 4): Commit 1f6d64829db7 ("xfs: block allocation work needs to be kswapd aware") added PF_MEMALLOC | PF_KSWAPD inheritance to xfs_btree_split_worker() so that block allocation offloaded from kswapd to a workqueue thread could access emergency memory reserves and avoid reclaim throttling. pageout() no longer calls ->writepage() for filesystem folios -- it returns PAGE_ACTIVATE for non-shmem, non-anon pages. kswapd therefore never enters XFS writeback and cannot reach btree split. The only path that offloads btree splits is unwritten extent conversion at IO completion (xfs_end_io -> xfs_iomap_write_unwritten), which runs in a workqueue context where current_is_kswapd() is always false. Let's remove the dead kswapd flag and related codes. Link: https://lore.kernel.org/20260902131653.1338227-1-wangkefeng.wang@huawei.com Link: https://lore.kernel.org/20260902131653.1338227-2-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Reviewed-by: Christoph Hellwig Reviewed-by: Shakeel Butt Cc: Brendan Jackman Cc: Carlos Maiolino Cc: Christian Brauner Cc: "Darrick J. Wong" Cc: David Hildenbrand Cc: Johannes Weiner Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Zi Yan --- fs/xfs/libxfs/xfs_btree.c | 18 +----------------- fs/xfs/xfs_platform.h | 4 ---- 2 files changed, 1 insertion(+), 21 deletions(-) diff --git a/fs/xfs/libxfs/xfs_btree.c b/fs/xfs/libxfs/xfs_btree.c index 60ef7f08b1d300..6738d9d1511bc7 100644 --- a/fs/xfs/libxfs/xfs_btree.c +++ b/fs/xfs/libxfs/xfs_btree.c @@ -2994,7 +2994,6 @@ struct xfs_btree_split_args { struct xfs_btree_cur **curp; int *stat; /* success/failure */ int result; - bool kswapd; /* allocation in kswapd context */ struct completion *done; struct work_struct work; }; @@ -3008,33 +3007,18 @@ xfs_btree_split_worker( { struct xfs_btree_split_args *args = container_of(work, struct xfs_btree_split_args, work); - unsigned long pflags; - unsigned long new_pflags = 0; - - /* - * we are in a transaction context here, but may also be doing work - * in kswapd context, and hence we may need to inherit that state - * temporarily to ensure that we don't block waiting for memory reclaim - * in any way. - */ - if (args->kswapd) - new_pflags |= PF_MEMALLOC | PF_KSWAPD; - - current_set_flags_nested(&pflags, new_pflags); xfs_trans_set_context(args->cur->bc_tp); args->result = __xfs_btree_split(args->cur, args->level, args->ptrp, args->key, args->curp, args->stat); xfs_trans_clear_context(args->cur->bc_tp); - current_restore_flags_nested(&pflags, new_pflags); /* * Do not access args after complete() has run here. We don't own args * and the owner may run and free args before we return here. */ complete(args->done); - } /* @@ -3078,7 +3062,7 @@ xfs_btree_split( args.curp = curp; args.stat = stat; args.done = &done; - args.kswapd = current_is_kswapd(); + INIT_WORK_ONSTACK(&args.work, xfs_btree_split_worker); queue_work(xfs_alloc_wq, &args.work); wait_for_completion(&done); diff --git a/fs/xfs/xfs_platform.h b/fs/xfs/xfs_platform.h index 745d715b4c646b..e961e36c53fce7 100644 --- a/fs/xfs/xfs_platform.h +++ b/fs/xfs/xfs_platform.h @@ -115,10 +115,6 @@ typedef __u32 xfs_nlink_t; #define xfs_blockgc_secs xfs_params.blockgc_timer.val #define current_cpu() (raw_smp_processor_id()) -#define current_set_flags_nested(sp, f) \ - (*(sp) = current->flags, current->flags |= (f)) -#define current_restore_flags_nested(sp, f) \ - (current->flags = ((current->flags & ~(f)) | (*(sp) & (f)))) #define NBBY 8 /* number of bits per byte */ From cea4beb008f4f7813f4e2a87902b7cba1321514e Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Wed, 2 Sep 2026 21:16:51 +0800 Subject: [PATCH 0426/1012] iomap: simplify writepages reclaim guard Now that kswapd can no longer reach filesystem writeback (pageout() returns PAGE_ACTIVATE for non-shmem, non-anon folios), any PF_MEMALLOC context in iomap_writepages() indicates a VM regression, simplify the check by removing PF_KSWAPD. Link: https://lore.kernel.org/20260902131653.1338227-3-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Reviewed-by: Shakeel Butt Cc: Brendan Jackman Cc: Carlos Maiolino Cc: Christian Brauner Cc: Christoph Hellwig Cc: "Darrick J. Wong" Cc: David Hildenbrand Cc: Johannes Weiner Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Vlastimil Babka (SUSE) Cc: Zi Yan --- fs/iomap/buffered-io.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/fs/iomap/buffered-io.c b/fs/iomap/buffered-io.c index 0a5ebfda90f12e..6306ca747f3ba5 100644 --- a/fs/iomap/buffered-io.c +++ b/fs/iomap/buffered-io.c @@ -2077,8 +2077,7 @@ iomap_writepages(struct iomap_writepage_ctx *wpc) * Writeback from reclaim context should never happen except in the case * of a VM regression so warn about it and refuse to write the data. */ - if (WARN_ON_ONCE((current->flags & (PF_MEMALLOC | PF_KSWAPD)) == - PF_MEMALLOC)) + if (WARN_ON_ONCE((current->flags & PF_MEMALLOC))) return -EIO; while ((folio = writeback_iter(mapping, wpc->wbc, folio, &error))) { From 2c6793d6c7315a96cfbdaa5329bdf4133a720ccb Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Wed, 2 Sep 2026 21:16:52 +0800 Subject: [PATCH 0427/1012] mm: replace PF_KSWAPD flag with kthread_func() check The preceding commits removed the last consumer that propagated PF_KSWAPD beyond kswapd itself (XFS btree split worker inheritance). The only remaining setter of PF_KSWAPD is kswapd(), and every current_is_kswapd() caller only needs to check whether the current task *is* the kswapd thread, not whether it inherited the flag. Replace the flag-based test with kthread_func(current) == kswapd, freeing the 0x00020000 PF flag bit. Link: https://lore.kernel.org/20260902131653.1338227-4-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Acked-by: Vlastimil Babka (SUSE) Acked-by: Zi Yan Cc: Brendan Jackman Cc: Carlos Maiolino Cc: Christian Brauner Cc: Christoph Hellwig Cc: "Darrick J. Wong" Cc: David Hildenbrand Cc: Johannes Weiner Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Suren Baghdasaryan --- include/linux/sched.h | 2 +- include/linux/swap.h | 7 +------ mm/vmscan.c | 10 ++++++++-- tools/sched_ext/include/scx/common.bpf.h | 1 - 4 files changed, 10 insertions(+), 10 deletions(-) diff --git a/include/linux/sched.h b/include/linux/sched.h index d35ae49a991f7d..90393bd53bbc67 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -1810,7 +1810,7 @@ extern struct pid __rcu *cad_pid; #define PF_USER_WORKER 0x00004000 /* Kernel thread cloned from userspace thread */ #define PF_NOFREEZE 0x00008000 /* This thread should not be frozen */ #define PF_KCOMPACTD 0x00010000 /* I am kcompactd */ -#define PF_KSWAPD 0x00020000 /* I am kswapd */ +#define PF__HOLE__00020000 0x00020000 #define PF_MEMALLOC_NOFS 0x00040000 /* All allocations inherit GFP_NOFS. See memalloc_nfs_save() */ #define PF_MEMALLOC_NOIO 0x00080000 /* All allocations inherit GFP_NOIO. See memalloc_noio_save() */ #define PF_LOCAL_THROTTLE 0x00100000 /* Throttle writes only against the bdi I write to, diff --git a/include/linux/swap.h b/include/linux/swap.h index 7a43409879caed..74ce794042474c 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -25,12 +25,6 @@ #define SWAP_FLAGS_VALID (SWAP_FLAG_PRIO_MASK | SWAP_FLAG_PREFER | \ SWAP_FLAG_DISCARD | SWAP_FLAG_DISCARD_ONCE | \ SWAP_FLAG_DISCARD_PAGES) - -static inline int current_is_kswapd(void) -{ - return current->flags & PF_KSWAPD; -} - /* * MAX_SWAPFILES defines the maximum number of swaptypes: things which can * be swapped to. The swap type and the offset into that swap type are @@ -339,6 +333,7 @@ void check_move_unevictable_folios(struct folio_batch *fbatch); extern void __meminit kswapd_run(int nid); extern void __meminit kswapd_stop(int nid); +bool current_is_kswapd(void); #ifdef CONFIG_SWAP int add_swap_extent(struct swap_info_struct *sis, unsigned long start_page, diff --git a/mm/vmscan.c b/mm/vmscan.c index ba7adf36e69f7b..245f68c75b2894 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -7539,7 +7539,7 @@ static int kswapd(void *p) * us from recursively trying to free more memory as we're * trying to free the first piece of memory in the first place). */ - tsk->flags |= PF_MEMALLOC | PF_KSWAPD; + tsk->flags |= PF_MEMALLOC; set_freezable(); WRITE_ONCE(pgdat->kswapd_order, 0); @@ -7589,11 +7589,17 @@ static int kswapd(void *p) goto kswapd_try_sleep; } - tsk->flags &= ~(PF_MEMALLOC | PF_KSWAPD); + tsk->flags &= ~PF_MEMALLOC; return 0; } +bool current_is_kswapd(void) +{ + return kthread_func(current) == kswapd; +} +EXPORT_SYMBOL_GPL(current_is_kswapd); + /* * A zone is low on free memory or too fragmented for high-order memory. If * kswapd should reclaim (direct reclaim is deferred), wake it up for the zone's diff --git a/tools/sched_ext/include/scx/common.bpf.h b/tools/sched_ext/include/scx/common.bpf.h index 22f24ebef8a9ae..5a28a36c5120d6 100644 --- a/tools/sched_ext/include/scx/common.bpf.h +++ b/tools/sched_ext/include/scx/common.bpf.h @@ -32,7 +32,6 @@ #define PF_IO_WORKER 0x00000010 /* Task is an IO worker */ #define PF_WQ_WORKER 0x00000020 /* I'm a workqueue worker */ #define PF_KCOMPACTD 0x00010000 /* I am kcompactd */ -#define PF_KSWAPD 0x00020000 /* I am kswapd */ #define PF_KTHREAD 0x00200000 /* I am a kernel thread */ #define PF_EXITING 0x00000004 #define CLOCK_MONOTONIC 1 From 2cb9609a267b4f0b00941de773ec08a4f8c18142 Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Wed, 2 Sep 2026 21:16:53 +0800 Subject: [PATCH 0428/1012] mm: replace PF_KCOMPACTD flag with kthread_func() check PF_KCOMPACTD was introduced by commit ce6d9c1c2b5c ("NFS: fix nfs_release_folio() to not deadlock via kcompactd writeback") so nfs_release_folio() could detect kcompactd context and skip writeback. The flag is only consumed by current_is_kcompactd(), whose sole caller is nfs_release_folio(). Replace the flag-based check with kthread_func(current) == kcompactd, freeing the 0x00010000 PF flag bit. Link: https://lore.kernel.org/20260902131653.1338227-5-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Acked-by: Zi Yan Acked-by: Vlastimil Babka (SUSE) Reviewed-by: David Hildenbrand (Arm) Cc: Brendan Jackman Cc: Carlos Maiolino Cc: Christian Brauner Cc: Christoph Hellwig Cc: "Darrick J. Wong" Cc: Johannes Weiner Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Suren Baghdasaryan --- include/linux/compaction.h | 11 ++++++----- include/linux/sched.h | 2 +- mm/compaction.c | 9 ++++++--- tools/sched_ext/include/scx/common.bpf.h | 1 - 4 files changed, 13 insertions(+), 10 deletions(-) diff --git a/include/linux/compaction.h b/include/linux/compaction.h index 66a2f70e9e019d..691c09f0a6971d 100644 --- a/include/linux/compaction.h +++ b/include/linux/compaction.h @@ -81,10 +81,6 @@ static inline unsigned long compact_gap(unsigned int order) return min(2UL << order, COMPACT_CLUSTER_MAX); } -static inline int current_is_kcompactd(void) -{ - return current->flags & PF_KCOMPACTD; -} #ifdef CONFIG_COMPACTION @@ -103,7 +99,7 @@ extern void compaction_defer_reset(struct zone *zone, int order, bool compaction_zonelist_suitable(struct alloc_context *ac, int order, int alloc_flags, gfp_t gfp_mask); - +bool current_is_kcompactd(void); extern void __meminit kcompactd_run(int nid); extern void __meminit kcompactd_stop(int nid); extern void wakeup_kcompactd(pg_data_t *pgdat, int order, int highest_zoneidx); @@ -120,6 +116,11 @@ static inline bool compaction_suitable(struct zone *zone, int order, return false; } +static inline bool current_is_kcompactd(void) +{ + return false; +} + static inline void kcompactd_run(int nid) { } diff --git a/include/linux/sched.h b/include/linux/sched.h index 90393bd53bbc67..f45b7d43113ca9 100644 --- a/include/linux/sched.h +++ b/include/linux/sched.h @@ -1809,7 +1809,7 @@ extern struct pid __rcu *cad_pid; #define PF_USED_MATH 0x00002000 /* If unset the fpu must be initialized before use */ #define PF_USER_WORKER 0x00004000 /* Kernel thread cloned from userspace thread */ #define PF_NOFREEZE 0x00008000 /* This thread should not be frozen */ -#define PF_KCOMPACTD 0x00010000 /* I am kcompactd */ +#define PF__HOLE__00010000 0x00010000 #define PF__HOLE__00020000 0x00020000 #define PF_MEMALLOC_NOFS 0x00040000 /* All allocations inherit GFP_NOFS. See memalloc_nfs_save() */ #define PF_MEMALLOC_NOIO 0x00080000 /* All allocations inherit GFP_NOIO. See memalloc_noio_save() */ diff --git a/mm/compaction.c b/mm/compaction.c index a049415512c672..4994e200bbecd7 100644 --- a/mm/compaction.c +++ b/mm/compaction.c @@ -3197,7 +3197,6 @@ static int kcompactd(void *p) long default_timeout = msecs_to_jiffies(HPAGE_FRAG_CHECK_INTERVAL_MSEC); long timeout = default_timeout; - current->flags |= PF_KCOMPACTD; set_freezable(); pgdat->kcompactd_max_order = 0; @@ -3254,11 +3253,15 @@ static int kcompactd(void *p) pgdat->proactive_compact_trigger = false; } - current->flags &= ~PF_KCOMPACTD; - return 0; } +bool current_is_kcompactd(void) +{ + return kthread_func(current) == kcompactd; +} +EXPORT_SYMBOL_GPL(current_is_kcompactd); + /* * This kcompactd start function will be called by init and node-hot-add. * On node-hot-add, kcompactd will moved to proper cpus if cpus are hot-added. diff --git a/tools/sched_ext/include/scx/common.bpf.h b/tools/sched_ext/include/scx/common.bpf.h index 5a28a36c5120d6..ab8d676d8cd63f 100644 --- a/tools/sched_ext/include/scx/common.bpf.h +++ b/tools/sched_ext/include/scx/common.bpf.h @@ -31,7 +31,6 @@ #define PF_IDLE 0x00000002 /* I am an IDLE thread */ #define PF_IO_WORKER 0x00000010 /* Task is an IO worker */ #define PF_WQ_WORKER 0x00000020 /* I'm a workqueue worker */ -#define PF_KCOMPACTD 0x00010000 /* I am kcompactd */ #define PF_KTHREAD 0x00200000 /* I am a kernel thread */ #define PF_EXITING 0x00000004 #define CLOCK_MONOTONIC 1 From baf5bef9f5df07af351c702d6d1bd8c451f99ed6 Mon Sep 17 00:00:00 2001 From: Ackerley Tng Date: Wed, 9 Sep 2026 10:11:00 -0700 Subject: [PATCH 0429/1012] mm: hugetlb: return -ENOSPC on memcg charge failure Patch series "Fix bugs in HugeTLB allocation when mem_cgroup_charge_hugetlb() fails", v2. In hugetlb_alloc_folio(), when mem_cgroup_charge_hugetlb() fails, there are 2 issues: 1. free_huge_folio() expects a non-refcounted folio and will VM_BUG_ON_FOLIO(). 2. -ENOMEM is returned, causing an infinite loop retrying the fault. This patch (of 2): When mem_cgroup_charge_hugetlb() fails with -ENOMEM, alloc_hugetlb_folio() currently propagates this error. This results in the page fault handler returning VM_FAULT_OOM. Because HugeTLB allocations are high-order and use __GFP_RETRY_MAYFAIL, they bypass the OOM killer. Returning VM_FAULT_OOM to the #PF handler without triggering the OOM killer (or having it make progress) leads to an infinite loop of retrying the fault. Avoid this loop by returning -ENOSPC when charging fails, which maps to VM_FAULT_SIGBUS, terminating the process cleanly. Make mem_cgroup_charge_hugetlb() fault handling use a common error handling path, the same handling used for hugetlb_cgroup_uncharge_cgroup{,_rsvd}(), which also don't trigger the OOM killer and hence opt to terminate the process with a SIGBUS. Link: https://lore.kernel.org/20260909-hugetlb-alloc-folio-memcg-charge-error-handling-v2-1-4b4a8a19a7f7@google.com Fixes: 991135774c0e ("memcg/hugetlb: introduce mem_cgroup_charge_hugetlb") Signed-off-by: Ackerley Tng Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Reviewed-by: Joshua Hahn Cc: Alex Shi Cc: David Hildenbrand Cc: David Rientjes Cc: Dongliang Mu Cc: Frank van der Linden Cc: Hongxiang Lou Cc: James Houghton Cc: Johannes Weiner Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Ma Wupeng Cc: Miaohe Lin Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Oscar Salvador Cc: Peter Xu Cc: Roman Gushchin Cc: Shakeel Butt Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Yanteng Si Cc: --- mm/hugetlb.c | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 7edc2a860a4007..dfee81e955fcc7 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -2851,7 +2851,6 @@ void wait_for_freed_hugetlb_folios(void) * * Return: A pointer to the allocated folio, or an ERR_PTR on failure. * -ENOSPC if cgroup charging fails or no folio is available. - * -ENOMEM if mem cgroup charging fails. */ struct folio *hugetlb_alloc_folio(struct hstate *h, struct mempolicy_interpreted *mpoli, u8 alloc_flags) @@ -2924,7 +2923,11 @@ struct folio *hugetlb_alloc_folio(struct hstate *h, * were committed to the folio and freeing the folio * would have cleared those up. */ - return ERR_PTR(ret); + /* + * Return -ENOSPC, since retrying the fault is futile: + * the OOM killer is not triggered for HugeTLB. + */ + return ERR_PTR(-ENOSPC); } return folio; From 5190e298b5d3844fcc214b0efb99be5b8eff58dc Mon Sep 17 00:00:00 2001 From: Ackerley Tng Date: Wed, 9 Sep 2026 10:11:01 -0700 Subject: [PATCH 0430/1012] mm: hugetlb: drop refcount before freeing on memcg charge failure When mem_cgroup_charge_hugetlb(folio, gfp) returns -ENOMEM, the folio has its refcount set to 1 via folio_ref_unfreeze(folio, 1). The error path calls free_huge_folio(folio) directly, which expects a refcount of 0. Hence, VM_BUG_ON_FOLIO(folio_ref_count(folio), folio) is triggered. Even with CONFIG_DEBUG_VM disabled, returning a folio with refcount 1 to the freelist can corrupt allocator state later. Use folio_put(folio) instead of free_huge_folio(folio) to properly drop the reference before freeing it. Link: https://lore.kernel.org/20260909-hugetlb-alloc-folio-memcg-charge-error-handling-v2-2-4b4a8a19a7f7@google.com Fixes: 991135774c0e ("memcg/hugetlb: introduce mem_cgroup_charge_hugetlb") Signed-off-by: Ackerley Tng Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Reviewed-by: Joshua Hahn Cc: Alex Shi Cc: David Hildenbrand Cc: David Rientjes Cc: Dongliang Mu Cc: Frank van der Linden Cc: Hongxiang Lou Cc: James Houghton Cc: Johannes Weiner Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Ma Wupeng Cc: Miaohe Lin Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Oscar Salvador Cc: Peter Xu Cc: Roman Gushchin Cc: Shakeel Butt Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Yanteng Si Cc: --- mm/hugetlb.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index dfee81e955fcc7..4d7c2ee126efda 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -2917,7 +2917,7 @@ struct folio *hugetlb_alloc_folio(struct hstate *h, lruvec_stat_mod_folio(folio, NR_HUGETLB, nr_pages); if (ret == -ENOMEM) { - free_huge_folio(folio); + folio_put(folio); /* * Skip uncharging hugetlb_cgroup since the charges * were committed to the folio and freeing the folio From 5eb2c61cd8fc209e98d8373b9ed9dbcca47eeb9f Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Wed, 2 Sep 2026 11:10:11 +0300 Subject: [PATCH 0431/1012] docs/core-api: memory-allocation: add k[mz]alloc_obj() and clarify kmalloc Patch series "docs/core-api: memory-allocation: add k[mz]alloc_obj() and clarify kmalloc", v2. Update the memory-allocation guide to describe k[mz]alloc_obj() and clarify description of kmalloc() size limitations. And when I realized that get_maintainers.pl does not list any of mm people for that patch I added the MAINTAINERS update as well :) This patch (of 2): Since v7.0 the most used memory allocation function is kzalloc_obj(). Update the memory-allocation guide to describe k[mz]alloc_obj() family and make kzalloc_obj() the first answer to "How should I allocate memory?" question. While on it, clarify description of kmalloc() size limitations. Link: https://lore.kernel.org/20260902-docs-memalloc-guide-v2-0-218c1a4dcb80@kernel.org Link: https://lore.kernel.org/20260902-docs-memalloc-guide-v2-1-218c1a4dcb80@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Suren Baghdasaryan Acked-by: SJ Park Acked-by: Vlastimil Babka (SUSE) Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Randy Dunlap --- Documentation/core-api/memory-allocation.rst | 30 +++++++++++++++----- 1 file changed, 23 insertions(+), 7 deletions(-) diff --git a/Documentation/core-api/memory-allocation.rst b/Documentation/core-api/memory-allocation.rst index 0f19dd52432394..823f7fa57429b7 100644 --- a/Documentation/core-api/memory-allocation.rst +++ b/Documentation/core-api/memory-allocation.rst @@ -19,6 +19,12 @@ Diversity of the allocation APIs combined with the numerous GFP flags makes the question "How should I allocate memory?" not that easy to answer, although very likely you should use +:: + + kzalloc_obj(); + +or + :: kzalloc(, GFP_KERNEL); @@ -139,10 +145,13 @@ allocate memory for an array, there are kmalloc_array() and kcalloc() helpers. The helpers struct_size(), array_size() and array3_size() can be used to safely calculate object sizes without overflowing. -The maximal size of a chunk that can be allocated with `kmalloc` is -limited. The actual limit depends on the hardware and the kernel -configuration, but it is a good practice to use `kmalloc` for objects -smaller than page size. +Since 7.0 there are type aware kmalloc-family helpers that let you safely and +conveniently allocate a single object or arrays of objects with kzalloc_obj() +and kmalloc_obj() and their array versions kzalloc_objs() and +kmalloc_objs(). These helpers only need the type of the object that should be +allocated and the count of elements in the array for the array versions. + +As of v7.2, vast majority of the memory allocations use kzalloc_obj(). The address of a chunk allocated with `kmalloc` is aligned to at least ARCH_KMALLOC_MINALIGN bytes. For sizes which are a power of two, the @@ -154,9 +163,16 @@ Chunks allocated with kmalloc() can be resized with krealloc(). Similarly to kmalloc_array(): a helper for resizing arrays is provided in the form of krealloc_array(). -For large allocations you can use vmalloc() and vzalloc(), or directly -request pages from the page allocator. The memory allocated by `vmalloc` -and related functions is not physically contiguous. +`kmalloc` always allocates physically contiguous memory and the maximal size of +a chunk that can be allocated with `kmalloc` is limited by `KMALLOC_MAX_SIZE`, +which matches the page allocator's MAX_PAGE_ORDER limit. + +Internally, the slab allocator differentiates allocations of different orders +and delegates larger allocations to the page allocator, but for the users of +`kmalloc` family it is entirely transparent. + +For large allocations that do not require physically contiguous memory you can +use vmalloc() and vzalloc() family. If you are not sure whether the allocation size is too large for `kmalloc`, it is possible to use kvmalloc() and its derivatives. It will From 90737c6dc7847c6b7099b64cea3ea83dcc214252 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Wed, 2 Sep 2026 11:10:12 +0300 Subject: [PATCH 0432/1012] MAINTAINERS: add memory related docs in core-mm/ to MM - MISC section Previous efforts to make sure that mm files are properly listed in MAINTAINERS missed Documentation/core-api/mm-api.rst Documentation/core-api/memory-allocation.rst Add them to "MEMORY MANAGEMENT - MISC" Link: https://lore.kernel.org/20260902-docs-memalloc-guide-v2-2-218c1a4dcb80@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: SJ Park Acked-by: Vlastimil Babka (SUSE) Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Michal Hocko Cc: Randy Dunlap Cc: Suren Baghdasaryan --- MAINTAINERS | 2 ++ 1 file changed, 2 insertions(+) diff --git a/MAINTAINERS b/MAINTAINERS index 3ff1a8f07e526f..ca76ab10fe68ed 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17267,6 +17267,8 @@ F: Documentation/ABI/testing/sysfs-kernel-mm F: Documentation/ABI/testing/sysfs-kernel-mm-cma F: Documentation/ABI/testing/sysfs-kernel-mm-numa F: Documentation/admin-guide/mm/ +F: Documentation/core-api/memory-allocation.rst +F: Documentation/core-api/mm-api.rst F: Documentation/mm/ F: mm/char-mem.c F: include/linux/cma.h From a78cdbdbc04f0d1be4a154de3d5ea27e0546c5b6 Mon Sep 17 00:00:00 2001 From: Meijing Zhao Date: Wed, 2 Sep 2026 14:45:08 +0800 Subject: [PATCH 0433/1012] mm: trace: decode arm64 and sparc64 VM_ARCH_1 flags Patch series "mm: trace: decode architecture-specific VMA flags", v2. show_vma_flags(), which is used by VMA tracepoints and %pGv, names generic VM_* bits but does not fully decode architecture-specific flag positions. As a result, trace output and VMA dumps can show a generic "arch_1" name or a raw hexadecimal value instead of the meaning assigned by the target architecture. This series adds symbolic names for the arm64 and sparc64 meanings of VM_ARCH_1, arm64 MTE flags, user shadow stack flags, and protection-key encoding bits under their corresponding configuration guards. This patch (of 3): The VM_ARCH_1 bit has architecture-specific meanings. show_vma_flags(), which is used by VMA tracepoints and %pGv, already reports the powerpc, parisc and no-MMU meanings, but falls back to the generic "arch_1" name on arm64 and sparc64. Report VM_ARM64_BTI as "bti" and VM_SPARC_ADI as "adi" so trace output and %pGv dumps expose the actual architecture-specific state. Link: https://lore.kernel.org/cover.1788330432.git.zhaomeijing@lixiang.com Link: https://lore.kernel.org/bcf0fea58240c4b6daa253739683ed7709e42fbb.1788330432.git.zhaomeijing@lixiang.com Signed-off-by: Meijing Zhao Signed-off-by: Andrew Morton Cc: Lorenzo Stoakes Cc: "Masami Hiramatsu (Google)" Cc: Mathieu Desnoyers Cc: Steven Rostedt --- include/trace/events/mmflags.h | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/include/trace/events/mmflags.h b/include/trace/events/mmflags.h index 935893e5ea53bc..02caeaf359ab18 100644 --- a/include/trace/events/mmflags.h +++ b/include/trace/events/mmflags.h @@ -168,6 +168,10 @@ IF_HAVE_PG_ARCH_3(arch_3) #define __VM_ARCH_SPECIFIC_1 {VM_SAO, "sao" } #elif defined(CONFIG_PARISC) #define __VM_ARCH_SPECIFIC_1 {VM_GROWSUP, "growsup" } +#elif defined(CONFIG_SPARC64) +#define __VM_ARCH_SPECIFIC_1 {VM_SPARC_ADI, "adi" } +#elif defined(CONFIG_ARM64) +#define __VM_ARCH_SPECIFIC_1 {VM_ARM64_BTI, "bti" } #elif !defined(CONFIG_MMU) #define __VM_ARCH_SPECIFIC_1 {VM_MAPPED_COPY,"mappedcopy" } #else From a85c1f9b4a9ab55e476050a5f06da21767d81b31 Mon Sep 17 00:00:00 2001 From: Meijing Zhao Date: Wed, 2 Sep 2026 14:45:09 +0800 Subject: [PATCH 0434/1012] mm: trace: decode MTE and shadow stack VMA flags show_vma_flags(), which is used by VMA tracepoints and %pGv, leaves architecture-specific HIGH_ARCH_* bits unnamed. Arm64 MTE flags and user shadow stack flags are therefore printed as raw hexadecimal values. These bit positions are shared between architectures. For example, bit 37 represents VM_MTE_ALLOWED on arm64 but VM_SHADOW_STACK on x86, so a shared HIGH_ARCH_* bit cannot be given an unconditional name. Add conditionally compiled names for VM_MTE, VM_MTE_ALLOWED and VM_SHADOW_STACK. Keep the configuration guards aligned with the definitions of these aliases so each shared bit position is decoded according to the target architecture. For example, an arm64 VMA containing VM_MTE_ALLOWED is now printed as: ...|account|mte_allowed|... instead of: ...|account|0x2000000000 Link: https://lore.kernel.org/1c7f7003cff8d00209baa8498f1efee113f5b5a4.1788330432.git.zhaomeijing@lixiang.com Signed-off-by: Meijing Zhao Signed-off-by: Andrew Morton Cc: Lorenzo Stoakes Cc: "Masami Hiramatsu (Google)" Cc: Mathieu Desnoyers Cc: Steven Rostedt --- include/trace/events/mmflags.h | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/include/trace/events/mmflags.h b/include/trace/events/mmflags.h index 02caeaf359ab18..2f980d20a9f640 100644 --- a/include/trace/events/mmflags.h +++ b/include/trace/events/mmflags.h @@ -202,6 +202,19 @@ IF_HAVE_PG_ARCH_3(arch_3) # define IF_HAVE_VM_DROPPABLE(flag, name) #endif +#ifdef CONFIG_ARM64_MTE +# define IF_HAVE_VM_MTE(flag, name) {flag, name}, +#else +# define IF_HAVE_VM_MTE(flag, name) +#endif + +#if defined(CONFIG_X86_USER_SHADOW_STACK) || defined(CONFIG_RISCV_USER_CFI) || \ + defined(CONFIG_ARM64_GCS) +# define IF_HAVE_VM_SHADOW_STACK(flag, name) {flag, name}, +#else +# define IF_HAVE_VM_SHADOW_STACK(flag, name) +#endif + #define __def_vmaflag_names \ {VM_READ, "read" }, \ {VM_WRITE, "write" }, \ @@ -237,6 +250,9 @@ IF_HAVE_VM_SOFTDIRTY(VM_SOFTDIRTY, "softdirty" ) \ {VM_HUGEPAGE, "hugepage" }, \ {VM_NOHUGEPAGE, "nohugepage" }, \ IF_HAVE_VM_DROPPABLE(VM_DROPPABLE, "droppable" ) \ +IF_HAVE_VM_MTE(VM_MTE, "mte") \ +IF_HAVE_VM_MTE(VM_MTE_ALLOWED, "mte_allowed") \ +IF_HAVE_VM_SHADOW_STACK(VM_SHADOW_STACK, "shadow_stack") \ {VM_MERGEABLE, "mergeable" } \ #define show_vma_flags(flags) \ From c3d398fb357e14fbf326cf527c5c2e5d026b8eb7 Mon Sep 17 00:00:00 2001 From: Meijing Zhao Date: Wed, 2 Sep 2026 14:45:10 +0800 Subject: [PATCH 0435/1012] mm: trace: name protection key encoding bits Protection keys are encoded in architecture-specific HIGH_ARCH_* VMA flag bits. show_vma_flags(), which is used by VMA tracepoints and %pGv, does not name those bits, leaving them as raw hexadecimal values. Name the protection-key encoding bits pkey_bit0 through pkey_bit4 under CONFIG_ARCH_HAS_PKEYS. The names make clear that these are bits of one protection-key value rather than independent protection keys. For example, protection key 3 is represented as: pkey_bit0|pkey_bit1 Honor CONFIG_ARCH_PKEY_BITS when exposing bit 3 and bit 4 so only bits provided by the architecture are included. Link: https://lore.kernel.org/1871da32001243b37147bac5af02d2f2ec64304e.1788330432.git.zhaomeijing@lixiang.com Signed-off-by: Meijing Zhao Signed-off-by: Andrew Morton Cc: Lorenzo Stoakes Cc: "Masami Hiramatsu (Google)" Cc: Mathieu Desnoyers Cc: Steven Rostedt --- include/trace/events/mmflags.h | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/include/trace/events/mmflags.h b/include/trace/events/mmflags.h index 2f980d20a9f640..ef9aa388b84f7d 100644 --- a/include/trace/events/mmflags.h +++ b/include/trace/events/mmflags.h @@ -202,6 +202,24 @@ IF_HAVE_PG_ARCH_3(arch_3) # define IF_HAVE_VM_DROPPABLE(flag, name) #endif +#ifdef CONFIG_ARCH_HAS_PKEYS +# define IF_HAVE_VM_PKEY(flag, name) {flag, name}, +#if CONFIG_ARCH_PKEY_BITS > 3 +# define IF_HAVE_VM_PKEY3(flag, name) {flag, name}, +#else +# define IF_HAVE_VM_PKEY3(flag, name) +#endif +#if CONFIG_ARCH_PKEY_BITS > 4 +# define IF_HAVE_VM_PKEY4(flag, name) {flag, name}, +#else +# define IF_HAVE_VM_PKEY4(flag, name) +#endif +#else +# define IF_HAVE_VM_PKEY(flag, name) +# define IF_HAVE_VM_PKEY3(flag, name) +# define IF_HAVE_VM_PKEY4(flag, name) +#endif + #ifdef CONFIG_ARM64_MTE # define IF_HAVE_VM_MTE(flag, name) {flag, name}, #else @@ -250,6 +268,11 @@ IF_HAVE_VM_SOFTDIRTY(VM_SOFTDIRTY, "softdirty" ) \ {VM_HUGEPAGE, "hugepage" }, \ {VM_NOHUGEPAGE, "nohugepage" }, \ IF_HAVE_VM_DROPPABLE(VM_DROPPABLE, "droppable" ) \ +IF_HAVE_VM_PKEY(VM_PKEY_BIT0, "pkey_bit0") \ +IF_HAVE_VM_PKEY(VM_PKEY_BIT1, "pkey_bit1") \ +IF_HAVE_VM_PKEY(VM_PKEY_BIT2, "pkey_bit2") \ +IF_HAVE_VM_PKEY3(VM_PKEY_BIT3, "pkey_bit3") \ +IF_HAVE_VM_PKEY4(VM_PKEY_BIT4, "pkey_bit4") \ IF_HAVE_VM_MTE(VM_MTE, "mte") \ IF_HAVE_VM_MTE(VM_MTE_ALLOWED, "mte_allowed") \ IF_HAVE_VM_SHADOW_STACK(VM_SHADOW_STACK, "shadow_stack") \ From 6abe3fdf93f4e6ec871a55aafcc3ec498629b2a4 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:34 -0700 Subject: [PATCH 0436/1012] mm/damon/core: use damon_nr_samples_per_aggr() for max merge threshold Patch series "mm/damon: cleanup code, add test cases, and update guidances in docs". Misc cleanup, improvements and updates of code, test, and documents. Patches 1-5 cleanup DAMON code. Patches 6-10 adds kunit and selftest test cases for recently fixed bugs and a new feature. Patches 11 and 12 update guidelines for AI review and what document to read, on DAMON documents. This patch (of 12): kdamond_merge_regions() open-codes max region merge threshold calculation. What it does is fundamentally the same as damon_nr_samples_per_aggr() but missing a few corner cases. The unhandled corner cases should be rare and make only a negligible level of monitoring results degradation. But having the inconsistency could increase future maintenance burden. Use the dedicated function. Link: https://lore.kernel.org/20260902054747.99370-1-sj@kernel.org Link: https://lore.kernel.org/20260902054747.99370-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index cf4ec122a7f99d..1cf4d41f24dcdc 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3533,8 +3533,7 @@ static void kdamond_merge_regions(struct damon_ctx *c, unsigned int threshold, unsigned int max_thres; bool count_age = true; - max_thres = c->attrs.aggr_interval / - (c->attrs.sample_interval ? c->attrs.sample_interval : 1); + max_thres = damon_nr_samples_per_aggr(&c->attrs); while (true) { nr_regions = 0; damon_for_each_target(t, c) { From 0a20670975d65f3ced9d37e38c1ff4a33f71690d Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:35 -0700 Subject: [PATCH 0437/1012] mm/damon/core: remove debug messages There are a few debug messages in DAMON core. Those have not really been used in a meaningful way for the last few years, though. Remove those. Link: https://lore.kernel.org/20260902054747.99370-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Kunwu Chan --- mm/damon/core.c | 9 --------- 1 file changed, 9 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 1cf4d41f24dcdc..4d817063a7a2e5 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3756,10 +3756,6 @@ static unsigned long damos_wmark_wait_us(struct damos *scheme) /* higher than high watermark or lower than low watermark */ if (metric > scheme->wmarks.high || scheme->wmarks.low > metric) { - if (scheme->wmarks.activated) - pr_debug("deactivate a scheme (%d) for %s wmark\n", - scheme->action, - str_high_low(metric > scheme->wmarks.high)); scheme->wmarks.activated = false; return scheme->wmarks.interval; } @@ -3769,8 +3765,6 @@ static unsigned long damos_wmark_wait_us(struct damos *scheme) !scheme->wmarks.activated) return scheme->wmarks.interval; - if (!scheme->wmarks.activated) - pr_debug("activate a scheme (%d)\n", scheme->action); scheme->wmarks.activated = true; return 0; } @@ -3913,8 +3907,6 @@ static int kdamond_fn(void *data) struct damon_ctx *ctx = data; unsigned long sz_limit = 0; - pr_debug("kdamond (%d) starts\n", current->pid); - mutex_lock(&ctx->call_controls_lock); ctx->call_controls_obsolete = false; mutex_unlock(&ctx->call_controls_lock); @@ -4068,7 +4060,6 @@ static int kdamond_fn(void *data) mutex_unlock(&ctx->walk_control_lock); damos_walk_cancel(ctx); - pr_debug("kdamond (%d) finishes\n", current->pid); mutex_lock(&ctx->kdamond_lock); ctx->kdamond = NULL; mutex_unlock(&ctx->kdamond_lock); From 8dad93ec65c3ffaf6bff1a700da24cc6149122cd Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 23:08:16 -0700 Subject: [PATCH 0438/1012] mm/damon/core: remove string_choices.h include It is no more being used. Link: https://lore.kernel.org/20260902061401.104419-1-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Kunwu Chan --- mm/damon/core.c | 1 - 1 file changed, 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 4d817063a7a2e5..88bd2f505cea07 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -15,7 +15,6 @@ #include #include #include -#include /* for damon_get_folio() used by node eligible memory metrics */ #include "ops-common.h" From c2fa366a2c9425518debced3fed3ed5d1bae49b1 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:36 -0700 Subject: [PATCH 0439/1012] mm/damon/vaddr: remove a debug message There is a debug message in the DAMON virtual address space operation set. It has not really been used in a meaningful way for the last few years, though. Remove it. Link: https://lore.kernel.org/20260902054747.99370-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/vaddr.c | 16 +++------------- 1 file changed, 3 insertions(+), 13 deletions(-) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 5b4d16c8db6285..91a0d441c1f940 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -189,22 +189,12 @@ static int damon_va_three_regions(struct damon_target *t, * * */ -static void __damon_va_init_regions(struct damon_ctx *ctx, - struct damon_target *t) +static void __damon_va_init_regions(struct damon_target *t) { - struct damon_target *ti; struct damon_addr_range regions[3]; - int tidx = 0; - if (damon_va_three_regions(t, regions)) { - damon_for_each_target(ti, ctx) { - if (ti == t) - break; - tidx++; - } - pr_debug("Failed to get three regions of %dth target\n", tidx); + if (damon_va_three_regions(t, regions)) return; - } damon_set_regions(t, regions, 3, DAMON_MIN_REGION_SZ); } @@ -217,7 +207,7 @@ static void damon_va_init(struct damon_ctx *ctx) damon_for_each_target(t, ctx) { /* the user may set the target regions as they want */ if (!damon_nr_regions(t)) - __damon_va_init_regions(ctx, t); + __damon_va_init_regions(t); } } From 8464081e7905b0e7eedddd12aa7bb55cb420989c Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:37 -0700 Subject: [PATCH 0440/1012] mm/damon/core: validate number of probes in valid_probe_params() Each DAMON context is allowed to have only up to DAMON_MAX_PROBES probes. The central place for validating DAMON probe parameters, damon_valid_probe_params(), is not validating the upper limit, though. Do the validation. Link: https://lore.kernel.org/20260902054747.99370-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 88bd2f505cea07..87cc5d7568932c 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1419,6 +1419,13 @@ static bool damon_valid_probe_params(struct damon_ctx *ctx) unsigned char max_probe_hits; struct damon_probe *probe; unsigned int wsum, wsum_to_add; + int nr_probes; + + nr_probes = 0; + damon_for_each_probe(probe, ctx) + nr_probes++; + if (nr_probes > DAMON_MAX_PROBES) + return false; if (!damon_has_probe_weights(ctx)) return true; From 94d100f93075b90e5f048d6476e7b903dcc06cc3 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:38 -0700 Subject: [PATCH 0441/1012] mm/damon/sysfs: remove probes number validation DAMON sysfs interface is disallowing >DAMON_MAX_PROBES nr_probes input, since DAMON_MAX_PROBES is the upper limit of probes per DAMON context. The core layer is validating the upper limit again, though. It is preferred to let DAMON API callers such as sysfs interface to set parameters in flexible ways, and do parameters validation in the core layer. Drop the duplicated validation in the sysfs interface. Link: https://lore.kernel.org/20260902054747.99370-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 7ec14f48d157a4..b576e97cbfdb8a 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -1479,7 +1479,7 @@ static ssize_t nr_probes_store(struct kobject *kobj, if (err) return err; - if (nr < 0 || nr > DAMON_MAX_PROBES) + if (nr < 0) return -EINVAL; probes = container_of(kobj, struct damon_sysfs_probes, kobj); From 5abb2181ca837a327a7be3fc439faec80ae6ff3b Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:39 -0700 Subject: [PATCH 0442/1012] mm/damon/tests/core-kunit: extend set_regions() test for error case damon_test_set_regions_for() is designed to test only success-expected damon_set_regions() calls. Extend it to cover error-expected calls, too. Link: https://lore.kernel.org/20260902054747.99370-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Kunwu Chan --- mm/damon/tests/core-kunit.h | 18 ++++++++++-------- 1 file changed, 10 insertions(+), 8 deletions(-) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index f2568fba552ec4..d20ea12ca61633 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -469,11 +469,12 @@ static void damon_test_set_regions_for(struct kunit *test, struct damon_addr_range *old_ranges, int sz_old_ranges, struct damon_addr_range *new_ranges, int sz_new_ranges, unsigned long min_region_sz, - struct damon_addr_range *expect_ranges, int sz_expect_ranges) + struct damon_addr_range *expect_ranges, int sz_expect_ranges, + int expect_err) { struct damon_target *t; struct damon_region *r; - int i; + int i, err; t = damon_new_target(); if (!t) @@ -487,7 +488,8 @@ static void damon_test_set_regions_for(struct kunit *test, damon_add_region(r, t); } - damon_set_regions(t, new_ranges, sz_new_ranges, min_region_sz); + err = damon_set_regions(t, new_ranges, sz_new_ranges, min_region_sz); + KUNIT_EXPECT_EQ(test, err, expect_err); KUNIT_EXPECT_EQ(test, damon_nr_regions(t), sz_expect_ranges); if (damon_nr_regions(t) != sz_expect_ranges) { @@ -516,7 +518,7 @@ static void damon_test_set_regions(struct kunit *test) (struct damon_addr_range[]){ {.start = 5, .end = 15}, {.start = 15, .end = 25}, - }, 2); + }, 2, 0); /* Un-intersecting regions should be removed. */ damon_test_set_regions_for(test, (struct damon_addr_range[]){ @@ -529,7 +531,7 @@ static void damon_test_set_regions(struct kunit *test) 1, (struct damon_addr_range[]){ {.start = 18, .end = 23}, - }, 1); + }, 1, 0); /* * Holes should be filled up with new regions. * @@ -550,7 +552,7 @@ static void damon_test_set_regions(struct kunit *test) {.start = 8, .end = 16}, {.start = 16, .end = 24}, {.start = 24, .end = 28}, - }, 3); + }, 3, 0); /* * New regions should be able to be appended. * @@ -572,7 +574,7 @@ static void damon_test_set_regions(struct kunit *test) {.start = 0, .end = 4}, {.start = 4, .end = 15}, {.start = 25, .end = 40}, - }, 3); + }, 3, 0); /* * New regions should be able to be inserted. * @@ -595,7 +597,7 @@ static void damon_test_set_regions(struct kunit *test) {.start = 0, .end = 15}, {.start = 25, .end = 40}, {.start = 44, .end = 50}, - }, 3); + }, 3, 0); } static void damon_test_update_monitoring_result(struct kunit *test) From 2b5cdc2fff9baa3f19de464d8227068e6e1235fc Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:40 -0700 Subject: [PATCH 0443/1012] mm/damon/tests/core-kunit: test <=0 size damon_set_regions() inputs Commit 1292c0ecb1ca ("mm/damon/core: validate ranges in damon_set_regions()") disallowed passing zero or negative size input ranges to damon_set_regions(). Add kunit test cases for those inputs. Link: https://lore.kernel.org/20260902054747.99370-8-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Kunwu Chan --- mm/damon/tests/core-kunit.h | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index d20ea12ca61633..89f364a8ffeb67 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -598,6 +598,20 @@ static void damon_test_set_regions(struct kunit *test) {.start = 25, .end = 40}, {.start = 44, .end = 50}, }, 3, 0); + /* Zero size regions should return -EINVAL. */ + damon_test_set_regions_for(test, + (struct damon_addr_range[]){}, 0, + (struct damon_addr_range[]){ + {.start = 42, .end = 42}, + }, 1, 1, + (struct damon_addr_range[]){}, 0, -EINVAL); + /* Negative size regions should return -EINVAL. */ + damon_test_set_regions_for(test, + (struct damon_addr_range[]){}, 0, + (struct damon_addr_range[]){ + {.start = 42, .end = 21}, + }, 1, 1, + (struct damon_addr_range[]){}, 0, -EINVAL); } static void damon_test_update_monitoring_result(struct kunit *test) From 324589c80efbfe727df2280b9b69119576275e5a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:41 -0700 Subject: [PATCH 0444/1012] mm/damon/tests/core-kunit: test overlapping ranges for set_regions() Commit 954157679ec3 ("mm/damon/core: disallow overlapping input ranges for damon_set_regions()") disallowed passing overlapping input ranges to damon_set_regions(). Add a kunit test case for the overlapping input. Link: https://lore.kernel.org/20260902054747.99370-9-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/tests/core-kunit.h | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 89f364a8ffeb67..ebb695090bb426 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -612,6 +612,17 @@ static void damon_test_set_regions(struct kunit *test) {.start = 42, .end = 21}, }, 1, 1, (struct damon_addr_range[]){}, 0, -EINVAL); + /* + * Regions resulting in same region after alignment should return + * -EINVAL. + */ + damon_test_set_regions_for(test, + (struct damon_addr_range[]){}, 0, + (struct damon_addr_range[]){ + {.start = 10, .end = 20}, + {.start = 20, .end = 30}, + }, 2, 4096, + (struct damon_addr_range[]){}, 0, -EINVAL); } static void damon_test_update_monitoring_result(struct kunit *test) From bbe77f9c989707abf8a4cc3d3ae594408915c7fa Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:42 -0700 Subject: [PATCH 0445/1012] mm/damon/tests/core-kunit: test damon_nr_samples_per_aggr() damon_max_nr_accesses(), which is a previous version of damon_nr_samples_per_aggr() before the renaming, was wrongly returning zero or random overflowed values for extreme intervals setup. Commit 35d4a3cf70a8 ("mm/damon/ops-common: handle extreme intervals in damon_hot_score()") updated the function to return correct or more valid values. Add a kunit test to ensure it is working as expected. Link: https://lore.kernel.org/20260902054747.99370-10-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/tests/core-kunit.h | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index ebb695090bb426..c01e6a75cadc1e 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -625,6 +625,27 @@ static void damon_test_set_regions(struct kunit *test) (struct damon_addr_range[]){}, 0, -EINVAL); } +static void damon_test_nr_samples_per_aggr(struct kunit *test) +{ + struct damon_attrs attrs = { + .sample_interval = 0, + .aggr_interval = 0, + }; + + /* Zero aggregation interval doesn't cause division by zero */ + KUNIT_EXPECT_EQ(test, damon_nr_samples_per_aggr(&attrs), 1); + + /* + * Too large aggregation interval on 64 bit system doesn't cause + * overflow + */ + if (ULONG_MAX > UINT_MAX) { + attrs.aggr_interval = (unsigned long)UINT_MAX + 1; + KUNIT_EXPECT_EQ(test, damon_nr_samples_per_aggr(&attrs), + UINT_MAX); + } +} + static void damon_test_update_monitoring_result(struct kunit *test) { struct damon_attrs old_attrs = { @@ -1858,6 +1879,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damon_test_split_above_half_progresses), KUNIT_CASE(damon_test_ops_registration), KUNIT_CASE(damon_test_set_regions), + KUNIT_CASE(damon_test_nr_samples_per_aggr), KUNIT_CASE(damon_test_update_monitoring_result), KUNIT_CASE(damon_test_set_attrs), KUNIT_CASE(damon_test_mvsum), From 8bd31a9a0277d0408b6129c478c49baed08e635d Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:43 -0700 Subject: [PATCH 0446/1012] selftests/damon/sysfs.sh: test hugepage_mem_bp quota goal DAMON sysfs quota goal target_metric file now accepts 'hugepage_mem_bp' input. Test it is accepted in fundamental DAMON sysfs file operation selftest. Link: https://lore.kernel.org/20260902054747.99370-11-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/damon/sysfs.sh | 1 + 1 file changed, 1 insertion(+) diff --git a/tools/testing/selftests/damon/sysfs.sh b/tools/testing/selftests/damon/sysfs.sh index ddebde6edabe4e..b66593c9ac471f 100755 --- a/tools/testing/selftests/damon/sysfs.sh +++ b/tools/testing/selftests/damon/sysfs.sh @@ -210,6 +210,7 @@ test_goal() ensure_write_succ "$fpath" "active_mem_bp" "valid input" ensure_write_succ "$fpath" "inactive_mem_bp" "valid input" ensure_write_succ "$fpath" "node_eligible_mem_bp" "valid input" + ensure_write_succ "$fpath" "hugepage_mem_bp" "valid input" ensure_write_fail "$fpath" "foo" "invalid input" ensure_file "$goal_dir/nid" "exist" "600" ensure_file "$goal_dir/path" "exist" "600" From 47693e31097971137249453ba8c917948173d7e8 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:44 -0700 Subject: [PATCH 0447/1012] Docs/mm/damon/maintainer-profile: update AI review for Sashiko replies In the past, sharing Sashiko review results was tedious. Also it was suggested to minimize the recipients on the sharing mails. Hence the AI review section of DAMON maintainer profile document was updated to give guidance about available tools for making the sharing easier, and how the recipients list should be managed. Now DAMON is onboarded [1] to Sashiko's automatic review replies feature. Sashiko directly sends its reviews as replies to the patch mail thread. It also reduces the recipients list to deliver the reporting to only the author and the mailing list. Remove the old guidance that is no more necessary but only confusing. Link: https://lore.kernel.org/20260902054747.99370-12-sj@kernel.org Link: https://github.com/sashiko-dev/sashiko/commit/b554c7b6e733 [1] Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Kunwu Chan --- Documentation/mm/damon/maintainer-profile.rst | 19 +++++-------------- 1 file changed, 5 insertions(+), 14 deletions(-) diff --git a/Documentation/mm/damon/maintainer-profile.rst b/Documentation/mm/damon/maintainer-profile.rst index fb2fa00cc9aa1b..a7c1352339378c 100644 --- a/Documentation/mm/damon/maintainer-profile.rst +++ b/Documentation/mm/damon/maintainer-profile.rst @@ -106,18 +106,9 @@ AI Review For patches that are publicly posted to DAMON mailing list (damon@lists.linux.dev), AI reviews of the patches will be available at -sashiko.dev. The reviews could also be sent as mails to the author of the -patch. - -Patch authors are encouraged to check the AI reviews and share their opinions. -The sharing could be done as a reply to the mail thread. Consider reducing the -recipients list for such sharing, since some people are not really interested -in AI reviews. As a rule of thumb, drop stable@vger.kernel.org and individuals -except DAMON maintainer. - -`hkml` also provides a `feature -`_ -for such sharing. Please feel free to use the feature. +sashiko.dev. The reviews will also be sent as replies to the author of the +patch and the mailing list. -It is only an optional recommendation. DAMON maintainer could also ask any -question about the AI reviews, though. +Patch authors are encouraged to check the AI reviews and share their opinions +by replying on the mail thread. It is only an optional recommendation. DAMON +maintainer could also ask any question about the AI reviews, though. From 2b4e934bd9ec3be60a03ab8bc37a1ea5a80c6419 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 1 Sep 2026 22:47:45 -0700 Subject: [PATCH 0448/1012] Docs/ABI/damon: recommend subsystem doc instead of admin-guide DAMON ABI doc is recommending DAMON admin-guide for people who are willing to know further about DAMON. Nowadays the subsystem doc (Documentation/mm/damon) is a more recommended place for even beginners. Recommend the subsystem doc over admin-guide. Link: https://lore.kernel.org/20260902054747.99370-13-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/ABI/testing/sysfs-kernel-mm-damon | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Documentation/ABI/testing/sysfs-kernel-mm-damon b/Documentation/ABI/testing/sysfs-kernel-mm-damon index f8d2601e829047..ad21f58f3c9126 100644 --- a/Documentation/ABI/testing/sysfs-kernel-mm-damon +++ b/Documentation/ABI/testing/sysfs-kernel-mm-damon @@ -3,7 +3,7 @@ Date: Mar 2022 Contact: SJ Park Description: Interface for Data Access MONitoring (DAMON). Contains files for controlling DAMON. For more details on DAMON itself, - please refer to Documentation/admin-guide/mm/damon/index.rst. + please refer to Documentation/mm/damon/index.rst. What: /sys/kernel/mm/damon/admin/ Date: Mar 2022 From f76589f266ee809b0e36db9f92e87ca817838435 Mon Sep 17 00:00:00 2001 From: zhaozhengzhuo Date: Wed, 2 Sep 2026 11:12:29 +0800 Subject: [PATCH 0449/1012] mm/migrate_device: fix function name in kernel-doc The kernel-doc for migrate_device_range() says that migrate_vma_setup() is similar to itself. Refer to migrate_device_range() as the subject of the comparison, making the distinction between virtual-address-based and device-PFN-based migration clear. Link: https://lore.kernel.org/13768B0F4A5FC1F5+20260902031229.1821112-1-zhaozhengzhuo@uniontech.com Fixes: e778406b40db ("mm/migrate_device.c: add migrate_device_range()") Signed-off-by: zhaozhengzhuo Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Alistair Popple Cc: David Hildenbrand Cc: Byungchul Park Cc: Gregory Price Cc: "Huang, Ying" Cc: Joshua Hahn Cc: Matthew Brost Cc: Rakie Kim --- mm/migrate_device.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/migrate_device.c b/mm/migrate_device.c index 009bfa8b212d5b..0c437004329d9c 100644 --- a/mm/migrate_device.c +++ b/mm/migrate_device.c @@ -1398,9 +1398,9 @@ static unsigned long migrate_device_pfn_lock(unsigned long pfn) * @start: starting pfn in the range to migrate. * @npages: number of pages to migrate. * - * migrate_vma_setup() is similar in concept to migrate_vma_setup() except that - * instead of looking up pages based on virtual address mappings a range of - * device pfns that should be migrated to system memory is used instead. + * migrate_device_range() is similar in concept to migrate_vma_setup(), except + * that instead of looking up pages based on virtual address mappings, a range + * of device pfns that should be migrated to system memory is used. * * This is useful when a driver needs to free device memory but doesn't know the * virtual mappings of every page that may be in device memory. For example this From a322f9617718bdc514cde3f380e14f4832f4d8e2 Mon Sep 17 00:00:00 2001 From: Wei Yang Date: Wed, 24 Jun 2026 08:23:59 +0000 Subject: [PATCH 0450/1012] mm/page_vma_mapped: guard check_pmd() with CONFIG_TRANSPARENT_HUGEPAGE The kernel test robot reported a build failure on the parisc architecture when expanding HPAGE_PMD_NR in check_pmd(). mm/page_vma_mapped.c:142:13: note: in expansion of macro 'HPAGE_PMD_NR' if ((pfn + HPAGE_PMD_NR - 1) < pvmw->pfn) ^~~~~~~~~~~~ The config [1] in report link shows neither TRANSPARENT_HUGEPAGE nor HUGETLB_PAGE is defined. Then trigger the BUILD_BUG. Fix it by define check_pmd() under CONFIG_TRANSPARENT_HUGEPAGE. Link: https://lore.kernel.org/20260624082359.2869-1-richard.weiyang@gmail.com Link: https://download.01.org/0day-ci/archive/20260624/202606240042.ffPsEXVc-lkp@intel.com/config [1] Fixes: 2aff7a4755be ("mm: Convert page_vma_mapped_walk to work on PFNs") Signed-off-by: Wei Yang Signed-off-by: Andrew Morton Reported-by: kernel test robot Closes: https://lore.kernel.org/oe-kbuild-all/202606240042.ffPsEXVc-lkp@intel.com/ Cc: David Hildenbrand Cc: Harry Yoo Cc: Jann Horn Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Matthew Wilcox (Oracle) Cc: Rik van Riel Cc: Vlastimil Babka --- mm/page_vma_mapped.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/page_vma_mapped.c b/mm/page_vma_mapped.c index 4e964545e5e85a..28e306fdb3a5b8 100644 --- a/mm/page_vma_mapped.c +++ b/mm/page_vma_mapped.c @@ -142,6 +142,7 @@ static bool check_pte(struct page_vma_mapped_walk *pvmw, unsigned long pte_nr) return true; } +#ifdef CONFIG_TRANSPARENT_HUGEPAGE /* Returns true if the two ranges overlap. Careful to not overflow. */ static bool check_pmd(unsigned long pfn, struct page_vma_mapped_walk *pvmw) { @@ -151,6 +152,12 @@ static bool check_pmd(unsigned long pfn, struct page_vma_mapped_walk *pvmw) return false; return true; } +#else +static bool check_pmd(unsigned long pfn, struct page_vma_mapped_walk *pvmw) +{ + return false; +} +#endif static void step_forward(struct page_vma_mapped_walk *pvmw, unsigned long size) { From f1d9864144785ae56ba137562b655ffaf6588fb3 Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Thu, 3 Sep 2026 21:52:51 +0800 Subject: [PATCH 0451/1012] selftests/mm: remove unreachable returns after ksft exit helpers The ksft_exit*() helpers such as ksft_exit_fail_msg() are declared __noreturn, and the ksft_exit() and ksft_finished() macros expand to calls of them, always terminating the process via exit(). Any return statements following such calls are unreachable, both at the end of main() and on error paths of helper functions. Remove all of them. No functional change. Link: https://lore.kernel.org/20260903135251.39593-1-zenghui.yu@linux.dev Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: SJ Park Reviewed-by: Zi Yan Assisted-by: GLM-5.3 OpenCode Cc: Kiryl Shutsemau Cc: Baolin Wang Cc: Barry Song Cc: David Hildenbrand Cc: Dev Jain Cc: Lance Yang Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/folio_split_race_test.c | 2 -- tools/testing/selftests/mm/mlock-random-test.c | 1 - tools/testing/selftests/mm/pkey_sighandler_tests.c | 1 - tools/testing/selftests/mm/split_huge_page_test.c | 4 ---- 4 files changed, 8 deletions(-) diff --git a/tools/testing/selftests/mm/folio_split_race_test.c b/tools/testing/selftests/mm/folio_split_race_test.c index 45b84f7b364e0a..1960635a953eb5 100644 --- a/tools/testing/selftests/mm/folio_split_race_test.c +++ b/tools/testing/selftests/mm/folio_split_race_test.c @@ -269,6 +269,4 @@ int main(void) NUM_ITERATIONS); ksft_exit(iter == NUM_ITERATIONS); - - return 0; } diff --git a/tools/testing/selftests/mm/mlock-random-test.c b/tools/testing/selftests/mm/mlock-random-test.c index 16294bc7dae6b2..58772914fd79e1 100644 --- a/tools/testing/selftests/mm/mlock-random-test.c +++ b/tools/testing/selftests/mm/mlock-random-test.c @@ -71,7 +71,6 @@ int get_proc_locked_vm_size(void) fclose(f); ksft_exit_fail_msg("cannot parse VmLck in /proc/self/status: %s\n", strerror(errno)); - return -1; } /* diff --git a/tools/testing/selftests/mm/pkey_sighandler_tests.c b/tools/testing/selftests/mm/pkey_sighandler_tests.c index 74bf79a5399dac..f9c728ba96a559 100644 --- a/tools/testing/selftests/mm/pkey_sighandler_tests.c +++ b/tools/testing/selftests/mm/pkey_sighandler_tests.c @@ -556,5 +556,4 @@ int main(int argc, char *argv[]) } ksft_finished(); - return 0; } diff --git a/tools/testing/selftests/mm/split_huge_page_test.c b/tools/testing/selftests/mm/split_huge_page_test.c index 86a6036928261d..c01d227d7fd6dd 100644 --- a/tools/testing/selftests/mm/split_huge_page_test.c +++ b/tools/testing/selftests/mm/split_huge_page_test.c @@ -101,7 +101,6 @@ static bool is_backed_by_folio(char *vaddr, int order, int pagemap_fd, return (pfn_flags & folio_tail_flags) != folio_tail_flags; fail: ksft_exit_fail_msg("Failed to get folio info\n"); - return false; } static int check_after_split_folio_orders(char *vaddr_start, size_t len, @@ -548,7 +547,6 @@ static int create_pagecache_thp_and_fd(const char *testfile, size_t fd_size, err_out_unlink: unlink(testfile); ksft_exit_fail_msg("Failed to create large pagecache folios\n"); - return -1; } static void split_thp_in_pagecache_to_order_at(size_t fd_size, @@ -711,6 +709,4 @@ int main(int argc, char **argv) free(expected_orders); ksft_finished(); - - return 0; } From b82d0a874eec5d05260e557fe7bec573b98a95e6 Mon Sep 17 00:00:00 2001 From: Christos Skarlos Date: Thu, 3 Sep 2026 12:22:00 +0300 Subject: [PATCH 0452/1012] mm/huge_memory: fix various coding style warnings Resolve coding style issues flagged by checkpatch.pl Specifically: - Add missing blank lines after variable declarations. - Remove unnecessary braces {} for a single statement block. No functional changes are introduced. Link: https://lore.kernel.org/20260903092200.88910-1-christosskarlos.kernel@gmail.com Signed-off-by: Christos Skarlos Signed-off-by: Andrew Morton Reviewed-by: Barry Song Reviewed-by: Zi Yan Reviewed-by: Lorenzo Stoakes (ARM) Cc: Baolin Wang Cc: David Hildenbrand Cc: Dev Jain Cc: Lance Yang Cc: Liam R. Howlett Cc: Ryan Roberts --- mm/huge_memory.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index c5d11147b69aec..dd66c6ad5af13c 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -1146,6 +1146,7 @@ subsys_initcall(hugepage_init); static int __init setup_transparent_hugepage(char *str) { int ret = 0; + if (!str) goto out; if (!strcmp(str, "always")) { @@ -1552,6 +1553,7 @@ static void set_huge_zero_folio(pgtable_t pgtable, struct mm_struct *mm, struct folio *zero_folio) { pmd_t entry; + entry = folio_mk_pmd(zero_folio, vma->vm_page_prot); entry = pmd_mkspecial(entry); pgtable_trans_huge_deposit(mm, pmd, pgtable); @@ -2661,6 +2663,7 @@ bool move_huge_pmd(struct vm_area_struct *vma, unsigned long old_addr, if (pmd_move_must_withdraw(new_ptl, old_ptl, vma)) { pgtable_t pgtable; + pgtable = pgtable_trans_huge_withdraw(mm, old_pmd); pgtable_trans_huge_deposit(mm, new_pmd, pgtable); } @@ -3949,9 +3952,8 @@ int folio_check_splittable(struct folio *folio, unsigned int new_order, * swapcache folio split. Only uniform split to order-0 can be used * here. */ - if ((split_type == SPLIT_TYPE_NON_UNIFORM || new_order) && folio_test_swapcache(folio)) { + if ((split_type == SPLIT_TYPE_NON_UNIFORM || new_order) && folio_test_swapcache(folio)) return -EINVAL; - } if (is_huge_zero_folio(folio)) return -EINVAL; From 5eaf613cf1a8c9353e0444b6ff9cbab287ac1eb5 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Thu, 3 Sep 2026 17:21:26 +0800 Subject: [PATCH 0453/1012] mm/page_owner: preserve original free_pid/free_tgid during folio migration __update_page_owner_free_handle() accepts pid and tgid, but ignores them, always storing current->pid and current->tgid. __folio_copy_owner() forwards the original free pid/tgid during folio migration, yet these values were overwritten by the migrating task. Store the passed-in pid/tgid so the migrated folio keeps the original free attribution. Link: https://lore.kernel.org/20260903092126.24685-1-hongfu.li@linux.dev Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Acked-by: Vlastimil Babka (SUSE) Cc: Brendan Jackman Cc: Johannes Weiner Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Zi Yan --- mm/page_owner.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/page_owner.c b/mm/page_owner.c index 3fc37d9b908ef0..cfc31c92d76570 100644 --- a/mm/page_owner.c +++ b/mm/page_owner.c @@ -307,8 +307,8 @@ static inline void __update_page_owner_free_handle(struct page *page, page_owner->free_handle = handle; } page_owner->free_ts_nsec = free_ts_nsec; - page_owner->free_pid = current->pid; - page_owner->free_tgid = current->tgid; + page_owner->free_pid = pid; + page_owner->free_tgid = tgid; } rcu_read_unlock(); } From eaa50792601e235f7274635557a15bf68474d633 Mon Sep 17 00:00:00 2001 From: Jinmeng Zhou Date: Thu, 3 Sep 2026 15:50:48 +0800 Subject: [PATCH 0454/1012] mm/hugetlb: charge folios to the target mm's memcg HugeTLB folios are currently charged to the memcg of the allocating task. This gives the wrong result when a userfaultfd handler populates a HugeTLB VMA that belongs to another process. The UFFDIO_COPY ioctl operates on the userfaultfd context's mm, but get_mem_cgroup_from_current() charges the folio to the handler's memcg instead. This can be reproduced by placing the faulting process and its userfaultfd handler in different memory cgroups. Have the target process register a HugeTLB mapping with userfaultfd, trigger a missing fault, and let the handler resolve it with UFFDIO_COPY. The hugepage usage is then reported in the handler's memory.current instead of the target's. The generic userfaultfd population path avoids this problem by charging folios to dst_vma->vm_mm. Pass the target mm through hugetlb_alloc_folio() and charge the folio by using get_mem_cgroup_from_mm(). This preserves the existing charge timing and error handling while making HugeTLB userfaultfd population consistent with the generic path. Link: https://lore.kernel.org/20260903075048.3316-1-zhoujinmeng@bytedance.com Fixes: 8cba9576df60 ("hugetlb: memcg: account hugetlb-backed memory in memory controller") Signed-off-by: Jinmeng Zhou Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Reviewed-by: Hongfu Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Michal Hocko Cc: Nhat Pham Cc: Oscar Salvador Cc: Roman Gushchin Cc: Shakeel Butt Cc: --- include/linux/hugetlb.h | 3 ++- include/linux/memcontrol.h | 8 +++++--- mm/hugetlb.c | 9 ++++++--- mm/memcontrol.c | 6 ++++-- 4 files changed, 17 insertions(+), 9 deletions(-) diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 900c95e346b2e3..80a5a03e9cee72 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -694,7 +694,8 @@ enum hugetlb_alloc_flag { #define HUGETLB_ALLOC_USE_GLOBAL_RESERVATIONS BIT(HUGETLB_ALLOC_USE_GLOBAL_RESERVATIONS_BIT) struct folio *hugetlb_alloc_folio(struct hstate *h, - struct mempolicy_interpreted *mpoli, u8 alloc_flags); + struct mempolicy_interpreted *mpoli, struct mm_struct *mm, + u8 alloc_flags); struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma, unsigned long addr, bool cow_from_owner); struct folio *alloc_hugetlb_folio_nodemask(struct hstate *h, int preferred_nid, diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index c799926435560f..058ebd73ff1605 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -647,7 +647,8 @@ static inline int mem_cgroup_charge(struct folio *folio, struct mm_struct *mm, return __mem_cgroup_charge(folio, mm, gfp); } -int mem_cgroup_charge_hugetlb(struct folio* folio, gfp_t gfp); +int mem_cgroup_charge_hugetlb(struct folio *folio, struct mm_struct *mm, + gfp_t gfp); int mem_cgroup_swapin_charge_folio(struct folio *folio, unsigned short id, struct mm_struct *mm, gfp_t gfp); @@ -1146,9 +1147,10 @@ static inline int mem_cgroup_charge(struct folio *folio, return 0; } -static inline int mem_cgroup_charge_hugetlb(struct folio* folio, gfp_t gfp) +static inline int mem_cgroup_charge_hugetlb(struct folio *folio, + struct mm_struct *mm, gfp_t gfp) { - return 0; + return 0; } static inline int mem_cgroup_swapin_charge_folio(struct folio *folio, diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 4d7c2ee126efda..afffaa3d2d3741 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -2844,6 +2844,7 @@ void wait_for_freed_hugetlb_folios(void) * hugetlb_alloc_folio - Allocate a hugetlb folio. * @h: Hugetlb state control block. * @mpoli: Interpreted memory policy to use for allocation. + * @mm: Memory descriptor of the allocation target. * @alloc_flags: Flags controlling the allocation behavior. * * Allocates a hugetlb folio and handles cgroup charging and global hstate @@ -2853,7 +2854,8 @@ void wait_for_freed_hugetlb_folios(void) * -ENOSPC if cgroup charging fails or no folio is available. */ struct folio *hugetlb_alloc_folio(struct hstate *h, - struct mempolicy_interpreted *mpoli, u8 alloc_flags) + struct mempolicy_interpreted *mpoli, struct mm_struct *mm, + u8 alloc_flags) { bool charge_hugetlb_cgroup_rsvd = alloc_flags & HUGETLB_ALLOC_CHARG_CGROUP_RSVD; @@ -2908,7 +2910,8 @@ struct folio *hugetlb_alloc_folio(struct hstate *h, spin_unlock_irq(&hugetlb_lock); - ret = mem_cgroup_charge_hugetlb(folio, gfp | __GFP_RETRY_MAYFAIL); + ret = mem_cgroup_charge_hugetlb(folio, mm, + gfp | __GFP_RETRY_MAYFAIL); /* * Unconditionally increment NR_HUGETLB here because if * mem_cgroup_charge_hugetlb failed, freeing the page will @@ -3051,7 +3054,7 @@ struct folio *alloc_hugetlb_folio(struct vm_area_struct *vma, .nodemask = nodemask, }; - folio = hugetlb_alloc_folio(h, &mpoli, alloc_flags); + folio = hugetlb_alloc_folio(h, &mpoli, vma->vm_mm, alloc_flags); mpol_cond_put(mpol); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index bd1e7e15442659..86ff580c70183a 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5265,6 +5265,7 @@ int __mem_cgroup_charge(struct folio *folio, struct mm_struct *mm, gfp_t gfp) /** * mem_cgroup_charge_hugetlb - charge the memcg for a hugetlb folio * @folio: folio being charged + * @mm: mm context of the allocation target * @gfp: reclaim mode * * This function is called when allocating a huge page folio, after the page has @@ -5274,9 +5275,10 @@ int __mem_cgroup_charge(struct folio *folio, struct mm_struct *mm, gfp_t gfp) * Returns ENOMEM if the memcg is already full. * Returns 0 if either the charge was successful, or if we skip the charging. */ -int mem_cgroup_charge_hugetlb(struct folio *folio, gfp_t gfp) +int mem_cgroup_charge_hugetlb(struct folio *folio, struct mm_struct *mm, + gfp_t gfp) { - struct mem_cgroup *memcg = get_mem_cgroup_from_current(); + struct mem_cgroup *memcg = get_mem_cgroup_from_mm(mm); int ret = 0; /* From bb2559eb16af29f338e77c4f7e7bbb31c2a5e2d8 Mon Sep 17 00:00:00 2001 From: Kaitao Cheng Date: Thu, 3 Sep 2026 13:35:34 +0800 Subject: [PATCH 0455/1012] mm/page-flags: define HWPoison test-and-change helpers unconditionally Page flag helpers for configuration-dependent flags provide false or no-op variants so that their users can be independent of the configuration. The HWPoison helpers do not fully follow this pattern. When CONFIG_MEMORY_FAILURE is enabled, PAGEFLAG() and TESTSCFLAG() provide the regular, test-and-set, and test-and-clear operations. When it is disabled, only PAGEFLAG_FALSE() is instantiated, leaving TestSetPageHWPoison() and TestClearPageHWPoison() undefined. Use TESTSCFLAG_FALSE() to provide the missing accessors when memory failure handling is disabled. Both accessors return false, which is consistent with HWPoison state being unavailable, and makes the accessor interface consistent across configurations. Also remove now-unneeded test_and_clear_pmem_poison() from nvdimm/pmem. Link: https://lore.kernel.org/20260903053535.17611-1-kaitao.cheng@linux.dev Link: https://lore.kernel.org/20260903053535.17611-2-kaitao.cheng@linux.dev Link: https://lore.kernel.org/20260903053535.17611-3-kaitao.cheng@linux.dev Signed-off-by: Kaitao Cheng Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: David Hildenbrand (Arm) Reviewed-by: Oscar Salvador Cc: Alison Schofield Cc: Dave Jiang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vishal Verma Cc: Vlastimil Babka Cc: Ira Weiny --- drivers/nvdimm/pmem.c | 2 +- drivers/nvdimm/pmem.h | 12 ------------ include/linux/page-flags.h | 1 + 3 files changed, 2 insertions(+), 13 deletions(-) diff --git a/drivers/nvdimm/pmem.c b/drivers/nvdimm/pmem.c index 30a51c365ce8ba..5fb86595e8bd7f 100644 --- a/drivers/nvdimm/pmem.c +++ b/drivers/nvdimm/pmem.c @@ -80,7 +80,7 @@ static void pmem_mkpage_present(struct pmem_device *pmem, phys_addr_t offset, * here since we're in the driver I/O path and * outstanding I/O requests pin the dev_pagemap. */ - if (test_and_clear_pmem_poison(page)) + if (TestClearPageHWPoison(page)) clear_mce_nospec(pfn); } } diff --git a/drivers/nvdimm/pmem.h b/drivers/nvdimm/pmem.h index a48509f901968e..76870505dd7971 100644 --- a/drivers/nvdimm/pmem.h +++ b/drivers/nvdimm/pmem.h @@ -1,7 +1,6 @@ /* SPDX-License-Identifier: GPL-2.0 */ #ifndef __NVDIMM_PMEM_H__ #define __NVDIMM_PMEM_H__ -#include #include #include #include @@ -31,15 +30,4 @@ long __pmem_direct_access(struct pmem_device *pmem, pgoff_t pgoff, long nr_pages, enum dax_access_mode mode, void **kaddr, unsigned long *pfn); -#ifdef CONFIG_MEMORY_FAILURE -static inline bool test_and_clear_pmem_poison(struct page *page) -{ - return TestClearPageHWPoison(page); -} -#else -static inline bool test_and_clear_pmem_poison(struct page *page) -{ - return false; -} -#endif #endif /* __NVDIMM_PMEM_H__ */ diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index ae2ebaed6d4d96..3dc79c0c5adf03 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -655,6 +655,7 @@ TESTSCFLAG(HWPoison, hwpoison, PF_ANY) #define __PG_HWPOISON (1UL << PG_hwpoison) #else PAGEFLAG_FALSE(HWPoison, hwpoison) +TESTSCFLAG_FALSE(HWPoison, hwpoison) #define __PG_HWPOISON 0 #endif From 1f5abd9da4fc7cec6fcaeebfecac73005f52754b Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Thu, 3 Sep 2026 11:01:34 +0800 Subject: [PATCH 0456/1012] mm/memfd: fix hugetlb reservation accounting in error paths If hugetlb_add_to_page_cache() in memfd_alloc_folio() fails with -EEXIST, a concurrent fault has already instantiated the folio in the page cache, and the reservation now belongs to that folio. Calling hugetlb_unreserve_pages() in that case incorrectly removes the region backing the cached folio and releases one reservation more than it should, leaving that folio in the page cache with no region recording it. That leaves resv_huge_pages one page short until that folio is removed from the page cache, so available_huge_pages() reports a page that is not actually free and the pool can grant one reservation more than it can back. Applications using hugetlb memfds can then fail to allocate or fault in a page they already reserved. They see -ENOMEM or -ENOSPC from the allocation or fault path even though the reservation and HugePages_Free still look healthy. So hold the hugetlb fault mutex from hugetlb_reserve_pages() until the error-path unreserve completes to make the reserve, allocate and instantiate steps atomic against concurrent faults. With the mutex held from the start, a concurrent fault can no longer consume the reservation between reserve and allocate/instantiate. If a fault completed before the mutex was taken, it has already added the region for that index, so hugetlb_reserve_pages() returns 0 and the error path leaves the region in place. Link: https://lore.kernel.org/20260927094757.31665-1-hongfu.li@linux.dev Link: https://lore.kernel.org/20260903030134.7407-1-hongfu.li@linux.dev Fixes: 717cf9357325 ("mm/memfd: reserve hugetlb folios before allocation") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Cc: Baolin Wang Cc: David Hildenbrand Cc: Hugh Dickins Cc: Muchun Song Cc: Oscar Salvador Cc: Vivek Kasireddy Cc: --- mm/memfd.c | 31 ++++++++++++++++--------------- 1 file changed, 16 insertions(+), 15 deletions(-) diff --git a/mm/memfd.c b/mm/memfd.c index c708d92533f4ab..0f6fff004f5e4c 100644 --- a/mm/memfd.c +++ b/mm/memfd.c @@ -82,22 +82,31 @@ struct folio *memfd_alloc_folio(struct file *memfd, pgoff_t idx) struct hstate *h = hstate_file(memfd); int err = -ENOMEM; long nr_resv; + u32 hash; gfp_mask = htlb_alloc_mask(h); gfp_mask &= ~(__GFP_HIGHMEM | __GFP_MOVABLE); idx >>= huge_page_order(h); + /* + * Serialize hugepage allocation and instantiation to prevent + * races with concurrent allocations, as required by all other + * callers of hugetlb_add_to_page_cache(). + */ + hash = hugetlb_fault_mutex_hash(memfd->f_mapping, idx); + mutex_lock(&hugetlb_fault_mutex_table[hash]); + nr_resv = hugetlb_reserve_pages(inode, idx, idx + 1, NULL, EMPTY_VMA_FLAGS); - if (nr_resv < 0) - return ERR_PTR(nr_resv); + if (nr_resv < 0) { + err = nr_resv; + goto out_unlock; + } folio = alloc_hugetlb_folio_reserve(h, numa_node_id(), NULL, gfp_mask); if (folio) { - u32 hash; - /* * Zero the folio to prevent information leaks to userspace. * Use folio_zero_user() which is optimized for huge/gigantic @@ -112,20 +121,9 @@ struct folio *memfd_alloc_folio(struct file *memfd, pgoff_t idx) */ __folio_mark_uptodate(folio); - /* - * Serialize hugepage allocation and instantiation to prevent - * races with concurrent allocations, as required by all other - * callers of hugetlb_add_to_page_cache(). - */ - hash = hugetlb_fault_mutex_hash(memfd->f_mapping, idx); - mutex_lock(&hugetlb_fault_mutex_table[hash]); - err = hugetlb_add_to_page_cache(folio, memfd->f_mapping, idx); - - mutex_unlock(&hugetlb_fault_mutex_table[hash]); - if (err) { folio_put(folio); goto err_unresv; @@ -133,11 +131,14 @@ struct folio *memfd_alloc_folio(struct file *memfd, pgoff_t idx) hugetlb_set_folio_subpool(folio, subpool_inode(inode)); folio_unlock(folio); + mutex_unlock(&hugetlb_fault_mutex_table[hash]); return folio; } err_unresv: if (nr_resv > 0) hugetlb_unreserve_pages(inode, idx, idx + 1, 0); +out_unlock: + mutex_unlock(&hugetlb_fault_mutex_table[hash]); return ERR_PTR(err); } #endif From 183efe8c903ddeb64f98d0dc1f7888c5a8394168 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:07:19 -0700 Subject: [PATCH 0457/1012] mm/damon/core: error damos_commit_quota_goal() for zero target_value Patch series "mm/damon: move zero damos quota target_value handling to the core layer". Having zero DAMOS quota target value can cause division by zero. DAMON API callers are checking the target value parameters to avoid that. It is easy to make mistakes in some of the multiple API callers. Move that to the core layer. Patch 1 adds the corner case handling into the core layer DAMON parameters validation logic. Patches 2 and 3 remove no more needed DAMON API callers side handling of the corner case in DAMON_LRU_SORT and DMON_SAMPLIE_MTIER, respectively. This patch (of 3): If a DAMOS scheme has a damos_quota_goal of zero target_value, damos_quota_goal() could trigger division-by-zero error. Hence each DAMON API callers should do the zero target_value validation. It is easy to make mistakes. Actually such bugs in DAMON_LRU_SORT and DAMON_SAMPLE_MTIER were found and fixed [1]. It is better to handle the corner case only once in the core layer, instead of multiple places in all DAMON API callers. One straightforward option is using an alternative denominator for the corner case in the damos_quota_goal(). However, the zero target_value is meaningless. In this case, the quota goal is always evaluated as achieved or over-achieved. The quota will only keep being reduced. Simply avoid using zero target_value by adding a check in the core layer DAMOS quota goal parameters validation/commit path, damos_commit_quota_goal(). Update it to return an error in the case. Also update its caller to propagate the error. Link: https://lore.kernel.org/20260903010722.94244-1-sj@kernel.org Link: https://lore.kernel.org/20260903010722.94244-2-sj@kernel.org Link: https://lore.kernel.org/20260803134034.15217-1-sj@kernel.org [1] Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/core.c | 22 ++++++++++++++++------ 1 file changed, 16 insertions(+), 6 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 87cc5d7568932c..15a773614935b9 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1211,14 +1211,17 @@ static void damos_commit_quota_goal_union( } } -static void damos_commit_quota_goal( +static int damos_commit_quota_goal( struct damos_quota_goal *dst, struct damos_quota_goal *src) { + if (!src->target_value) + return -EINVAL; dst->metric = src->metric; dst->target_value = src->target_value; if (dst->metric == DAMOS_QUOTA_USER_INPUT) dst->current_value = src->current_value; damos_commit_quota_goal_union(dst, src); + return 0; } /** @@ -1236,14 +1239,17 @@ static void damos_commit_quota_goal( int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src) { struct damos_quota_goal *dst_goal, *next, *src_goal, *new_goal; - int i = 0, j = 0; + int i = 0, j = 0, err; damos_for_each_quota_goal_safe(dst_goal, next, dst) { src_goal = damos_nth_quota_goal(i++, src); - if (src_goal) - damos_commit_quota_goal(dst_goal, src_goal); - else + if (src_goal) { + err = damos_commit_quota_goal(dst_goal, src_goal); + if (err) + return err; + } else { damos_destroy_quota_goal(dst_goal); + } } damos_for_each_quota_goal_safe(src_goal, next, src) { if (j++ < i) @@ -1252,7 +1258,11 @@ int damos_commit_quota_goals(struct damos_quota *dst, struct damos_quota *src) src_goal->metric, src_goal->target_value); if (!new_goal) return -ENOMEM; - damos_commit_quota_goal(new_goal, src_goal); + err = damos_commit_quota_goal(new_goal, src_goal); + if (err) { + damos_free_quota_goal(new_goal); + return err; + } damos_add_quota_goal(dst, new_goal); } return 0; From c43b49f60e9f925d5aafc9449beb92d55a0110ea Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:07:20 -0700 Subject: [PATCH 0458/1012] Revert "mm/damon/lru_sort: error out for >10000 active_mem_bp" This reverts commit 06befa61c427e74319781e6f35a364cfc32dbae8. The commit was made to avoid zero damos quota goal target value, because it can trigger division-by-zero. Now the core layer handles the corner case. It returns an error for any attempt setting the aero target_value. The corner case handling in DAMON_LRU_SORT is hence no more needed. Remove it. Note that this slightly changes the user behavior. It still disallows active_mem_bp of 10,002. But now it allows other >10,000 active_mem_bp values. Setting >10,000 active_mem_bp makes not much sense. But it doesn't cause critical problems such as memory leak or crash, either. Arguably that doesn't deserve additional code complexity. Just allow it. Link: https://lore.kernel.org/20260903010722.94244-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/lru_sort.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/damon/lru_sort.c b/mm/damon/lru_sort.c index bd847829a99072..7df45f9a0b3aeb 100644 --- a/mm/damon/lru_sort.c +++ b/mm/damon/lru_sort.c @@ -233,8 +233,6 @@ static int damon_lru_sort_add_quota_goals(struct damos *hot_scheme, if (!active_mem_bp) return 0; - if (10000 < active_mem_bp) - return -EINVAL; goal = damos_new_quota_goal(DAMOS_QUOTA_ACTIVE_MEM_BP, active_mem_bp); if (!goal) return -ENOMEM; From 10cf63bfc48f156a26388d0af0f1b02fb889c344 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:07:21 -0700 Subject: [PATCH 0459/1012] Revert "samples/damon/mtier: error out for zero quota goal target values" This reverts commit a16fd3ad9d89b05475864da97327870464611736. The commit was made to avoid zero damos quota goal target value, because it can trigger division-by-zero. Now the core layer handles the corner case. It returns an error for any attempt setting the aero target_value. The corner case handling in DAMON_SAMPLE_MTIER is hence no more needed. Remove it. Link: https://lore.kernel.org/20260903010722.94244-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- samples/damon/mtier.c | 3 --- 1 file changed, 3 deletions(-) diff --git a/samples/damon/mtier.c b/samples/damon/mtier.c index bea45c87cc9bed..27dc88bdf7a0ef 100644 --- a/samples/damon/mtier.c +++ b/samples/damon/mtier.c @@ -161,9 +161,6 @@ static struct damon_ctx *damon_sample_mtier_build_ctx(bool promote) if (!scheme) goto free_out; damon_set_schemes(ctx, &scheme, 1); - /* zero target value causes division by zero in damos_quota_store() */ - if (!node0_mem_used_bp || !node0_mem_free_bp) - goto free_out; quota_goal = damos_new_quota_goal( promote ? DAMOS_QUOTA_NODE_MEM_USED_BP : DAMOS_QUOTA_NODE_MEM_FREE_BP, From e2b75c83a2c1e7d7291ad3da6a93840589d1fe26 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:03:30 -0700 Subject: [PATCH 0460/1012] mm/damon/core: handle NULL ctx parameter in damon_call() Patch series "mm/damon: allow NULL or unstarted damon_ctx parameter for damon_call()". Callers of damon_call() should validate the damon_ctx object parameter. If it is NULL or never damon_start()-ed object, damon_call() could dereference the NULL pointer or indefinitely hang. Ensuring all callers doing the validation correctly has turned out to be difficult. Handle the corner cases inside the core layer and remove callers' validations. Patches 1 and 2 respectively allow passing NULL and not yet damon_start()-ed ctx parameter to damon_call(). Patches 3 and 4 remove the callers side validations in DMON_RECLAIM and DASMON_LRU_SORT, respectively. This patch (of 4): When NULL damon_ctx pointer parameter is passed, damon_call() could do NULL dereference. The caller is responsible to avoid that. It is easy to forget, and there are many damon_call() callers. Meanwhile, damon_call() is never meant to be performance critical. It uses mutex and completion. Add the NULL pointer check inside damon_call() so that callers can pass the parameter without NULL checks. Link: https://lore.kernel.org/20260903010334.93622-1-sj@kernel.org Link: https://lore.kernel.org/20260903010334.93622-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/core.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 15a773614935b9..a4f78d10ef4e5f 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2216,6 +2216,8 @@ int damon_kdamond_pid(struct damon_ctx *ctx) */ int damon_call(struct damon_ctx *ctx, struct damon_call_control *control) { + if (!ctx) + return -EINVAL; if (!control->repeat) init_completion(&control->completion); control->canceled = false; From 82a77e2901432332cad6a1d819a09b4bebaa3dcf Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:03:31 -0700 Subject: [PATCH 0461/1012] mm/damon/core: set ctx->call_controls_obsolete in damon_new_ctx() damon_ctx->call_controls_obsolete is used to disallow damon_call() requests when the request cannot be served. The field is unset and set when the context execution is started and terminated, respectively. The intention is to allow damon_call() requests only while the context is actively being executed. damon_ctx constructor, damon_new_ctx() unsets the field, though. As a result, passing the damon_ctx parameter that never successfully damon_start()-ed to damon_call() can indefinitely hang. The callers should ensure to avoid the case. Such parameter validation is not always simple. Actually such bugs in DAMON_RECLAIM and DAMON_LRU_SORT have been found and fixed [1]. Set the field in damon_new_ctx(), so that DAMON API callers can pass the context parameter to damon_call() without the additional check. Link: https://lore.kernel.org/20260903010334.93622-3-sj@kernel.org Link: https://lore.kernel.org/20260803134646.16640-1-sj@kernel.org [1] Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/core.c | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index a4f78d10ef4e5f..ea6df4311ceb7f 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -933,6 +933,7 @@ struct damon_ctx *damon_new_ctx(void) INIT_LIST_HEAD(&ctx->adaptive_targets); INIT_LIST_HEAD(&ctx->schemes); + ctx->call_controls_obsolete = true; prandom_seed_state(&ctx->rnd_state, get_random_u64()); return ctx; @@ -2206,10 +2207,6 @@ int damon_kdamond_pid(struct damon_ctx *ctx) * synchronization. The return value of the function will be saved in * &damon_call_control->return_code. * - * Note that this function should be called only after damon_start() with the - * @ctx has succeeded. Otherwise, this function could fall into an indefinite - * wait. - * * When this function is failed, the @ctx is guaranteed to be stopped. * * Return: 0 on success, negative error code otherwise. From 07bbd0d6e2d6a8db017a710a5d33a36b9778442a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:03:32 -0700 Subject: [PATCH 0462/1012] mm/damon/reclaim: remove unnecessary damon_call() param validation DAMON_RECLAIM avoids passing NULL or unstarted damon_ctx to damon_call() with its own validation. The validation is no longer needed, because the DAMON core layer now handles the corner cases itself. Remove the unnecessary check. Link: https://lore.kernel.org/20260903010334.93622-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/reclaim.c | 8 -------- 1 file changed, 8 deletions(-) diff --git a/mm/damon/reclaim.c b/mm/damon/reclaim.c index 45d5557cc575a0..42a2c9cb134310 100644 --- a/mm/damon/reclaim.c +++ b/mm/damon/reclaim.c @@ -271,8 +271,6 @@ static int damon_reclaim_commit_inputs_fn(void *arg) return damon_reclaim_apply_parameters(); } -static bool damon_reclaim_damon_has_started; - static int damon_reclaim_commit_inputs_store(const char *val, const struct kernel_param *kp) { @@ -293,10 +291,6 @@ static int damon_reclaim_commit_inputs_store(const char *val, if (!commit_inputs_request) return 0; - /* Skip damon_call() if ctx has not successfully started. */ - if (!damon_reclaim_damon_has_started) - return -EINVAL; - err = damon_call(ctx, &control); return err ? err : control.return_code; @@ -343,8 +337,6 @@ static int damon_reclaim_turn(bool on) err = damon_start(&ctx, 1, true); if (err) return err; - if (!damon_reclaim_damon_has_started) - damon_reclaim_damon_has_started = true; return damon_call(ctx, &call_control); } From a7519b6354e5bea402823dcdcae6c797032a8dc0 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 2 Sep 2026 18:03:33 -0700 Subject: [PATCH 0463/1012] mm/damon/lru_sort: remove unnecessary damon_call() param validation DAMON_LRU_SORT avoids passing NULL or unstarted damon_ctx to damon_call() with its own validation. The validation is no longer needed, because the DAMON core layer now handles the corner cases itself. Remove the unnecessary check. Link: https://lore.kernel.org/20260903010334.93622-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- mm/damon/lru_sort.c | 8 -------- 1 file changed, 8 deletions(-) diff --git a/mm/damon/lru_sort.c b/mm/damon/lru_sort.c index 7df45f9a0b3aeb..ad8e86dd3a93e1 100644 --- a/mm/damon/lru_sort.c +++ b/mm/damon/lru_sort.c @@ -344,8 +344,6 @@ static int damon_lru_sort_commit_inputs_fn(void *arg) return damon_lru_sort_apply_parameters(); } -static bool damon_lru_sort_damon_has_started; - static int damon_lru_sort_commit_inputs_store(const char *val, const struct kernel_param *kp) { @@ -366,10 +364,6 @@ static int damon_lru_sort_commit_inputs_store(const char *val, if (!commit_inputs_request) return 0; - /* Skip damon_call() if ctx has not successfully started. */ - if (!damon_lru_sort_damon_has_started) - return -EINVAL; - err = damon_call(ctx, &control); return err ? err : control.return_code; @@ -420,8 +414,6 @@ static int damon_lru_sort_turn(bool on) err = damon_start(&ctx, 1, true); if (err) return err; - if (!damon_lru_sort_damon_has_started) - damon_lru_sort_damon_has_started = true; return damon_call(ctx, &call_control); } From 3cbad7c015892b8ccda826afce75c78074542b8c Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Wed, 2 Sep 2026 15:55:07 -0400 Subject: [PATCH 0464/1012] mm/memory_hotplug: factor out node_is_memoryless() A memoryless node neither spans present pages (populated or ZONE_DEVICE) nor has an offline-but-added memory block still linked to it in sysfs. try_offline_node() presently open-codes this memoryless check. Pull that into a node_is_memoryless() helper and pull the existing check_no_memblock_for_node_cb() helper ahead of the add/online path so it's clearer what is happening here. No functional change. Link: https://lore.kernel.org/20260902195507.88655-1-gourry@gourry.net Signed-off-by: Gregory Price Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador --- mm/memory_hotplug.c | 60 +++++++++++++++++++++++---------------------- 1 file changed, 31 insertions(+), 29 deletions(-) diff --git a/mm/memory_hotplug.c b/mm/memory_hotplug.c index 226ab9cb078ad1..d0e94057682af6 100644 --- a/mm/memory_hotplug.c +++ b/mm/memory_hotplug.c @@ -1491,6 +1491,36 @@ static int create_altmaps_and_memory_blocks(int nid, struct memory_group *group, return ret; } +static int check_no_memblock_for_node_cb(struct memory_block *mem, void *arg) +{ + int nid = *(int *)arg; + + /* + * If a memory block belongs to multiple nodes, the stored nid is not + * reliable. However, such blocks are always online (e.g., cannot get + * offlined) and, therefore, are still spanned by the node. + */ + return mem->nid == nid ? -EEXIST : 0; +} + +/* Caller must hold the memory hotplug lock for this check. */ +static bool node_is_memoryless(int nid) +{ + /* + * A node still spanning pages (especially ZONE_DEVICE) is not + * memoryless. A node spans memory after move_pfn_range_to_zone(), + * e.g. once a memory block has been onlined. + */ + if (node_spanned_pages(nid)) + return false; + /* + * Offline memory blocks may not be spanned by the node yet, but they + * link to it in sysfs and can be onlined later, so the node is not + * memoryless while any remain. + */ + return !for_each_memory_block(&nid, check_no_memblock_for_node_cb); +} + /* * NOTE: The caller must call lock_device_hotplug() to serialize hotplug * and online/offline operations (triggered e.g. by sysfs). @@ -2214,18 +2244,6 @@ static int check_cpu_on_node(int nid) return 0; } -static int check_no_memblock_for_node_cb(struct memory_block *mem, void *arg) -{ - int nid = *(int *)arg; - - /* - * If a memory block belongs to multiple nodes, the stored nid is not - * reliable. However, such blocks are always online (e.g., cannot get - * offlined) and, therefore, are still spanned by the node. - */ - return mem->nid == nid ? -EEXIST : 0; -} - /** * try_offline_node * @nid: the node ID @@ -2237,23 +2255,7 @@ static int check_no_memblock_for_node_cb(struct memory_block *mem, void *arg) */ void try_offline_node(int nid) { - int rc; - - /* - * If the node still spans pages (especially ZONE_DEVICE), don't - * offline it. A node spans memory after move_pfn_range_to_zone(), - * e.g., after the memory block was onlined. - */ - if (node_spanned_pages(nid)) - return; - - /* - * Especially offline memory blocks might not be spanned by the - * node. They will get spanned by the node once they get onlined. - * However, they link to the node in sysfs and can get onlined later. - */ - rc = for_each_memory_block(&nid, check_no_memblock_for_node_cb); - if (rc) + if (!node_is_memoryless(nid)) return; if (check_cpu_on_node(nid)) From b0a37bea82f094e37159fe8db81cb778ee05849b Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Fri, 4 Sep 2026 18:54:43 -0700 Subject: [PATCH 0465/1012] mm-memory_hotplug-factor-out-node_is_memoryless-fix move node_is_memoryless() inside CONFIG_MEMORY_HOTREMOVE Reported-by: kernel test robot Closes: https://lore.kernel.org/oe-kbuild-all/202609050628.ywCLhOj5-lkp@intel.com/ Cc: David Hildenbrand Cc: Gregory Price Cc: Oscar Salvador Signed-off-by: Andrew Morton --- mm/memory_hotplug.c | 61 +++++++++++++++++++++++---------------------- 1 file changed, 31 insertions(+), 30 deletions(-) diff --git a/mm/memory_hotplug.c b/mm/memory_hotplug.c index d0e94057682af6..b428da66d279c0 100644 --- a/mm/memory_hotplug.c +++ b/mm/memory_hotplug.c @@ -1491,36 +1491,6 @@ static int create_altmaps_and_memory_blocks(int nid, struct memory_group *group, return ret; } -static int check_no_memblock_for_node_cb(struct memory_block *mem, void *arg) -{ - int nid = *(int *)arg; - - /* - * If a memory block belongs to multiple nodes, the stored nid is not - * reliable. However, such blocks are always online (e.g., cannot get - * offlined) and, therefore, are still spanned by the node. - */ - return mem->nid == nid ? -EEXIST : 0; -} - -/* Caller must hold the memory hotplug lock for this check. */ -static bool node_is_memoryless(int nid) -{ - /* - * A node still spanning pages (especially ZONE_DEVICE) is not - * memoryless. A node spans memory after move_pfn_range_to_zone(), - * e.g. once a memory block has been onlined. - */ - if (node_spanned_pages(nid)) - return false; - /* - * Offline memory blocks may not be spanned by the node yet, but they - * link to it in sysfs and can be onlined later, so the node is not - * memoryless while any remain. - */ - return !for_each_memory_block(&nid, check_no_memblock_for_node_cb); -} - /* * NOTE: The caller must call lock_device_hotplug() to serialize hotplug * and online/offline operations (triggered e.g. by sysfs). @@ -1815,6 +1785,37 @@ bool mhp_range_allowed(u64 start, u64 size, bool need_mapping) } #ifdef CONFIG_MEMORY_HOTREMOVE + +static int check_no_memblock_for_node_cb(struct memory_block *mem, void *arg) +{ + int nid = *(int *)arg; + + /* + * If a memory block belongs to multiple nodes, the stored nid is not + * reliable. However, such blocks are always online (e.g., cannot get + * offlined) and, therefore, are still spanned by the node. + */ + return mem->nid == nid ? -EEXIST : 0; +} + +/* Caller must hold the memory hotplug lock for this check. */ +static bool node_is_memoryless(int nid) +{ + /* + * A node still spanning pages (especially ZONE_DEVICE) is not + * memoryless. A node spans memory after move_pfn_range_to_zone(), + * e.g. once a memory block has been onlined. + */ + if (node_spanned_pages(nid)) + return false; + /* + * Offline memory blocks may not be spanned by the node yet, but they + * link to it in sysfs and can be onlined later, so the node is not + * memoryless while any remain. + */ + return !for_each_memory_block(&nid, check_no_memblock_for_node_cb); +} + /* * Scan pfn range [start,end) to find movable/migratable pages (LRU and * hugetlb folio, movable_ops pages). Will skip over most unmovable From b4801e6f7506fbb6b08edb5802583a58cbaad4f3 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sat, 5 Sep 2026 15:08:30 -0400 Subject: [PATCH 0466/1012] mm: remove PageWriteback The last caller of PageWriteback() was removed in commit f75987e543c2 ("libceph: remove pinning assertion in ceph_msg_data_iter_next()"). This flag is now only used on folios, so we can remove all the page accessors. folio_test_clear_writeback() is not used, so don't add FOLIO_TEST_CLEAR_FLAG() for it. Link: https://lore.kernel.org/20260905-remove-pagewriteback-v1-1-06f41c00db03@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: SJ Park Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/page-flags.h | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index 3dc79c0c5adf03..86dd0470da1173 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -588,8 +588,8 @@ FOLIO_FLAG(owner_2, FOLIO_HEAD_PAGE) * Only test-and-set exist for PG_writeback. The unconditional operators are * risky: they bypass page accounting. */ -TESTPAGEFLAG(Writeback, writeback, PF_NO_TAIL) - TESTSCFLAG(Writeback, writeback, PF_NO_TAIL) +FOLIO_TEST_FLAG(writeback, FOLIO_HEAD_PAGE) + FOLIO_TEST_SET_FLAG(writeback, FOLIO_HEAD_PAGE) FOLIO_FLAG(mappedtodisk, FOLIO_HEAD_PAGE) /* PG_readahead is only used for reads; PG_reclaim is only for writes */ From 375c816bef9c87a152677516b4c4eb5777ee369f Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Sun, 6 Sep 2026 00:51:06 +0800 Subject: [PATCH 0467/1012] mm/memcontrol: move the lru_zone_size sanity check to the reader side Patch series "mm/mglru: clean up folio counters and flag usage", v6. This is a cleanup series separated out from the MGLRU-FG series [1]. As that series is getting too long in following updates, separate out the clean up part for easier review and merge. No feature change is intended, except one bugfix. It mostly replaces the open-coded bit operations scattered throughout the MGLRU code with new helpers, with proper kdocs, sanity debug checks, and hardens a few MGLRU functions. A subtle generation counter leak is also found during the refactoring and the fix is included. Also collected review feedbacks on the cleanup part from the posted series. This patch (of 6): Instead of using an unsigned long and checking the counter value at the updater side, turn the counter into a signed long and check at the reader side. This reduces overhead and simplifies the code. commit ca707239e8a7 ("mm: update_lru_size warn and reset bad lru_size") added a sanity check for memcg counter underflow: lru_zone_size is unsigned, so an underflow wraps it around and returns an enormously large number, then the memcg shrinker loops almost forever as the calculated number of folios to shrink is huge. It also checked if a zero value matches the empty LRU list, so the positive and negative deltas had to be handled separately. However that emptiness check was already removed by commit b4536f0c829c ("mm, memcg: fix the active list aging for lowmem requests when memcg is enabled"), so handling the deltas separately is no longer needed. The remaining update-side check is costly and cannot really catch the leak it is after anyway. It runs on every LRU folio, and if a folio was removed without updating the counter while other folios remain on the LRU, the WARN only triggers much later, from a likely innocent callsite. While readers are much rarer than writers, only the reclaim and reparenting paths read it, once per batch. Checking at the reader side instead leaves the update path a plain addition, and puts the warning where the value is actually consumed. Note this changes the behavior on underflow: the correction is removed and a negative value is kept. A massive leak of the LRU size counter would indicate that something else has gone very wrong, and one should fix that leaking site instead. Besides, the original behavior might cause false positives, or make things worse if the accounting happens after the actual insertion: the value is not leaked, just delayed, so force-fixing it would cause a bigger problem. The warning now only kicks in when a consumer actually uses it, in which case the reader gets zero. Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-0-9aacbd77d4ca@tencent.com Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-1-9aacbd77d4ca@tencent.com Link: https://lore.kernel.org/linux-mm/20260804-mglru-fg-v1-0-4d8dad39dad6@tencent.com/ [1] Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Ridong Chen Reviewed-by: Barry Song Acked-by: Shakeel Butt Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: Zi Yan Cc: Lian Wang Cc: Qi Zheng --- include/linux/memcontrol.h | 9 +++++++-- mm/memcontrol.c | 18 +----------------- 2 files changed, 8 insertions(+), 19 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 058ebd73ff1605..a03b6e3e547078 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -100,7 +100,7 @@ struct mem_cgroup_per_node { /* Fields which get updated often at the end. */ struct lruvec lruvec; CACHELINE_PADDING(_pad2_); - unsigned long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS]; + long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS]; struct mem_cgroup_reclaim_iter iter; /* @@ -888,10 +888,15 @@ static inline unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec, enum lru_list lru, int zone_idx) { + long val; struct mem_cgroup_per_node *mz; mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec); - return READ_ONCE(mz->lru_zone_size[zone_idx][lru]); + val = READ_ONCE(mz->lru_zone_size[zone_idx][lru]); + if (WARN_ON_ONCE(val < 0)) + return 0; + + return val; } void __mem_cgroup_handle_over_high(gfp_t gfp_mask); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 86ff580c70183a..4cb2db8c0923a4 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -1529,28 +1529,12 @@ void mem_cgroup_update_lru_size(struct lruvec *lruvec, enum lru_list lru, int zid, long nr_pages) { struct mem_cgroup_per_node *mz; - unsigned long *lru_size; - long size; if (mem_cgroup_disabled()) return; mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec); - lru_size = &mz->lru_zone_size[zid][lru]; - - if (nr_pages < 0) - *lru_size += nr_pages; - - size = *lru_size; - if (WARN_ONCE(size < 0, - "%s(%p, %d, %ld): lru_size %ld\n", - __func__, lruvec, lru, nr_pages, size)) { - VM_BUG_ON(1); - *lru_size = 0; - } - - if (nr_pages > 0) - *lru_size += nr_pages; + mz->lru_zone_size[zid][lru] += nr_pages; } /** From 03839e67cabfcd385caddf060fbadd525583d8e5 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Sun, 6 Sep 2026 00:51:07 +0800 Subject: [PATCH 0468/1012] mm/mglru: introduce helpers for manipulating gen and refs flags Instead of doing bit ops on folio->flags.f, introduce helpers for adjusting a folio's refs and generation info, making the code easier to debug and understand. No functional change is intended: some combined atomic operations are split into two, which only creates harmless transient states. There is no measurable performance impact, and some paths even look slightly better in the generated assembly. Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-2-9aacbd77d4ca@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Reviewed-by: Baolin Wang Reviewed-by: Barry Song Cc: Axel Rasmussen Cc: Baoquan He Cc: Chris Li Cc: David Hildenbrand (Arm) Cc: Johannes Weiner Cc: Kairui Song Cc: Liam R. Howlett Cc: Lian Wang Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Ridong Chen Cc: Roman Gushchin Cc: Shakeel Butt Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: Zi Yan --- include/linux/mm_inline.h | 84 +++++++++++++++++++++++++++++++++++---- include/linux/mmzone.h | 1 + mm/folio.c | 19 +++++---- mm/vmscan.c | 60 ++++++++++++++++------------ 4 files changed, 122 insertions(+), 42 deletions(-) diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h index 621c8653d8f7eb..3f4bd5b02b54da 100644 --- a/include/linux/mm_inline.h +++ b/include/linux/mm_inline.h @@ -142,10 +142,66 @@ static inline int lru_tier_from_refs(int refs, bool workingset) return workingset ? MAX_NR_TIERS - 1 : order_base_2(refs); } -static inline int folio_lru_refs(const struct folio *folio) +/** + * lru_set_gen_flags - Set the LRU generation number to specified folio flags. + * @flags: pointer to the folio flags + * @gen: generation number, between 0 and (MAX_NR_GENS - 1), inclusive. + */ +static inline void lru_set_gen_flags(unsigned long *flags, int gen) +{ + BUILD_BUG_ON(LRU_GEN_MASK & LRU_REFS_MASK); + VM_WARN_ON_ONCE(gen >= MAX_NR_GENS || gen < 0); + /* Store gen offset by 1, zero means the folio is off-list. */ + *flags &= ~LRU_GEN_MASK; + *flags |= (gen + 1UL) << LRU_GEN_PGOFF; +} + +/** + * lru_get_gen_flags - Return the LRU generation number from folio flags. + * @flags: folio flags + * + * Returns: A number between 0 and (MAX_NR_GENS - 1), inclusive. Returns + * -1 if the flags indicate the folio is off the list (e.g., isolated). + */ +static inline int lru_get_gen_flags(unsigned long flags) { - unsigned long flags = READ_ONCE(folio->flags.f); + int gen = ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; + /* Exclude the legal -1 from the unsigned MAX_NR_GENS comparison */ + VM_WARN_ON_ONCE(gen != -1 && gen >= MAX_NR_GENS); + return gen; +} + +/** + * lru_set_refs_flags - Set the LRU referenced count to folio flags. + * @flags: pointer to the folio flags + * @refs: referenced / access count number, between 0 and LRU_REFS_MAX, inclusive. + * + * For MGLRU, PG_referenced holds the first ref, and the extra bits hold the + * remaining refs. For classical LRU the extra bits are not used, so it can + * also be seen as the refs count never exceeds 1. In both cases, refs == 1 + * means PG_referenced is set and the extra bits are zero, and refs == 0 means + * PG_referenced and the extra bits are all unset. + */ +static inline void lru_set_refs_flags(unsigned long *flags, unsigned int refs) +{ + VM_WARN_ON_ONCE(refs > LRU_REFS_MAX); + BUILD_BUG_ON(LRU_REFS_MAX != (LRU_REFS_MASK >> LRU_REFS_PGOFF) + 1); + + *flags &= ~LRU_REFS_FLAGS; + if (!refs) + return; + *flags |= (BIT(PG_referenced) | ((refs - 1UL) << LRU_REFS_PGOFF)); +} + +/** + * lru_get_refs_flags - Return LRU referenced / access count from folio flags. + * @flags: folio flags + * + * Reads the LRU referenced count set by lru_set_refs_flags(). + */ +static inline int lru_get_refs_flags(unsigned long flags) +{ if (!(flags & BIT(PG_referenced))) return 0; /* @@ -155,11 +211,24 @@ static inline int folio_lru_refs(const struct folio *folio) return ((flags & LRU_REFS_MASK) >> LRU_REFS_PGOFF) + 1; } -static inline int folio_lru_gen(const struct folio *folio) +static inline int folio_lru_refs(const struct folio *folio) { - unsigned long flags = READ_ONCE(folio->flags.f); + return lru_get_refs_flags(READ_ONCE(*const_folio_flags(folio, 0))); +} + +static inline void folio_set_lru_refs(struct folio *folio, unsigned int refs) +{ + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); + + do { + new_flags = old_flags; + lru_set_refs_flags(&new_flags, refs); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); +} - return ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; +static inline int folio_lru_gen(const struct folio *folio) +{ + return lru_get_gen_flags(READ_ONCE(*const_folio_flags(folio, 0))); } static inline bool lru_gen_is_active(const struct lruvec *lruvec, int gen) @@ -270,7 +339,7 @@ static inline bool lru_gen_add_folio(struct lruvec *lruvec, struct folio *folio, gen = lru_gen_from_seq(seq); flags = (gen + 1UL) << LRU_GEN_PGOFF; /* see the comment on MIN_NR_GENS about PG_active */ - set_mask_bits(&folio->flags.f, LRU_GEN_MASK | BIT(PG_active), flags); + set_mask_bits(folio_flags(folio, 0), LRU_GEN_MASK | BIT(PG_active), flags); lru_gen_update_size(lruvec, folio, -1, gen); /* for folio_rotate_reclaimable() */ @@ -295,7 +364,7 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, /* for folio_migrate_flags() */ flags = !reclaiming && lru_gen_is_active(lruvec, gen) ? BIT(PG_active) : 0; - flags = set_mask_bits(&folio->flags.f, LRU_GEN_MASK, flags); + flags = set_mask_bits(folio_flags(folio, 0), LRU_GEN_MASK, flags); gen = ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; lru_gen_update_size(lruvec, folio, gen, -1); @@ -339,7 +408,6 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, static inline void folio_migrate_refs(struct folio *new, const struct folio *old) { - } #endif /* CONFIG_LRU_GEN */ diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index c070b867e2f349..593cc92fe47146 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -500,6 +500,7 @@ enum lruvec_flags { #define LRU_GEN_MASK ((BIT(LRU_GEN_WIDTH) - 1) << LRU_GEN_PGOFF) #define LRU_REFS_MASK ((BIT(LRU_REFS_WIDTH) - 1) << LRU_REFS_PGOFF) +#define LRU_REFS_MAX BIT(LRU_REFS_WIDTH) /* * For folios accessed multiple times through file descriptors, diff --git a/mm/folio.c b/mm/folio.c index 50a6dbe55998e7..47a437e0f7fde6 100644 --- a/mm/folio.c +++ b/mm/folio.c @@ -354,26 +354,28 @@ static void __lru_cache_activate_folio(struct folio *folio) static void lru_gen_inc_refs(struct folio *folio) { - unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f); + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); + int refs; if (folio_test_unevictable(folio)) return; /* see the comment on LRU_REFS_FLAGS */ - if (!folio_test_referenced(folio)) { - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced)); + if (!folio_lru_refs(folio)) { + folio_set_lru_refs(folio, 1); return; } do { - if ((old_flags & LRU_REFS_MASK) == LRU_REFS_MASK) { + new_flags = old_flags; + refs = lru_get_refs_flags(old_flags); + if (refs == LRU_REFS_MAX) { if (!folio_test_workingset(folio)) folio_set_workingset(folio); return; } - - new_flags = old_flags + BIT(LRU_REFS_PGOFF); - } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags)); + lru_set_refs_flags(&new_flags, refs + 1); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); } static bool lru_gen_clear_refs(struct folio *folio) @@ -385,7 +387,8 @@ static bool lru_gen_clear_refs(struct folio *folio) if (gen < 0) return true; - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS | BIT(PG_workingset), 0); + folio_set_lru_refs(folio, 0); + folio_clear_workingset(folio); rcu_read_lock(); seq = READ_ONCE(folio_lruvec(folio)->lrugen.min_seq[type]); diff --git a/mm/vmscan.c b/mm/vmscan.c index 245f68c75b2894..c6e03a5047aeb9 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -863,19 +863,22 @@ static bool lru_gen_set_refs(struct folio *folio, const vma_flags_t *vma_flags) if (!folio_test_referenced(folio) && !folio_test_workingset(folio)) { /* Activate file-backed executable folios after first usage. */ if (is_exec_file_folio(folio, vma_flags)) { - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_workingset)); + folio_set_workingset(folio); + folio_set_lru_refs(folio, 0); return true; } - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced)); + folio_set_lru_refs(folio, 1); return false; } /* Promote on second access */ - if (folio_lru_refs(folio) > 1) - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_workingset)); - else + if (folio_lru_refs(folio) > 1) { + folio_set_workingset(folio); + folio_set_lru_refs(folio, 0); + } else { folio_mark_accessed(folio); + } return true; } #else @@ -3291,11 +3294,10 @@ static bool positive_ctrl_err(struct ctrl_pos *sp, struct ctrl_pos *pv) ******************************************************************************/ /* promote pages accessed through page tables */ -static int folio_update_gen(struct folio *folio, int gen, const vma_flags_t *vma_flags) +static int folio_update_gen(struct folio *folio, int new_gen, const vma_flags_t *vma_flags) { - unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f); - - VM_WARN_ON_ONCE(gen >= MAX_NR_GENS); + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); + int old_gen; /* * See the comment on LRU_REFS_FLAGS, and activate file-backed @@ -3304,31 +3306,34 @@ static int folio_update_gen(struct folio *folio, int gen, const vma_flags_t *vma */ if (!folio_test_referenced(folio) && !folio_test_workingset(folio) && !is_exec_file_folio(folio, vma_flags)) { - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced)); + folio_set_lru_refs(folio, 1); return -1; } do { + old_gen = lru_get_gen_flags(old_flags); + new_flags = old_flags; + /* lru_gen_del_folio() has isolated this page? */ - if (!(old_flags & LRU_GEN_MASK)) - return -1; + if (old_gen < 0) + break; - new_flags = old_flags & ~(LRU_GEN_MASK | LRU_REFS_FLAGS); - new_flags |= ((gen + 1UL) << LRU_GEN_PGOFF) | BIT(PG_workingset); - } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags)); + lru_set_gen_flags(&new_flags, new_gen); + lru_set_refs_flags(&new_flags, 0); + new_flags |= BIT(PG_workingset); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); - return ((old_flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; + return old_gen; } static int __folio_inc_gen(struct folio *folio, int old_gen, bool *increased) { - unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f); + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); int new_gen; - VM_WARN_ON_ONCE_FOLIO(!(old_flags & LRU_GEN_MASK), folio); - do { - new_gen = ((old_flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; + new_gen = lru_get_gen_flags(old_flags); + /* folio_update_gen() has promoted this page? */ if (new_gen >= 0 && new_gen != old_gen) { if (increased) @@ -3336,11 +3341,12 @@ static int __folio_inc_gen(struct folio *folio, int old_gen, bool *increased) return new_gen; } + new_flags = old_flags; new_gen = (old_gen + 1) % MAX_NR_GENS; - new_flags = old_flags & ~(LRU_GEN_MASK | LRU_REFS_FLAGS); - new_flags |= (new_gen + 1UL) << LRU_GEN_PGOFF; - } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags)); + lru_set_gen_flags(&new_flags, new_gen); + lru_set_refs_flags(&new_flags, 0); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); if (increased) *increased = true; @@ -4785,7 +4791,7 @@ static bool isolate_folio(struct lruvec *lruvec, struct folio *folio, struct sca /* see the comment on LRU_REFS_FLAGS */ if (!folio_test_referenced(folio)) - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, 0); + folio_set_lru_refs(folio, 0); success = lru_gen_del_folio(lruvec, folio, true); VM_WARN_ON_ONCE_FOLIO(!success, folio); @@ -5017,8 +5023,10 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec, } /* don't add rejected folios to the oldest generation */ - if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_active)); + if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) { + folio_set_lru_refs(folio, 0); + folio_set_active(folio); + } } move_folios_to_lru(&list); From 599d414f3ae30fe7d8c7a930e9a38022a31e7586 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Sun, 6 Sep 2026 00:51:08 +0800 Subject: [PATCH 0469/1012] mm/migrate: copy all referenced state via folio_migrate_lru_refs folio_migrate_flags() copies PG_referenced separately from the MGLRU refs counter, which folio_migrate_refs() transfers. Yet under MGLRU, PG_referenced and the refs counter bits together describe the referenced status of a folio. Consolidate the two: rename folio_migrate_refs() to folio_migrate_lru_refs() and let it copy the complete referenced status, i.e., the MGLRU refs count including PG_referenced, or just PG_referenced for the active/inactive LRU. Drop the open-coded PG_referenced copy so the referenced status is transferred in one place. No behavior change is intended: under the active/inactive LRU the extra bits are unused, so operating on them is a noop. Transfer the reference state first, before the destination folio is marked up to date, as a best effort to retain referenced status. Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-3-9aacbd77d4ca@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Reviewed-by: Baolin Wang Acked-by: David Hildenbrand (Arm) Reviewed-by: Lian Wang Reviewed-by: Barry Song Reviewed-by: Ridong Chen Cc: Axel Rasmussen Cc: Chris Li Cc: Johannes Weiner Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Qi Zheng Cc: Roman Gushchin Cc: Shakeel Butt Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: Zi Yan --- include/linux/mm_inline.h | 20 +++++++++++++++----- mm/migrate.c | 6 +++--- 2 files changed, 18 insertions(+), 8 deletions(-) diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h index 3f4bd5b02b54da..047295ae6e8ac8 100644 --- a/include/linux/mm_inline.h +++ b/include/linux/mm_inline.h @@ -373,11 +373,19 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, return true; } -static inline void folio_migrate_refs(struct folio *new, const struct folio *old) +/** + * folio_migrate_lru_refs - copy the reference state to a new folio + * @new: the destination folio + * @old: the source folio + * + * Transfer the reference state to @new during migration: the MGLRU + * refs count, including PG_referenced, or just PG_referenced for the + * active/inactive LRU. + */ +static inline void folio_migrate_lru_refs(struct folio *new, const struct folio *old) { - unsigned long refs = READ_ONCE(old->flags.f) & LRU_REFS_MASK; - - set_mask_bits(&new->flags.f, LRU_REFS_MASK, refs); + BUILD_BUG_ON(LRU_REFS_MASK & BIT(PG_referenced)); + folio_set_lru_refs(new, folio_lru_refs(old)); } #else /* !CONFIG_LRU_GEN */ @@ -406,8 +414,10 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, return false; } -static inline void folio_migrate_refs(struct folio *new, const struct folio *old) +static inline void folio_migrate_lru_refs(struct folio *new, const struct folio *old) { + if (folio_test_referenced(old)) + folio_set_referenced(new); } #endif /* CONFIG_LRU_GEN */ diff --git a/mm/migrate.c b/mm/migrate.c index 15b45832bcfa7e..a369d0c95c3860 100644 --- a/mm/migrate.c +++ b/mm/migrate.c @@ -776,8 +776,9 @@ void folio_migrate_flags(struct folio *newfolio, struct folio *folio) { int cpupid; - if (folio_test_referenced(folio)) - folio_set_referenced(newfolio); + /* Copy the reference state, including PG_referenced */ + folio_migrate_lru_refs(newfolio, folio); + if (folio_test_uptodate(folio)) folio_mark_uptodate(newfolio); if (folio_test_clear_active(folio)) { @@ -807,7 +808,6 @@ void folio_migrate_flags(struct folio *newfolio, struct folio *folio) if (folio_test_idle(folio)) folio_set_idle(newfolio); - folio_migrate_refs(newfolio, folio); /* * Copy NUMA information to the new page, to prevent over-eager * future migrations of this same page. From 76117ea6582b29d849cf5e3a55c7c1e1a3ce53f0 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Sun, 6 Sep 2026 00:51:09 +0800 Subject: [PATCH 0470/1012] mm/mglru: move max_seq read into walk_update_folio walk_pte_range(), walk_pmd_range_locked(), and lru_gen_look_around() each read lrugen->max_seq to compute the target generation used by walk_update_folio(), then pass it as a parameter. Move the read into walk_update_folio() itself so the callers no longer need to compute or pass the value. The max_seq read now happens once per folio update rather than once per walk range, so folios always get promoted to the current youngest generation. Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-4-9aacbd77d4ca@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Reviewed-by: Baolin Wang Reviewed-by: Ridong Chen Reviewed-by: Lian Wang Reviewed-by: Barry Song Cc: Axel Rasmussen Cc: Chris Li Cc: David Hildenbrand (Arm) Cc: Johannes Weiner Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Qi Zheng Cc: Roman Gushchin Cc: Shakeel Butt Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: Zi Yan --- mm/vmscan.c | 29 ++++++++++++----------------- 1 file changed, 12 insertions(+), 17 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index c6e03a5047aeb9..4732b987aa67b8 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3557,13 +3557,15 @@ static bool suitable_to_scan(int total, int young) } static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struct *vma, - struct folio *folio, int new_gen, bool dirty) + struct lruvec *lruvec, struct folio *folio, bool dirty) { - int old_gen; + int new_gen, old_gen; if (!folio) return; + new_gen = lru_gen_from_seq(READ_ONCE(lruvec->lrugen.max_seq)); + if (dirty && !folio_test_dirty(folio) && !(folio_test_anon(folio) && folio_test_swapbacked(folio) && !folio_test_swapcache(folio))) @@ -3594,8 +3596,6 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end, struct lru_gen_mm_walk *walk = args->private; struct mem_cgroup *memcg = lruvec_memcg(walk->lruvec); struct pglist_data *pgdat = lruvec_pgdat(walk->lruvec); - DEFINE_MAX_SEQ(walk->lruvec); - int gen = lru_gen_from_seq(max_seq); unsigned int nr; pmd_t pmdval; @@ -3646,7 +3646,7 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end, continue; if (last != folio) { - walk_update_folio(walk, args->vma, last, gen, dirty); + walk_update_folio(walk, args->vma, walk->lruvec, last, dirty); last = folio; dirty = false; @@ -3659,7 +3659,7 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end, walk->mm_stats[MM_LEAF_YOUNG] += nr; } - walk_update_folio(walk, args->vma, last, gen, dirty); + walk_update_folio(walk, args->vma, walk->lruvec, last, dirty); last = NULL; if (i < PTRS_PER_PTE && get_next_vma(PMD_MASK, PAGE_SIZE, args, &start, &end)) @@ -3682,8 +3682,6 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area struct lru_gen_mm_walk *walk = args->private; struct mem_cgroup *memcg = lruvec_memcg(walk->lruvec); struct pglist_data *pgdat = lruvec_pgdat(walk->lruvec); - DEFINE_MAX_SEQ(walk->lruvec); - int gen = lru_gen_from_seq(max_seq); VM_WARN_ON_ONCE(pud_leaf(*pud)); @@ -3737,7 +3735,7 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area goto next; if (last != folio) { - walk_update_folio(walk, vma, last, gen, dirty); + walk_update_folio(walk, vma, walk->lruvec, last, dirty); last = folio; dirty = false; @@ -3751,7 +3749,7 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area i = i > MIN_LRU_BATCH ? 0 : find_next_bit(bitmap, MIN_LRU_BATCH, i) + 1; } while (i <= MIN_LRU_BATCH); - walk_update_folio(walk, vma, last, gen, dirty); + walk_update_folio(walk, vma, walk->lruvec, last, dirty); lazy_mmu_mode_disable(); spin_unlock(ptl); @@ -4357,8 +4355,6 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) struct pglist_data *pgdat = folio_pgdat(folio); struct lruvec *lruvec; struct lru_gen_mm_state *mm_state; - unsigned long max_seq; - int gen; lockdep_assert_held(pvmw->ptl); VM_WARN_ON_ONCE_FOLIO(folio_test_lru(folio), folio); @@ -4395,8 +4391,6 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) memcg = get_mem_cgroup_from_folio(folio); lruvec = mem_cgroup_lruvec(memcg, pgdat); - max_seq = READ_ONCE((lruvec)->lrugen.max_seq); - gen = lru_gen_from_seq(max_seq); mm_state = get_mm_state(lruvec); lazy_mmu_mode_enable(); @@ -4428,7 +4422,7 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) continue; if (last != folio) { - walk_update_folio(walk, vma, last, gen, dirty); + walk_update_folio(walk, vma, lruvec, last, dirty); last = folio; dirty = false; @@ -4440,13 +4434,14 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) young += nr; } - walk_update_folio(walk, vma, last, gen, dirty); + walk_update_folio(walk, vma, lruvec, last, dirty); lazy_mmu_mode_disable(); /* feedback from rmap walkers to page table walkers */ if (mm_state && suitable_to_scan(i, young)) - update_bloom_filter(mm_state, max_seq, pvmw->pmd); + update_bloom_filter(mm_state, READ_ONCE(lruvec->lrugen.max_seq), + pvmw->pmd); mem_cgroup_put(memcg); From a3e82eb632bde03f9584f288b0b075c42c2cd7ce Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Sun, 6 Sep 2026 00:51:10 +0800 Subject: [PATCH 0471/1012] mm/mglru: use explicit tier range in read_ctrl_pos() read_ctrl_pos() encodes the tier range in a single "tier" parameter via "tier % MAX_NR_TIERS" as the start and "min(tier, MAX_NR_TIERS-1)" as the end. This is hard to follow, maintain, or extend. Tier values 0..3 select a single tier, while tier == MAX_NR_TIERS selects the full range. Replace it with explicit (tier_min, tier_max) parameters using a closed [tier_min, tier_max] interval, and add LRU_TIER_MIN and LRU_TIER_MAX for the tier bounds. The call sites now become self-documenting: - get_tier_idx: (LRU_TIER_MIN, LRU_TIER_MIN) for the first tier, (tier, tier) for each subsequent tier - get_type_to_scan: (LRU_TIER_MIN, LRU_TIER_MAX) for the full range No functional change. Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-5-9aacbd77d4ca@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Reviewed-by: Baoquan He Reviewed-by: Barry Song Reviewed-by: Ridong Chen Cc: Axel Rasmussen Cc: Chris Li Cc: David Hildenbrand (Arm) Cc: Johannes Weiner Cc: Kairui Song Cc: Liam R. Howlett Cc: Lian Wang Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Qi Zheng Cc: Roman Gushchin Cc: Shakeel Butt Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: Zi Yan --- include/linux/mmzone.h | 2 ++ mm/vmscan.c | 18 ++++++++++-------- 2 files changed, 12 insertions(+), 8 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 593cc92fe47146..acd94cecc0d399 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -495,6 +495,8 @@ enum lruvec_flags { * folio->flags, masked by LRU_REFS_MASK. */ #define MAX_NR_TIERS 4U +#define LRU_TIER_MIN 0U +#define LRU_TIER_MAX (MAX_NR_TIERS - 1) #ifndef __GENERATING_BOUNDS_H diff --git a/mm/vmscan.c b/mm/vmscan.c index 4732b987aa67b8..ec5832b0d8673f 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3223,8 +3223,8 @@ struct ctrl_pos { int gain; }; -static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier, int gain, - struct ctrl_pos *pos) +static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier_min, + int tier_max, int gain, struct ctrl_pos *pos) { int i; struct lru_gen_folio *lrugen = &lruvec->lrugen; @@ -3233,7 +3233,7 @@ static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier, int gain, pos->gain = gain; pos->refaulted = pos->total = 0; - for (i = tier % MAX_NR_TIERS; i <= min(tier, MAX_NR_TIERS - 1); i++) { + for (i = tier_min; i <= tier_max; i++) { pos->refaulted += lrugen->avg_refaulted[type][i] + atomic_long_read(&lrugen->refaulted[hist][type][i]); pos->total += lrugen->avg_total[type][i] + @@ -4879,9 +4879,9 @@ static int get_tier_idx(struct lruvec *lruvec, int type) * This value is chosen because any other tier would have at least twice * as many refaults as the first tier. */ - read_ctrl_pos(lruvec, type, 0, 2, &sp); - for (tier = 1; tier < MAX_NR_TIERS; tier++) { - read_ctrl_pos(lruvec, type, tier, 3, &pv); + read_ctrl_pos(lruvec, type, LRU_TIER_MIN, LRU_TIER_MIN, 2, &sp); + for (tier = LRU_TIER_MIN + 1; tier <= LRU_TIER_MAX; tier++) { + read_ctrl_pos(lruvec, type, tier, tier, 3, &pv); if (!positive_ctrl_err(&sp, &pv)) break; } @@ -4902,8 +4902,10 @@ static int get_type_to_scan(struct lruvec *lruvec, int swappiness) * Compare the sum of all tiers of anon with that of file to determine * which type to scan. */ - read_ctrl_pos(lruvec, LRU_GEN_ANON, MAX_NR_TIERS, swappiness, &sp); - read_ctrl_pos(lruvec, LRU_GEN_FILE, MAX_NR_TIERS, MAX_SWAPPINESS - swappiness, &pv); + read_ctrl_pos(lruvec, LRU_GEN_ANON, LRU_TIER_MIN, LRU_TIER_MAX, + swappiness, &sp); + read_ctrl_pos(lruvec, LRU_GEN_FILE, LRU_TIER_MIN, LRU_TIER_MAX, + MAX_SWAPPINESS - swappiness, &pv); return positive_ctrl_err(&sp, &pv); } From 7a13797ad227e5b2d674038e6b3d05ea7263e345 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Sun, 6 Sep 2026 00:51:11 +0800 Subject: [PATCH 0472/1012] mm/mglru: fix potential generation folio number leak Each generation of MGLRU accounts anon and file folio numbers separately, and the page table walker updates each generation's counters in batch once the walk is done. The walker promotes a folio's generation with a cmpxchg on folio->flags, and update_batch_size() then reads the live flags again to pick the anon/file column to charge. The walk holds neither the lruvec lock nor the folio lock, so the type can flip between the cmpxchg and that read: the lazyfree path clears PG_swapbacked, and reclaim sets it back on a dirty lazyfree folio. The batched delta pair is then recorded in the wrong type column. Nothing reconciles it afterwards, permanently skewing lrugen->nr_pages and the reclaim budgets derived from it. Fix it by capturing the type from the flags snapshot the cmpxchg linearized against: folio_update_gen() returns the type of the state it transitioned from, and update_batch_size() accounts with that. A folio's type only changes while it is off the LRU list, inside a del/add pair under the lruvec lock, with the gen bits cleared in between. The generation and PG_swapbacked sit in the same folio->flags word, so the cmpxchg snapshot captures them together. Let G be the generation that snapshot captured (old_gen) and G' the one it wrote (new_gen); the CAS can land in only three places: - before the del: the folio is anon at G; the batch records anon G -> G', and the del later removes the folio from the anon counters; - between del and add: gen == -1, so folio_update_gen() returns -1 without touching the flags and no batch is recorded; the del/add pair accounts for the move alone; - after the add: the folio is file at the fresh generation the add charged; the batch records file, that gen -> G', matching that charge. Unlike the drift of lazy promotions, which sort_folio() repairs under the lruvec lock, the phantom deltas from before this fix land in a column the folio never occupies again, so nothing ever repairs them. Link: https://lore.kernel.org/20260906-mglru-flags-cleanup-v6-6-9aacbd77d4ca@tencent.com Fixes: bd74fdaea146 ("mm: multi-gen LRU: support page table walks") Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Barry Song Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Chris Li Cc: David Hildenbrand (Arm) Cc: Johannes Weiner Cc: Kairui Song Cc: Liam R. Howlett Cc: Lian Wang Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Qi Zheng Cc: Ridong Chen Cc: Roman Gushchin Cc: Shakeel Butt Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: Zi Yan --- include/linux/mm_inline.h | 11 +++++++++++ mm/vmscan.c | 13 +++++++------ 2 files changed, 18 insertions(+), 6 deletions(-) diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h index 047295ae6e8ac8..ab69b9930893ff 100644 --- a/include/linux/mm_inline.h +++ b/include/linux/mm_inline.h @@ -30,6 +30,17 @@ static inline int folio_is_file_lru(const struct folio *folio) return !folio_test_swapbacked(folio); } +/** + * folio_flags_is_file_lru - Should the folio be on a file LRU or anon LRU? + * @flags: The folio's flags. + * + * Just like folio_is_file_lru but take the folio flags directly instead. + */ +static inline int folio_flags_is_file_lru(const unsigned long *flags) +{ + return !test_bit(PG_swapbacked, flags); +} + static __always_inline void __update_lru_size(struct lruvec *lruvec, enum lru_list lru, enum zone_type zid, long nr_pages) diff --git a/mm/vmscan.c b/mm/vmscan.c index ec5832b0d8673f..40d3f1b48a74cf 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3294,7 +3294,8 @@ static bool positive_ctrl_err(struct ctrl_pos *sp, struct ctrl_pos *pv) ******************************************************************************/ /* promote pages accessed through page tables */ -static int folio_update_gen(struct folio *folio, int new_gen, const vma_flags_t *vma_flags) +static int folio_update_gen(struct folio *folio, int new_gen, int *type, + const vma_flags_t *vma_flags) { unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); int old_gen; @@ -3323,6 +3324,7 @@ static int folio_update_gen(struct folio *folio, int new_gen, const vma_flags_t new_flags |= BIT(PG_workingset); } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); + *type = folio_flags_is_file_lru(&old_flags); return old_gen; } @@ -3368,9 +3370,8 @@ static int folio_inc_gen(struct lruvec *lruvec, struct folio *folio) } static void update_batch_size(struct lru_gen_mm_walk *walk, struct folio *folio, - int old_gen, int new_gen) + int old_gen, int new_gen, int type) { - int type = folio_is_file_lru(folio); int zone = folio_zonenum(folio); int delta = folio_nr_pages(folio); @@ -3559,7 +3560,7 @@ static bool suitable_to_scan(int total, int young) static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struct *vma, struct lruvec *lruvec, struct folio *folio, bool dirty) { - int new_gen, old_gen; + int new_gen, old_gen, type; if (!folio) return; @@ -3572,9 +3573,9 @@ static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struc folio_mark_dirty(folio); if (walk) { - old_gen = folio_update_gen(folio, new_gen, &vma->flags); + old_gen = folio_update_gen(folio, new_gen, &type, &vma->flags); if (old_gen >= 0 && old_gen != new_gen) - update_batch_size(walk, folio, old_gen, new_gen); + update_batch_size(walk, folio, old_gen, new_gen, type); } else if (lru_gen_set_refs(folio, &vma->flags)) { old_gen = folio_lru_gen(folio); if (old_gen >= 0 && old_gen != new_gen) From e5e40390b3c3ce6a1e896059d3bbe1ffc79046c0 Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Wed, 16 Sep 2026 10:34:46 +0200 Subject: [PATCH 0473/1012] mm/vmscan: avoid false-positive -Wuninitialized warning, again I previously worked around a false-postive gcc-16 warning in the get_tier_idx() function, by adding a fake initializer. This happens with the -fsanitize=bounds sanitizer when the compiler creates a specialized variant of isolate_folios(): In function 'get_tier_idx', inlined from 'isolate_folios.constprop' at mm/vmscan.c:4982:9: mm/vmscan.c:4934:9: error: 'sp.refaulted' is used uninitialized [-Werror=uninitialized] 4934 | read_ctrl_pos(lruvec, type, LRU_TIER_MIN, LRU_TIER_MIN, 2, &sp); | ^~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ mm/vmscan.c: In function 'isolate_folios.constprop': mm/vmscan.c:4946:25: note: 'sp.refaulted' was declared here 4946 | struct ctrl_pos sp, pv = {}; | ^~ Adding another "= {}" would solve the problem as well, but to prevent this from happening again after the next code refactoring, try instead to prevent this by forbidding interprocedural optimizations on this function. Link: https://lore.kernel.org/all/20260213123902.3466040-1-arnd@kernel.org/ Link: https://lore.kernel.org/20260916083456.4136132-1-arnd@kernel.org Fixes: 3de705a43a46 ("mm/vmscan: avoid false-positive -Wuninitialized warning") Signed-off-by: Arnd Bergmann Signed-off-by: Andrew Morton Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie Cc: --- mm/vmscan.c | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 40d3f1b48a74cf..4059d130078960 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3223,8 +3223,11 @@ struct ctrl_pos { int gain; }; -static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier_min, - int tier_max, int gain, struct ctrl_pos *pos) +/* + * __noipa works around gcc-16 warning for uninitialized use of pos->refaulted + */ +static void __noipa read_ctrl_pos(struct lruvec *lruvec, int type, int tier_min, + int tier_max, int gain, struct ctrl_pos *pos) { int i; struct lru_gen_folio *lrugen = &lruvec->lrugen; @@ -4873,7 +4876,7 @@ static int scan_folios(unsigned long nr_to_scan, struct lruvec *lruvec, static int get_tier_idx(struct lruvec *lruvec, int type) { int tier; - struct ctrl_pos sp, pv = {}; + struct ctrl_pos sp, pv; /* * To leave a margin for fluctuations, use a larger gain factor (2:3). @@ -4892,7 +4895,7 @@ static int get_tier_idx(struct lruvec *lruvec, int type) static int get_type_to_scan(struct lruvec *lruvec, int swappiness) { - struct ctrl_pos sp, pv = {}; + struct ctrl_pos sp, pv; if (swappiness <= MIN_SWAPPINESS + 1) return LRU_GEN_FILE; From 5108503607db418730050bcb477e5d1a31b3671a Mon Sep 17 00:00:00 2001 From: Luxiao Xu Date: Thu, 10 Sep 2026 11:42:50 +0800 Subject: [PATCH 0474/1012] mm/memory: constrain generic_access_phys() to page boundary This patch addresses an issue in generic_access_phys() where accessing memory across page boundaries in PFNMAP VMAs assumes physical pages are contiguous, which can lead to accessing unintended physical memory or exceeding VMA boundaries. generic_access_phys() improperly validates the memory access range: it only validates the start address using follow_pfnmap_start() and passes PAGE_ALIGN(len + offset) to ioremap_prot(). This poses two problems: 1. In PFNMAP VMAs, consecutive virtual pages are not guaranteed to be physically contiguous, and individual PTEs may have different access permissions or writability. 2. The mapping may cross VMA boundaries if len extends beyond vma->vm_end. Constrain the access in generic_access_phys() to at most the current page boundary (PAGE_SIZE - offset) and map only a single PAGE_SIZE via ioremap_prot(). Since the caller __access_remote_vm() already loops over the requested length and handles partial transfers, it will naturally iterate over the remaining pages. Also add a missing (resource_size_t) cast during PFN re-validation to avoid truncation on 32-bit PAE systems. Link: https://lore.kernel.org/e06e28a46c2a176238f03b5740df0913e57c2861.1788842306.git.rakukuip@gmail.com Fixes: 9cb12d7b4cca ("mm/memory.c: actually remap enough memory") Signed-off-by: Luxiao Xu Signed-off-by: Ren Wei Signed-off-by: Andrew Morton Reported-by: Vega Suggested-by: David Hildenbrand Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Grazvydas Ignotas Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: --- mm/memory.c | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/mm/memory.c b/mm/memory.c index ec63dd6212ac5a..926276d4192026 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -7131,6 +7131,12 @@ int generic_access_phys(struct vm_area_struct *vma, unsigned long addr, bool writable; struct follow_pfnmap_args args = { .vma = vma, .address = addr }; + /* + * Limit access to one page at a time, as that's what follow_pfnmap_start() + * guarantees; expect the caller to retry to read larger ranges. + */ + len = min_t(int, len, PAGE_SIZE - offset); + retry: if (follow_pfnmap_start(&args)) return -EINVAL; @@ -7142,7 +7148,7 @@ int generic_access_phys(struct vm_area_struct *vma, unsigned long addr, if ((write & FOLL_WRITE) && !writable) return -EINVAL; - maddr = ioremap_prot(phys_addr, PAGE_ALIGN(len + offset), prot); + maddr = ioremap_prot(phys_addr, PAGE_SIZE, prot); if (!maddr) return -ENOMEM; @@ -7150,7 +7156,7 @@ int generic_access_phys(struct vm_area_struct *vma, unsigned long addr, goto out_unmap; if ((pgprot_val(prot) != pgprot_val(args.pgprot)) || - (phys_addr != (args.pfn << PAGE_SHIFT)) || + (phys_addr != ((resource_size_t)args.pfn << PAGE_SHIFT)) || (writable != args.writable)) { follow_pfnmap_end(&args); iounmap(maddr); From ff7fe958e174a1e9edc671f727c2ce26c4dd6a67 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 5 Sep 2026 10:40:34 +0200 Subject: [PATCH 0475/1012] docs/mm: ksm: use the renamed ksm structure names The Reference section of ksm.rst asks mm/ksm.c for mm_slot, stable_node and rmap_item. Commit 21fbd59136e0 ("ksm: add the ksm prefix to the names of the ksm private structures") renamed them to ksm_mm_slot, ksm_stable_node and ksm_rmap_item. Since then only ksm_scan is rendered and the three structures are missing from the generated documentation. Use the current names. Link: https://lore.kernel.org/20260905084034.39521-1-kmehltretter@gmail.com Fixes: 21fbd59136e0 ("ksm: add the ksm prefix to the names of the ksm private structures") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Reviewed-by: Randy Dunlap Acked-by: David Hildenbrand (Arm) Assisted-by: LLM --- Documentation/mm/ksm.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Documentation/mm/ksm.rst b/Documentation/mm/ksm.rst index 2b4f72f1f95343..5a80573162def0 100644 --- a/Documentation/mm/ksm.rst +++ b/Documentation/mm/ksm.rst @@ -78,7 +78,7 @@ The frequency of such scans is defined by Reference --------- .. kernel-doc:: mm/ksm.c - :functions: mm_slot ksm_scan stable_node rmap_item + :functions: ksm_mm_slot ksm_scan ksm_stable_node ksm_rmap_item -- Izik Eidus, From a3a98dddd83739152a28fdbea21f214edf6127e4 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 4 Sep 2026 20:05:17 -0700 Subject: [PATCH 0476/1012] memcg: move per-node objcg to the read-mostly fields Patch series "memcg: group struct fields by access pattern". Every so often we get a memcg performance regression caused by nothing more than a field moving. Someone adds a field, removes one, or puts a few behind a config option. The layout shifts, fields with different access patterns land on the same cache line, and a bot reports a regression. Two examples. commit 98c9daf5ae6b ("mm: memcg: guard memcg1-specific members of struct mem_cgroup_per_node") moved lruvec next to lru_zone_size[] and needed commit f59adcf59332 ("mm: memcg: add cacheline padding after lruvec in mem_cgroup_per_node") to fix it. commit c1afbd5de131 ("mm/memcontrol: avoid false sharing between vmstats and events") had to add ____cacheline_aligned_in_smp for the same reason. Each fix was correct but nothing stops the next field addition from undoing it. This series makes the layout a contract the compiler checks, the same way struct net_device does it. Fields are sorted into named cache line groups by access pattern, and memcg_struct_check() verifies at build time that every field sits in its group. A field added in the wrong place now breaks the build instead of quietly costing a few percent. struct mem_cgroup gets three groups: memcg_write_hot written on the charge, reclaim and socket paths memcg_cold only the cgroup control paths touch these memcg_read_mostly set when the memcg is created, then only read Testing ======= The cgroup selftests give identical results with and without the series. For performance, two identical 30 core Xeon machines each ran both kernels, with the boot order swapped between them so that machine and order effects cancel. The useful tests run two workloads at once in one cgroup, because false sharing only shows up when one side reads a field that the other side writes. slab allocs + page faults, slab side +1.2% page faults + memory.stat readers, fault side +1.3% page faults + memory.stat readers, reader side +0.8% everything else no change No test regressed. The gains are small but the point of the series is the build time contract. This patch (of 6): current_obj_cgroup() reads memcg->nodeinfo[nid]->objcg on every accounted allocation. The field sits at the end of struct mem_cgroup_per_node, on the same cache line as lru_zone_size[] and iter. lru_zone_size[] is written on every LRU add and remove, and iter is written on every reclaim iteration. Move objcg next to the other read-mostly pointers at the start of the struct. No functional change. Link: https://lore.kernel.org/20260905030522.1887837-1-shakeel.butt@linux.dev Link: https://lore.kernel.org/20260905030522.1887837-2-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index a03b6e3e547078..cfa1877915d45b 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -94,6 +94,7 @@ struct mem_cgroup_per_node { struct lruvec_stats_percpu __percpu *lruvec_stats_percpu; struct lruvec_stats *lruvec_stats; struct shrinker_info __rcu *shrinker_info; + struct obj_cgroup __rcu *objcg; CACHELINE_PADDING(_pad1_); @@ -104,11 +105,9 @@ struct mem_cgroup_per_node { struct mem_cgroup_reclaim_iter iter; /* - * objcg is wiped out as a part of the objcg repaprenting process. * orig_objcg preserves a pointer (and a reference) to the original - * objcg until the end of live of memcg. + * objcg until the end of life of memcg. */ - struct obj_cgroup __rcu *objcg; struct obj_cgroup *orig_objcg; /* list of inherited objcgs, protected by objcg_lock */ struct list_head objcg_list; From 5daea15f68e551d671f7661b7f2967f6ae18ddf1 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 4 Sep 2026 20:05:18 -0700 Subject: [PATCH 0477/1012] memcg: split mem_cgroup_private_id into two fields The two members of struct mem_cgroup_private_id have different access patterns. The id is read on every eviction and refault through mem_cgroup_private_id(), and is only written when the memcg is created and destroyed. The ref is written on every swap charge and uncharge. Split them into private_id and private_id_ref so a later patch can put them into different cache line groups. A struct member cannot be split across two groups. No functional change. Link: https://lore.kernel.org/20260905030522.1887837-3-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 10 +++------- mm/memcontrol.c | 18 +++++++++--------- 2 files changed, 12 insertions(+), 16 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index cfa1877915d45b..32a959f1890abd 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -66,11 +66,6 @@ struct mem_cgroup_reclaim_cookie { #define MEM_CGROUP_ID_SHIFT 16 -struct mem_cgroup_private_id { - int id; - refcount_t ref; -}; - struct memcg_vmstats_percpu; struct memcg1_events_percpu; struct memcg_vmstats; @@ -189,7 +184,8 @@ struct mem_cgroup { struct cgroup_subsys_state css; /* Private memcg ID. Used to ID objects that outlive the cgroup */ - struct mem_cgroup_private_id id; + int private_id; + refcount_t private_id_ref; /* Accounted resources */ struct page_counter memory; /* Both v1 & v2 */ @@ -810,7 +806,7 @@ static inline unsigned short mem_cgroup_private_id(struct mem_cgroup *memcg) if (mem_cgroup_disabled()) return 0; - return memcg->id.id; + return memcg->private_id; } struct mem_cgroup *mem_cgroup_from_private_id(unsigned short id); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 4cb2db8c0923a4..2f11a3a79a6bc9 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -3773,7 +3773,7 @@ static void memcg_online_kmem(struct mem_cgroup *memcg) static_branch_enable(&memcg_kmem_online_key); - memcg->kmemcg_id = memcg->id.id; + memcg->kmemcg_id = memcg->private_id; } static void memcg_offline_kmem(struct mem_cgroup *memcg) @@ -4032,15 +4032,15 @@ static DEFINE_XARRAY_ALLOC1(mem_cgroup_private_ids); static void mem_cgroup_private_id_remove(struct mem_cgroup *memcg) { - if (memcg->id.id > 0) { - xa_erase(&mem_cgroup_private_ids, memcg->id.id); - memcg->id.id = 0; + if (memcg->private_id > 0) { + xa_erase(&mem_cgroup_private_ids, memcg->private_id); + memcg->private_id = 0; } } static inline void mem_cgroup_private_id_put(struct mem_cgroup *memcg, unsigned int n) { - if (refcount_sub_and_test(n, &memcg->id.ref)) { + if (refcount_sub_and_test(n, &memcg->private_id_ref)) { mem_cgroup_private_id_remove(memcg); /* Memcg ID pins CSS */ @@ -4050,7 +4050,7 @@ static inline void mem_cgroup_private_id_put(struct mem_cgroup *memcg, unsigned struct mem_cgroup *mem_cgroup_private_id_get_online(struct mem_cgroup *memcg, unsigned int n) { - while (!refcount_add_not_zero(n, &memcg->id.ref)) { + while (!refcount_add_not_zero(n, &memcg->private_id_ref)) { /* * The root cgroup cannot be destroyed, so it's refcount must * always be >= 1. @@ -4174,7 +4174,7 @@ static struct mem_cgroup *mem_cgroup_alloc(struct mem_cgroup *parent) if (!memcg) return ERR_PTR(-ENOMEM); - error = xa_alloc(&mem_cgroup_private_ids, &memcg->id.id, NULL, + error = xa_alloc(&mem_cgroup_private_ids, &memcg->private_id, NULL, XA_LIMIT(1, MEM_CGROUP_ID_MAX), GFP_KERNEL); if (error) goto fail; @@ -4320,7 +4320,7 @@ static int mem_cgroup_css_online(struct cgroup_subsys_state *css) lru_gen_online_memcg(memcg); /* Online state pins memcg ID, memcg ID pins CSS */ - refcount_set(&memcg->id.ref, 1); + refcount_set(&memcg->private_id_ref, 1); css_get(css); /* @@ -4333,7 +4333,7 @@ static int mem_cgroup_css_online(struct cgroup_subsys_state *css) * publish it here at the end of onlining. This matches the * regular ID destruction during offlining. */ - xa_store(&mem_cgroup_private_ids, memcg->id.id, memcg, GFP_KERNEL); + xa_store(&mem_cgroup_private_ids, memcg->private_id, memcg, GFP_KERNEL); return 0; free_objcg: From e53922db1ae70d2561f709270e982526f45a15d6 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 4 Sep 2026 20:05:19 -0700 Subject: [PATCH 0478/1012] memcg: group the write-hot fields of struct mem_cgroup These fields are written on the charge, reclaim and socket paths: socket_pressure written by reclaim, read on every socket charge memory_events bumped for this memcg and every ancestor, so a busy child dirties the whole chain memory_events_local vmpressure written on every reclaim iteration private_id_ref written on every swap charge and uncharge kmem_stat high_irq_work, high_work They are spread over the struct today and share cache lines with read-mostly fields. Put them in one cache line group. socket_pressure is kept next to memory_events because mem_cgroup_sk_under_memory_pressure() reads one and bumps the other. Add memcg_struct_check() so the build fails if a field lands outside its group. No functional change. Link: https://lore.kernel.org/20260905030522.1887837-4-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 58 ++++++++++++++++++++++---------------- mm/memcontrol.c | 32 +++++++++++++++++++++ 2 files changed, 66 insertions(+), 24 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 32a959f1890abd..66431981f4f764 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -185,7 +185,6 @@ struct mem_cgroup { /* Private memcg ID. Used to ID objects that outlive the cgroup */ int private_id; - refcount_t private_id_ref; /* Accounted resources */ struct page_counter memory; /* Both v1 & v2 */ @@ -195,14 +194,45 @@ struct mem_cgroup { struct page_counter memsw; /* v1 only */ }; + /* Written on the charge, reclaim and socket paths. */ + __cacheline_group_begin_aligned(memcg_write_hot); + /* + * Hint of reclaim pressure for socket memory management. Note + * that this indicator should NOT be used in legacy cgroup mode + * where socket memory is accounted/charged separately. + */ + u64 socket_pressure; +#if BITS_PER_LONG < 64 + seqlock_t socket_pressure_seqlock; +#endif + /* + * memory.events is bumped for this memcg and all its ancestors, so a + * busy child dirties every ancestor. + */ + atomic_long_t memory_events[MEMCG_NR_MEMORY_EVENTS]; + atomic_long_t memory_events_local[MEMCG_NR_MEMORY_EVENTS]; + + /* vmpressure notifications. Written on every reclaim iteration. */ + struct vmpressure vmpressure; + + /* Written on every swap charge and uncharge. */ + refcount_t private_id_ref; + +#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC + /* MEMCG_KMEM for nmi context */ + atomic_t kmem_stat; +#endif + + /* Range enforcement for interrupt charges */ + struct work_struct high_work; + + __cacheline_group_end_aligned(memcg_write_hot); + /* registered local peak watchers */ struct list_head memory_peaks; struct list_head swap_peaks; spinlock_t peaks_lock; - /* Range enforcement for interrupt charges */ - struct work_struct high_work; - #ifdef CONFIG_ZSWAP unsigned long zswap_max; @@ -213,9 +243,6 @@ struct mem_cgroup { bool zswap_writeback; #endif - /* vmpressure notifications */ - struct vmpressure vmpressure; - /* * Should the OOM killer kill all belonging tasks, had it kill one? */ @@ -231,23 +258,6 @@ struct mem_cgroup { /* memory.stat */ struct memcg_vmstats *vmstats; - /* memory.events */ - atomic_long_t memory_events[MEMCG_NR_MEMORY_EVENTS]; - atomic_long_t memory_events_local[MEMCG_NR_MEMORY_EVENTS]; - -#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC - /* MEMCG_KMEM for nmi context */ - atomic_t kmem_stat; -#endif - /* - * Hint of reclaim pressure for socket memroy management. Note - * that this indicator should NOT be used in legacy cgroup mode - * where socket memory is accounted/charged separately. - */ - u64 socket_pressure; -#if BITS_PER_LONG < 64 - seqlock_t socket_pressure_seqlock; -#endif int kmemcg_id; #ifdef CONFIG_CGROUP_WRITEBACK diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 2f11a3a79a6bc9..56fc6d8c02a86a 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5700,6 +5700,36 @@ __setup("cgroup.memory=", cgroup_memory); * basically everything that doesn't depend on a specific mem_cgroup structure * should be initialized from here. */ +/* + * Fields are grouped by access pattern. Putting a field in the wrong group + * breaks the build here. + */ +static void __init memcg_struct_check(void) +{ + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + socket_pressure); +#if BITS_PER_LONG < 64 + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + socket_pressure_seqlock); +#endif + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + memory_events); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + memory_events_local); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + vmpressure); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + private_id_ref); +#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + kmem_stat); +#endif + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + high_irq_work); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, + high_work); +} + int __init mem_cgroup_init(void) { unsigned int memcg_size; @@ -5713,6 +5743,8 @@ int __init mem_cgroup_init(void) */ BUILD_BUG_ON(MEMCG_CHARGE_BATCH > S32_MAX / PAGE_SIZE); + memcg_struct_check(); + cpuhp_setup_state_nocalls(CPUHP_MM_MEMCQ_DEAD, "mm/memctrl:dead", NULL, memcg_hotplug_cpu_dead); From 318ca8acc8eacfceae2bc64a55b9432766cbc41c Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 4 Sep 2026 20:05:20 -0700 Subject: [PATCH 0479/1012] memcg: group the cold fields of struct mem_cgroup These fields are only touched by the cgroup control paths: memory_peaks, swap_peaks, peaks_lock memory.peak open/read/release events_file, events_local_file, swap_events_file cgroup_file_notify() cgwb_list, cgwb_domain, cgwb_frn writeback setup and the foreign dirty slow path mm_list MGLRU mm list They sit in the middle of the struct today. The three cgroup_file members alone are 192 bytes of notify state next to the vmstats pointer. Put them in one cache line group. No functional change. Link: https://lore.kernel.org/20260905030522.1887837-5-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 49 ++++++++++++++++++++++---------------- mm/memcontrol.c | 25 +++++++++++++++++++ 2 files changed, 53 insertions(+), 21 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 66431981f4f764..3a7f8097130258 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -228,11 +228,37 @@ struct mem_cgroup { __cacheline_group_end_aligned(memcg_write_hot); + /* + * Off the charge and fault paths. Not write free: cgwb_domain is + * written on every writeout completion and mm_list on fork, exit and + * MGLRU aging. They are grouped here so those writes cannot land on + * a line that the fast paths read. + */ + __cacheline_group_begin_aligned(memcg_cold); /* registered local peak watchers */ struct list_head memory_peaks; struct list_head swap_peaks; spinlock_t peaks_lock; + /* memory.events and memory.events.local */ + struct cgroup_file events_file; + struct cgroup_file events_local_file; + + /* handle for "memory.swap.events" */ + struct cgroup_file swap_events_file; + +#ifdef CONFIG_CGROUP_WRITEBACK + struct list_head cgwb_list; + struct wb_domain cgwb_domain; + struct memcg_cgwb_frn cgwb_frn[MEMCG_CGWB_FRN_CNT]; +#endif + +#ifdef CONFIG_LRU_GEN_WALKS_MMU + /* per-memcg mm_struct list */ + struct lru_gen_mm_list mm_list; +#endif + __cacheline_group_end_aligned(memcg_cold); + #ifdef CONFIG_ZSWAP unsigned long zswap_max; @@ -248,37 +274,18 @@ struct mem_cgroup { */ bool oom_group; - /* memory.events and memory.events.local */ - struct cgroup_file events_file; - struct cgroup_file events_local_file; - - /* handle for "memory.swap.events" */ - struct cgroup_file swap_events_file; - /* memory.stat */ struct memcg_vmstats *vmstats; int kmemcg_id; -#ifdef CONFIG_CGROUP_WRITEBACK - struct list_head cgwb_list; -#endif - /* Keep the hot per-CPU stats pointer away from memory event counters. */ struct memcg_vmstats_percpu __percpu *vmstats_percpu ____cacheline_aligned_in_smp; -#ifdef CONFIG_CGROUP_WRITEBACK - struct wb_domain cgwb_domain; - struct memcg_cgwb_frn cgwb_frn[MEMCG_CGWB_FRN_CNT]; -#endif - -#ifdef CONFIG_LRU_GEN_WALKS_MMU - /* per-memcg mm_struct list */ - struct lru_gen_mm_list mm_list; -#endif - #ifdef CONFIG_MEMCG_V1 + /* v1 only. Not grouped: v1 is legacy, sorting it is not worth it. */ + /* Legacy consumer-oriented counters */ struct page_counter kmem; /* v1 only */ struct page_counter tcpmem; /* v1 only */ diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 56fc6d8c02a86a..51539265bfa9d5 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5728,6 +5728,31 @@ static void __init memcg_struct_check(void) high_irq_work); CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, high_work); + + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + memory_peaks); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + swap_peaks); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + peaks_lock); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + events_file); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + events_local_file); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + swap_events_file); +#ifdef CONFIG_CGROUP_WRITEBACK + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + cgwb_list); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + cgwb_domain); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + cgwb_frn); +#endif +#ifdef CONFIG_LRU_GEN_WALKS_MMU + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, + mm_list); +#endif } int __init mem_cgroup_init(void) From 37571f441eb2db2333dac7a53415e0c22b63bc39 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 4 Sep 2026 20:05:21 -0700 Subject: [PATCH 0480/1012] memcg: group the read-mostly fields of struct mem_cgroup These fields are set when the memcg is created and only read after that: vmstats_percpu read on every stat update vmstats zswap_max, zswap_writeback private_id read on every eviction and refault kmemcg_id read on every list_lru lookup oom_group Put them in one cache line group at the end of the struct, right before nodeinfo[]. nodeinfo[] is read-mostly too but it is a flexible array, so it cannot sit inside a group. The group ends without padding so the two share a line. This also drops the ____cacheline_aligned_in_smp on vmstats_percpu added by commit c1afbd5de131 ("mm/memcontrol: avoid false sharing between vmstats and events"). That only aligned the start of the field. cgwb_domain followed it on the same line and is written on every writeout completion. A group boundary covers both sides. No functional change. Link: https://lore.kernel.org/20260905030522.1887837-6-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 62 +++++++++++++++++++++----------------- mm/memcontrol.c | 17 +++++++++++ 2 files changed, 52 insertions(+), 27 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 3a7f8097130258..73594783a2ef0e 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -183,9 +183,6 @@ struct obj_cgroup { struct mem_cgroup { struct cgroup_subsys_state css; - /* Private memcg ID. Used to ID objects that outlive the cgroup */ - int private_id; - /* Accounted resources */ struct page_counter memory; /* Both v1 & v2 */ @@ -259,30 +256,6 @@ struct mem_cgroup { #endif __cacheline_group_end_aligned(memcg_cold); -#ifdef CONFIG_ZSWAP - unsigned long zswap_max; - - /* - * Prevent pages from this memcg from being written back from zswap to - * swap, and from being swapped out on zswap store failures. - */ - bool zswap_writeback; -#endif - - /* - * Should the OOM killer kill all belonging tasks, had it kill one? - */ - bool oom_group; - - /* memory.stat */ - struct memcg_vmstats *vmstats; - - int kmemcg_id; - - /* Keep the hot per-CPU stats pointer away from memory event counters. */ - struct memcg_vmstats_percpu __percpu *vmstats_percpu - ____cacheline_aligned_in_smp; - #ifdef CONFIG_MEMCG_V1 /* v1 only. Not grouped: v1 is legacy, sorting it is not worth it. */ @@ -322,6 +295,41 @@ struct mem_cgroup { int swappiness; #endif /* CONFIG_MEMCG_V1 */ + /* + * Set when the memcg is created and cleared when it is offlined. + * Never written on a hot path. + */ + __cacheline_group_begin_aligned(memcg_read_mostly); + /* Read on every stat update */ + struct memcg_vmstats_percpu __percpu *vmstats_percpu; + + /* memory.stat */ + struct memcg_vmstats *vmstats; + +#ifdef CONFIG_ZSWAP + unsigned long zswap_max; +#endif + + /* Private memcg ID. Used to ID objects that outlive the cgroup */ + int private_id; + + int kmemcg_id; + + /* + * Should the OOM killer kill all belonging tasks, had it kill one? + */ + bool oom_group; + +#ifdef CONFIG_ZSWAP + /* + * Prevent pages from this memcg from being written back from zswap to + * swap, and from being swapped out on zswap store failures. + */ + bool zswap_writeback; +#endif + /* Not padded: nodeinfo[] is read-mostly too, let it share the line. */ + __cacheline_group_end(memcg_read_mostly); + struct mem_cgroup_per_node *nodeinfo[]; }; diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 51539265bfa9d5..a392f06a521344 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5753,6 +5753,23 @@ static void __init memcg_struct_check(void) CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_cold, mm_list); #endif + + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + vmstats_percpu); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + vmstats); +#ifdef CONFIG_ZSWAP + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + zswap_max); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + zswap_writeback); +#endif + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + private_id); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + kmemcg_id); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + oom_group); } int __init mem_cgroup_init(void) From ae6e2d8bfc9d27f39d7ebaee7cb897710caaf696 Mon Sep 17 00:00:00 2001 From: Shakeel Butt Date: Fri, 4 Sep 2026 20:05:22 -0700 Subject: [PATCH 0481/1012] memcg: group the fields of struct mem_cgroup_per_node Replace the two ad-hoc CACHELINE_PADDING members with named cache line groups: memcg_pn_read_mostly memcg, lruvec_stats_percpu, lruvec_stats, shrinker_info, objcg memcg_pn_lruvec lruvec memcg_pn_write_hot lru_zone_size, iter, nmi slab stats memcg_pn_cold orig_objcg, objcg_list The group markers give the same isolation the padding did, but they are named and the build now checks them. lruvec still gets its own lines. Commit f59adcf59332 ("mm: memcg: add cacheline padding after lruvec in mem_cgroup_per_node") showed why that matters: lru_zone_size[] is written under lru_lock but read without it by lruvec_lru_size(), so it must not share a line with lruvec. Splitting the cold fields out costs one extra cache line per node per memcg. No functional change. Link: https://lore.kernel.org/20260905030522.1887837-7-shakeel.butt@linux.dev Signed-off-by: Shakeel Butt Signed-off-by: Andrew Morton Cc: Johannes Weiner Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 32 ++++++++++++++++++++++---------- mm/memcontrol.c | 30 ++++++++++++++++++++++++++++++ 2 files changed, 52 insertions(+), 10 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 73594783a2ef0e..c0c9805b6f0320 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -82,7 +82,8 @@ struct mem_cgroup_reclaim_iter { * per-node information in memory controller. */ struct mem_cgroup_per_node { - /* Keep the read-only fields at the start */ + /* Set when the memcg is created, then only read. */ + __cacheline_group_begin_aligned(memcg_pn_read_mostly); struct mem_cgroup *memcg; /* Back pointer, we cannot */ /* use container_of */ @@ -91,14 +92,30 @@ struct mem_cgroup_per_node { struct shrinker_info __rcu *shrinker_info; struct obj_cgroup __rcu *objcg; - CACHELINE_PADDING(_pad1_); + __cacheline_group_end_aligned(memcg_pn_read_mostly); - /* Fields which get updated often at the end. */ + /* + * Keep lruvec on its own lines. Sharing them with lru_zone_size[] + * regressed, see commit f59adcf59332 ("mm: memcg: add cacheline + * padding after lruvec in mem_cgroup_per_node"). + */ + __cacheline_group_begin_aligned(memcg_pn_lruvec); struct lruvec lruvec; - CACHELINE_PADDING(_pad2_); + __cacheline_group_end_aligned(memcg_pn_lruvec); + + /* Written on every LRU update and on every reclaim iteration. */ + __cacheline_group_begin_aligned(memcg_pn_write_hot); long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS]; struct mem_cgroup_reclaim_iter iter; +#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC + /* slab stats for nmi context */ + atomic_t slab_reclaimable; + atomic_t slab_unreclaimable; +#endif + __cacheline_group_end_aligned(memcg_pn_write_hot); + /* Touched only when the memcg is reparented or freed. */ + __cacheline_group_begin_aligned(memcg_pn_cold); /* * orig_objcg preserves a pointer (and a reference) to the original * objcg until the end of life of memcg. @@ -106,12 +123,7 @@ struct mem_cgroup_per_node { struct obj_cgroup *orig_objcg; /* list of inherited objcgs, protected by objcg_lock */ struct list_head objcg_list; - -#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC - /* slab stats for nmi context */ - atomic_t slab_reclaimable; - atomic_t slab_unreclaimable; -#endif + __cacheline_group_end_aligned(memcg_pn_cold); }; struct mem_cgroup_threshold { diff --git a/mm/memcontrol.c b/mm/memcontrol.c index a392f06a521344..e539881f58f147 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5770,6 +5770,36 @@ static void __init memcg_struct_check(void) kmemcg_id); CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, oom_group); + + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_read_mostly, memcg); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_read_mostly, lruvec_stats_percpu); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_read_mostly, lruvec_stats); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_read_mostly, shrinker_info); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_read_mostly, objcg); + + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_lruvec, lruvec); + + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_write_hot, lru_zone_size); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_write_hot, iter); +#ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_write_hot, slab_reclaimable); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_write_hot, slab_unreclaimable); +#endif + + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_cold, orig_objcg); + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup_per_node, + memcg_pn_cold, objcg_list); } int __init mem_cgroup_init(void) From c26093329ecbf88c2291b323c9e52d46b8d281fb Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Fri, 4 Sep 2026 23:20:47 +0000 Subject: [PATCH 0482/1012] mm/zswap: convert zswap_store_page() and zswap_compress() to take a folio Currently, zswap_store() iterates through a folio and extracts individual struct page pointers using folio_page() to pass into zswap_store_page() and zswap_compress(). Pass the folio and the page index directly into these functions instead. This eliminates the need to materialize struct page pointers inside the zswap_store() loop and replaces legacy page-based accessors with their folio equivalents: - sg_set_page() -> sg_set_folio() - kmap_local_page() -> kmap_local_folio() - page_to_nid() -> folio_nid() - page_swap_entry() -> calculated via folio->swap and index Link: https://lore.kernel.org/20260904232108.3034333-1-wfelipe@google.com Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton Reviewed-by: Tal Zussman Cc: Chengming Zhou Cc: Johannes Weiner Cc: Nhat Pham --- mm/zswap.c | 26 ++++++++++++-------------- 1 file changed, 12 insertions(+), 14 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index f3ae3c81e48eac..b894de1786fd3f 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -824,8 +824,8 @@ static int zswap_cpu_comp_prepare(unsigned int cpu, struct hlist_node *node) return ret; } -static bool zswap_compress(struct page *page, struct zswap_entry *entry, - struct zswap_pool *pool) +static bool zswap_compress(struct folio *folio, long index, + struct zswap_entry *entry, struct zswap_pool *pool) { struct crypto_acomp_ctx *acomp_ctx; struct scatterlist input, output; @@ -841,7 +841,7 @@ static bool zswap_compress(struct page *page, struct zswap_entry *entry, dst = acomp_ctx->buffer; sg_init_table(&input, 1); - sg_set_page(&input, page, PAGE_SIZE, 0); + sg_set_folio(&input, folio, PAGE_SIZE, index * PAGE_SIZE); sg_init_one(&output, dst, PAGE_SIZE); acomp_request_set_params(acomp_ctx->req, &input, &output, PAGE_SIZE, dlen); @@ -870,8 +870,7 @@ static bool zswap_compress(struct page *page, struct zswap_entry *entry, */ if (comp_ret || !dlen || dlen >= PAGE_SIZE) { rcu_read_lock(); - if (!mem_cgroup_zswap_writeback_enabled( - folio_memcg(page_folio(page)))) { + if (!mem_cgroup_zswap_writeback_enabled(folio_memcg(folio))) { rcu_read_unlock(); comp_ret = comp_ret ? comp_ret : -EINVAL; goto unlock; @@ -879,12 +878,12 @@ static bool zswap_compress(struct page *page, struct zswap_entry *entry, rcu_read_unlock(); comp_ret = 0; dlen = PAGE_SIZE; - dst = kmap_local_page(page); + dst = kmap_local_folio(folio, index * PAGE_SIZE); mapped = true; } gfp = GFP_NOWAIT | __GFP_NORETRY | __GFP_HIGHMEM | __GFP_MOVABLE; - handle = zs_malloc(pool->zs_pool, dlen, gfp, page_to_nid(page)); + handle = zs_malloc(pool->zs_pool, dlen, gfp, folio_nid(folio)); if (IS_ERR_VALUE(handle)) { alloc_ret = PTR_ERR((void *)handle); goto unlock; @@ -1392,21 +1391,22 @@ static void shrink_worker(struct work_struct *w) * main API **********************************/ -static bool zswap_store_page(struct page *page, +static bool zswap_store_page(struct folio *folio, long index, struct obj_cgroup *objcg, struct zswap_pool *pool) { - swp_entry_t page_swpentry = page_swap_entry(page); + swp_entry_t page_swpentry = swp_entry(swp_type(folio->swap), + swp_offset(folio->swap) + index); struct zswap_entry *entry, *old; /* allocate entry */ - entry = zswap_entry_cache_alloc(GFP_KERNEL, page_to_nid(page)); + entry = zswap_entry_cache_alloc(GFP_KERNEL, folio_nid(folio)); if (!entry) { zswap_reject_kmemcache_fail++; return false; } - if (!zswap_compress(page, entry, pool)) + if (!zswap_compress(folio, index, entry, pool)) goto compress_failed; old = xa_store(swap_zswap_tree(page_swpentry), @@ -1515,9 +1515,7 @@ bool zswap_store(struct folio *folio) } for (index = 0; index < nr_pages; ++index) { - struct page *page = folio_page(folio, index); - - if (!zswap_store_page(page, objcg, pool)) + if (!zswap_store_page(folio, index, objcg, pool)) goto put_pool; } From 01b0c80d45cb7df9ddbcd3e1710d0bf21febc4f3 Mon Sep 17 00:00:00 2001 From: David Stevens Date: Fri, 4 Sep 2026 10:31:45 -0700 Subject: [PATCH 0483/1012] memcg: don't call schedule_work when no spinning is allowed Memcg charging can be done from any context, but calling schedule_work() isn't safe from an NMI. If memory.high is breached from a context where spinning isn't allowed, use irq_work to schedule the reclaim work. Found this via code inspection. I spent a little bit trying to trigger it for real, but the only way I managed was by writing a hacky driver absuing alloc_pages_nolock(). Link: https://lore.kernel.org/20260904173145.2028377-1-stevensd@google.com Fixes: 3ac4638a734a ("memcg: make memcg_rstat_updated nmi safe") Signed-off-by: David Stevens Signed-off-by: Andrew Morton Acked-by: Michal Hocko Reviewed-by: Johannes Weiner Acked-by: Shakeel Butt Cc: Lorenzo Stoakes Cc: Muchun Song Cc: Roman Gushchin --- include/linux/memcontrol.h | 2 ++ mm/memcontrol.c | 13 ++++++++++++- 2 files changed, 14 insertions(+), 1 deletion(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index c0c9805b6f0320..4f720791a31c95 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -23,6 +23,7 @@ #include #include #include +#include struct mem_cgroup; struct obj_cgroup; @@ -233,6 +234,7 @@ struct mem_cgroup { #endif /* Range enforcement for interrupt charges */ + struct irq_work high_irq_work; struct work_struct high_work; __cacheline_group_end_aligned(memcg_write_hot); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index e539881f58f147..3e14e0ce9e7ec4 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -62,6 +62,7 @@ #include #include #include +#include #include "internal.h" #include "swap.h" #include "swap_table.h" @@ -2424,6 +2425,11 @@ static void high_work_func(struct work_struct *work) reclaim_high(memcg, MEMCG_CHARGE_BATCH, GFP_KERNEL); } +static void high_irq_work_func(struct irq_work *work) +{ + schedule_work(&container_of(work, struct mem_cgroup, high_irq_work)->high_work); +} + /* * Clamp the maximum sleep time per allocation batch to 2 seconds. This is * enough to still cause a significant slowdown in most cases, while still @@ -2832,7 +2838,10 @@ static int try_charge_memcg(struct mem_cgroup *memcg, gfp_t gfp_mask, /* Don't bother a random interrupted task */ if (!in_task()) { if (mem_high) { - schedule_work(&memcg->high_work); + if (allow_spinning) + schedule_work(&memcg->high_work); + else + irq_work_queue(&memcg->high_irq_work); break; } continue; @@ -4207,6 +4216,7 @@ static struct mem_cgroup *mem_cgroup_alloc(struct mem_cgroup *parent) goto fail; INIT_WORK(&memcg->high_work, high_work_func); + init_irq_work(&memcg->high_irq_work, high_irq_work_func); vmpressure_init(&memcg->vmpressure); INIT_LIST_HEAD(&memcg->memory_peaks); INIT_LIST_HEAD(&memcg->swap_peaks); @@ -4415,6 +4425,7 @@ static void mem_cgroup_css_free(struct cgroup_subsys_state *css) static_branch_dec(&memcg_bpf_enabled_key); vmpressure_cleanup(&memcg->vmpressure); + irq_work_sync(&memcg->high_irq_work); cancel_work_sync(&memcg->high_work); free_shrinker_info(memcg); mem_cgroup_free(memcg); From 57dbd48861f4cad2de12085c7ad6a1915aea87a2 Mon Sep 17 00:00:00 2001 From: Joanne Koong Date: Thu, 3 Sep 2026 14:56:16 -0700 Subject: [PATCH 0484/1012] mm/memcontrol: skip non-hierarchical memcg-wide stats when v1 is unavailable memcg_vmstats keeps a non-hierarchical copy of every memcg-wide stat item and event alongside the hierarchical one. The only readers however are the legacy memory.stat and memory.numa_stat, and reparenting on offline. All of them live under CONFIG_MEMCG_V1, and their accessors (memcg_page_state_local() and memcg_events_local()) are already compiled out with it. This means on a CONFIG_MEMCG_V1=n kernel, memcg_vmstats's non-hierarchical arrays are written to on every rstat flush, despite their values never being read / accessed. The same holds when the kernel does support v1 but the controller has been blocked from v1 hierarchies with the boot param cgroup_no_v1={memory,all}. A v1 mount is refused in that case, so the legacy memory.stat can never exist and the arrays are just as unread / unaccessed. Compile out the non-hierarchical memcg-wide arrays if CONFIG_MEMCG_V1 is not set. If it is set but cgroup_no_v1= has blocked the controller, skip updates on the arrays. This makes flushes cheaper. mem_cgroup_stat_aggregate() can now skip the read-modify-write of ac->local[i]. Nothing else accesses state_local or events_local, so those cachelines were getting pulled in solely for the writes and they are separate from the ones the loop is already walking. On an 80-cpu x86_64 machine with 500 cgroups each running a workload that dirties anon, file, dirty/writeback, slab, kmem, mlock, and reclaim counters, timing mem_cgroup_css_rstat_flush() in-kernel in TSC ticks per flush showed roughly before after delta memcg-wide aggregation 1231 1180 -4.1% overall flush function 2452 2397 -2.2% These numbers are from taking the median of 70 samples, one per 20s window on each kernel. The 95% intervals observed on the two deltas are [-5.41%, -1.76%] and [-4.53%, -0.14%]. The values above include the timing overhead itself, so only the delta is meaningful here. Counting the items that actually changed, a median of 1.5 of the 77 memcg-wide items (57 state + 20 events) had a non-zero per-cpu delta at each flush, which means the benchmarks above are with one or two fewer cachelines pulled in per flush. The count is low because the benchmark reads memory.stat in a loop to keep the flush rate up. For cases where flushes are triggered only by the 2s periodic worker, more changes will have accumulated between flushes, so more cachelines are skipped and the per-flush saving should be larger. Link: https://lore.kernel.org/20260903215616.1456239-1-joannelkoong@gmail.com Signed-off-by: Joanne Koong Signed-off-by: Andrew Morton Reviewed-by: Johannes Weiner Acked-by: Shakeel Butt Reviewed-by: Yosry Ahmed Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin --- include/linux/cgroup.h | 3 +++ kernel/cgroup/cgroup-internal.h | 1 - mm/memcontrol.c | 48 ++++++++++++++++++++++++++++----- 3 files changed, 44 insertions(+), 8 deletions(-) diff --git a/include/linux/cgroup.h b/include/linux/cgroup.h index 5dfa915a630e8d..2afb4cb2bb4fa4 100644 --- a/include/linux/cgroup.h +++ b/include/linux/cgroup.h @@ -154,6 +154,9 @@ struct cgroup *cgroup_get_from_path(const char *path); struct cgroup *cgroup_get_from_fd(int fd); struct cgroup *cgroup_v1v2_get_from_fd(int fd); +/* Was this controller blocked from v1 hierarchies by cgroup_no_v1= ? */ +bool cgroup1_ssid_disabled(int ssid); + int cgroup_attach_task_all(struct task_struct *from, struct task_struct *); int cgroup_transfer_tasks(struct cgroup *to, struct cgroup *from); diff --git a/kernel/cgroup/cgroup-internal.h b/kernel/cgroup/cgroup-internal.h index 58797123b752f3..7c367c8d0cbe1d 100644 --- a/kernel/cgroup/cgroup-internal.h +++ b/kernel/cgroup/cgroup-internal.h @@ -285,7 +285,6 @@ extern struct kernfs_syscall_ops cgroup1_kf_syscall_ops; extern const struct fs_parameter_spec cgroup1_fs_parameters[]; int proc_cgroupstats_show(struct seq_file *m, void *v); -bool cgroup1_ssid_disabled(int ssid); void cgroup1_pidlist_destroy_all(struct cgroup *cgrp); void cgroup1_release_agent(struct work_struct *work); void cgroup1_check_for_release(struct cgroup *cgrp); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 3e14e0ce9e7ec4..3e0f6edfd350d7 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -676,9 +676,11 @@ struct memcg_vmstats { long state[MEMCG_VMSTAT_SIZE]; unsigned long events[NR_MEMCG_EVENTS]; +#ifdef CONFIG_MEMCG_V1 /* Non-hierarchical (CPU aggregated) page state & events */ long state_local[MEMCG_VMSTAT_SIZE]; unsigned long events_local[NR_MEMCG_EVENTS]; +#endif /* Pending child counts during tree propagation */ long state_pending[MEMCG_VMSTAT_SIZE]; @@ -688,6 +690,31 @@ struct memcg_vmstats { atomic_long_t stats_updates; }; +/* + * The non-hierarchical memcg-wide counters are read back only by the legacy + * memory.stat and memory.numa_stat, and by reparenting on offline, all of which + * are v1-only. If the kernel is built without CONFIG_MEMCG_V1, or if the boot + * param cgroup_no_v1= has blocked the memory controller from v1 hierarchies, + * then nothing reads them and writers can skip the updates. + */ +static long *memcg_state_local_array(struct mem_cgroup *memcg) +{ +#ifdef CONFIG_MEMCG_V1 + if (!cgroup1_ssid_disabled(memory_cgrp_id)) + return memcg->vmstats->state_local; +#endif + return NULL; +} + +static unsigned long *memcg_events_local_array(struct mem_cgroup *memcg) +{ +#ifdef CONFIG_MEMCG_V1 + if (!cgroup1_ssid_disabled(memory_cgrp_id)) + return memcg->vmstats->events_local; +#endif + return NULL; +} + /* * memcg and lruvec stats flushing * @@ -4469,7 +4496,10 @@ static void mem_cgroup_css_reset(struct cgroup_subsys_state *css) struct aggregate_control { /* pointer to the aggregated (CPU and subtree aggregated) counters */ long *aggregate; - /* pointer to the non-hierarchichal (CPU aggregated) counters */ + /* + * pointer to the non-hierarchical (CPU aggregated) counters or NULL to + * skip updating them (see memcg_state_local_array()) + */ long *local; /* pointer to the pending child counters during tree propagation */ long *pending; @@ -4508,7 +4538,7 @@ static void mem_cgroup_stat_aggregate(struct aggregate_control *ac) } /* Aggregate counts on this level and propagate upwards */ - if (delta_cpu) + if (delta_cpu && ac->local) ac->local[i] += delta_cpu; if (delta) { @@ -4522,6 +4552,7 @@ static void mem_cgroup_stat_aggregate(struct aggregate_control *ac) #ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC static void flush_nmi_stats(struct mem_cgroup *memcg, struct mem_cgroup *parent) { + long *state_local = memcg_state_local_array(memcg); int nid; if (atomic_read(&memcg->kmem_stat)) { @@ -4529,7 +4560,8 @@ static void flush_nmi_stats(struct mem_cgroup *memcg, struct mem_cgroup *parent) int index = memcg_stats_index(MEMCG_KMEM); memcg->vmstats->state[index] += kmem; - memcg->vmstats->state_local[index] += kmem; + if (state_local) + state_local[index] += kmem; if (parent) parent->vmstats->state_pending[index] += kmem; } @@ -4551,7 +4583,8 @@ static void flush_nmi_stats(struct mem_cgroup *memcg, struct mem_cgroup *parent) if (plstats) plstats->state_pending[index] += slab; memcg->vmstats->state[index] += slab; - memcg->vmstats->state_local[index] += slab; + if (state_local) + state_local[index] += slab; if (parent) parent->vmstats->state_pending[index] += slab; } @@ -4564,7 +4597,8 @@ static void flush_nmi_stats(struct mem_cgroup *memcg, struct mem_cgroup *parent) if (plstats) plstats->state_pending[index] += slab; memcg->vmstats->state[index] += slab; - memcg->vmstats->state_local[index] += slab; + if (state_local) + state_local[index] += slab; if (parent) parent->vmstats->state_pending[index] += slab; } @@ -4589,7 +4623,7 @@ static void mem_cgroup_css_rstat_flush(struct cgroup_subsys_state *css, int cpu) ac = (struct aggregate_control) { .aggregate = memcg->vmstats->state, - .local = memcg->vmstats->state_local, + .local = memcg_state_local_array(memcg), .pending = memcg->vmstats->state_pending, .ppending = parent ? parent->vmstats->state_pending : NULL, .cstat = statc->state, @@ -4600,7 +4634,7 @@ static void mem_cgroup_css_rstat_flush(struct cgroup_subsys_state *css, int cpu) ac = (struct aggregate_control) { .aggregate = memcg->vmstats->events, - .local = memcg->vmstats->events_local, + .local = memcg_events_local_array(memcg), .pending = memcg->vmstats->events_pending, .ppending = parent ? parent->vmstats->events_pending : NULL, .cstat = statc->events, From 6ed0933bb4200ab5fdfaf43e4a664b6598cbfa04 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 3 Sep 2026 20:08:39 +0100 Subject: [PATCH 0485/1012] mm/madvise: swap in CoW'd MAP_PRIVATE-file mappings on MADV_WILLNEED Currently MADV_WILLNEED treats file-backed and pure anonymous mappings entirely separately - using POSIX_FADV_WILLNEED (equivalent of a readahead) for the former and a tree walk and swap in to swap cache for the latter. MAP_PRIVATE-file backed mappings straddle the two and currently get treated as if they were purely file-backed, meaning any swapped out private pages remain swapped out. Resolve the issue by explicitly checking for CoW'd MAP_PRIVATE-file backed mappings and performing both walks in this case. Since the logic checks for vma->anon_vma this means un-CoW'd MAP_PRIVATE-file backed mappings retain only the single file walk. Link: https://lore.kernel.org/aprjOxDy3JCPb2oa@gremlin Reported-by: Mike Kaplinskiy Closes: https://lore.kernel.org/all/CABeknB_S2XJSHFgnHdgnN0rjzHhH4oQJs_APq9fvxHztQ_pgiA@mail.gmail.com/ Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Vlastimil Babka (SUSE) Reviewed-by: Pedro Falcato Cc: David Hildenbrand Cc: Jann Horn Cc: Liam R. Howlett --- mm/madvise.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/mm/madvise.c b/mm/madvise.c index 73c2901b9adbf0..963337f93a7a11 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -297,11 +297,12 @@ static long madvise_willneed(struct madvise_behavior *madv_behavior) loff_t offset; #ifdef CONFIG_SWAP - if (!file) { + if (vma_is_cow_mapping(vma) && vma->anon_vma) { walk_page_range_vma(vma, start, end, &swapin_walk_ops, vma); lru_add_drain(); /* Push any new pages onto the LRU now */ - return 0; } + if (!file) + return 0; if (shmem_mapping(file->f_mapping)) { shmem_swapin_range(vma, start, end, file->f_mapping); From 04bc6301237b712c31358ab05e368f4e50e3c6c1 Mon Sep 17 00:00:00 2001 From: Hui Zhu Date: Fri, 11 Sep 2026 16:00:48 +0800 Subject: [PATCH 0486/1012] mm: memcg: redirect stats updates of dying memcgs for all hierarchies Patch series "mm: workingset: fix the shadow node budget under MGLRU", v5. Commit 7404bd37cfbe ("mm: workingset: use lruvec_lru_size() to get the number of lru pages") broke the workingset shadow node budget under MGLRU: lruvec_lru_size() reads mz->lru_zone_size, which MGLRU never maintains, so count_shadow_nodes() sees the evictable LRU lists as empty and the shadow shrinker reclaims eviction tokens almost as fast as they are created, losing thrashing protection. Patch 1 extends the dying-mcg stat redirection (previously cgroup v1 only) to all hierarchies, addressing the reparenting race that motivated 7404bd37cfbe. Patch 2 then switches count_shadow_nodes() back to lruvec_page_state_local(), which both classic LRU and MGLRU maintain. Patch 3 recovers the performance. Patch 1 added an unconditional rcu_read_lock() to the stat update fast path; patch 3 checks memcg_is_dying() first and takes the RCU lock only on the rare dying path. Patch 4 closes an accounting gap that patch 2 makes visible: on cgroup v2, reparent_state_local() never moves the dying memcg's non-hierarchical lruvec stats to the parent, so the parent receives the uncharges without the matching charges and its state_local underflows. Patch 4 reparents those stats, mirroring cgroup v1. Performance testing =================== The test script and the raw results are available at [1]. Environment: 10-vCPU QEMU guest, 8 GiB RAM, cgroup v2; 7 runs per configuration, medians reported. Workloads: w1-anon-churn: single-threaded anon fault/charge loop in a memcg (MADV_DONTNEED + re-fault, no reclaim). Every touch is a real fault with charge and memcg stat updates, so it stresses exactly the fast path patch 1 changes. This is the meaningful signal: it is not reclaim-bound, so the small fast-path overhead is not drowned out. w2-file-churn: file read loop under memory.high pressure (reclaim-bound; the differences below are within run-to-run noise and are shown for completeness only). w3-reparent: reparent accounting sanity check. w1-anon-churn (pages/s): classic LRU MGLRU base 4393028 4377122 patches 1-2 4385996 (-0.2%) 4352887 (-0.6%) patches 1-3 4381832 (-0.3%) 4377053 (+0.0%) w2-file-churn (MB/s): classic LRU MGLRU base 8277 8226 patches 1-2 8226 (-0.6%) 8123 (-1.3%) patches 1-3 8157 (-1.4%) 8294 (+0.8%) w3-reparent passed on all kernels. The small overhead visible with patches 1-2 comes from the redirection added by patch 1; patch 3 brings w1 back to the base level in both LRU configurations. The w2-file-churn differences are within run-to-run noise: that workload is reclaim-bound and too noisy to expose the small fast-path overhead, so w1-anon-churn is the meaningful signal. Patch 4 only touches the memcg offline path and is not exercised by these workloads. This patch (of 4): get_non_dying_memcg_start() redirects the stat updates of a dying memcg to its closest non-dying ancestor, but only on cgroup v1; on cgroup v2 the stats keep being accounted to the dying memcg itself. A later patch in this series restores lruvec_page_state_local() in count_shadow_nodes() to fix the broken workingset shadow node budget under MGLRU. count_shadow_nodes() is the only reader of those non-hierarchical state_locals on cgroup v2: when a memcg is offlined, its pages are reparented to the ancestor but their stat updates keep being accounted to the dying memcg, so count_shadow_nodes() computes a wrong shadow node budget and workingset thrashing protection is lost. This is user visible as premature reclaim of hot page cache and degraded performance under memory pressure. Apply the redirection to all hierarchies to fix this. Offlining is rare, so the added cost on the stat update fast path is limited to an rcu_read_lock() and a css_is_dying() check; the upward walk happens only while a memcg is dying. Link: https://lore.kernel.org/cover.1789096175.git.zhuhui@kylinos.cn Link: https://lore.kernel.org/c1ef4ef6a84cac479e573f4423b734dc8176f7d5.1789096175.git.zhuhui@kylinos.cn Link: https://gist.github.com/teawater/32f373ec41d185d840455eb167321a5a [1] Fixes: 7404bd37cfbe ("mm: workingset: use lruvec_lru_size() to get the number of lru pages") Signed-off-by: Hui Zhu Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Wei Xu Cc: Yuanchu Xie Cc: --- mm/memcontrol.c | 30 +++++------------------------- 1 file changed, 5 insertions(+), 25 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 3e0f6edfd350d7..8a571e80a1030b 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -873,20 +873,14 @@ static long memcg_state_val_in_pages(int idx, long val) return val < 0 ? -res : res; } -#ifdef CONFIG_MEMCG_V1 /* - * Used in mod_memcg_state() and mod_memcg_lruvec_state() to avoid race with - * reparenting of non-hierarchical state_locals. + * Used in mod_memcg_state() and mod_memcg_lruvec_state() to avoid race + * with reparenting of non-hierarchical state_locals. Offlining a + * memcg is rare, so do the redirection for all cgroup hierarchies. */ -static inline struct mem_cgroup *get_non_dying_memcg_start(struct mem_cgroup *memcg, - bool *rcu_locked) +static inline struct mem_cgroup * +get_non_dying_memcg_start(struct mem_cgroup *memcg, bool *rcu_locked) { - /* Rebinding can cause this value to be changed at runtime */ - if (cgroup_subsys_on_dfl(memory_cgrp_subsys)) { - *rcu_locked = false; - return memcg; - } - rcu_read_lock(); *rcu_locked = true; @@ -898,22 +892,8 @@ static inline struct mem_cgroup *get_non_dying_memcg_start(struct mem_cgroup *me static inline void get_non_dying_memcg_end(bool rcu_locked) { - if (!rcu_locked) - return; - rcu_read_unlock(); } -#else -static inline struct mem_cgroup *get_non_dying_memcg_start(struct mem_cgroup *memcg, - bool *rcu_locked) -{ - return memcg; -} - -static inline void get_non_dying_memcg_end(bool rcu_locked) -{ -} -#endif static void __mod_memcg_state(struct mem_cgroup *memcg, enum memcg_stat_item idx, long val) From bfb522ce99952f3328c582564056f08d97d4ae1c Mon Sep 17 00:00:00 2001 From: Hui Zhu Date: Fri, 11 Sep 2026 16:00:49 +0800 Subject: [PATCH 0487/1012] mm: workingset: use lruvec_page_state_local() to count lru pages Commit 7404bd37cfbe ("mm: workingset: use lruvec_lru_size() to get the number of lru pages") switched count_shadow_nodes() to lruvec_lru_size(). With CONFIG_MEMCG enabled, lruvec_lru_size() reads mz->lru_zone_size, which only the classic LRU paths maintain. MGLRU accounts its pages through __update_lru_size(), which skips that array, so with MGLRU on the four evictable LRU lists are always seen as empty. The shadow node budget (pages >> 3) then collapses to slab plus unevictable pages, and the workingset shadow shrinker reclaims eviction tokens almost as fast as they are created, losing thrashing protection. lruvec_page_state_local() reads lruvec_stats->state_local instead, which both classic LRU and MGLRU maintain. Switch back to it. The reparenting race this re-exposes on cgroup v2 is closed by the preceding patch that redirects dying-memcg stat updates for all hierarchies. Link: https://lore.kernel.org/2ed42f96aca124856ea30f774afb55cbe6d8ba58.1789096175.git.zhuhui@kylinos.cn Fixes: 7404bd37cfbe ("mm: workingset: use lruvec_lru_size() to get the number of lru pages") Signed-off-by: Hui Zhu Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Wei Xu Cc: Yuanchu Xie Cc: --- mm/workingset.c | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/mm/workingset.c b/mm/workingset.c index 7ac2b88c80ae56..8412f4840ae35c 100644 --- a/mm/workingset.c +++ b/mm/workingset.c @@ -688,10 +688,9 @@ static unsigned long count_shadow_nodes(struct shrinker *shrinker, mem_cgroup_flush_stats_ratelimited(sc->memcg); lruvec = mem_cgroup_lruvec(sc->memcg, NODE_DATA(sc->nid)); - for (pages = 0, i = 0; i < NR_LRU_LISTS; i++) - pages += lruvec_lru_size(lruvec, i, MAX_NR_ZONES - 1); - + pages += lruvec_page_state_local(lruvec, + NR_LRU_BASE + i); pages += lruvec_page_state_local( lruvec, NR_SLAB_RECLAIMABLE_B) >> PAGE_SHIFT; pages += lruvec_page_state_local( From c21dc516464dcc3328230b654a825dfd5c795768 Mon Sep 17 00:00:00 2001 From: Hui Zhu Date: Fri, 11 Sep 2026 16:00:50 +0800 Subject: [PATCH 0488/1012] mm: memcg: skip the RCU lock when the memcg is not dying get_non_dying_memcg_start() takes rcu_read_lock() on every stat update, but the lock only protects the upward walk to a non-dying ancestor, which happens solely while a memcg is being offlined. The dying check itself reads the CSS_DYING flag of a memcg the caller already holds a reference to, so it is safe without the lock. Check memcg_is_dying() first and return immediately when the memcg is alive, taking the RCU lock only on the rare dying path. On an anon fault/charge churn workload in a memcg this recovers the ~0.6% overhead added by the previous patch (4368077 vs 4343159 pages/s before, back to ~4377000 pages/s after). Link: https://lore.kernel.org/9ffdbdfc96312e3e13cb8f056bfe26649492d949.1789096175.git.zhuhui@kylinos.cn Fixes: 7404bd37cfbe ("mm: workingset: use lruvec_lru_size() to get the number of lru pages") Signed-off-by: Hui Zhu Signed-off-by: Andrew Morton Acked-by: Shakeel Butt Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Wei Xu Cc: Yuanchu Xie Cc: --- mm/memcontrol.c | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 8a571e80a1030b..307203301a2a56 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -881,6 +881,17 @@ static long memcg_state_val_in_pages(int idx, long val) static inline struct mem_cgroup * get_non_dying_memcg_start(struct mem_cgroup *memcg, bool *rcu_locked) { + /* + * Fast path: the caller holds a reference to @memcg, so reading + * its CSS_DYING flag without the RCU lock is safe. The RCU lock + * is only needed to walk up to a non-dying ancestor, which + * happens only while a memcg is actually being offlined. + */ + if (!memcg_is_dying(memcg)) { + *rcu_locked = false; + return memcg; + } + rcu_read_lock(); *rcu_locked = true; @@ -892,6 +903,9 @@ get_non_dying_memcg_start(struct mem_cgroup *memcg, bool *rcu_locked) static inline void get_non_dying_memcg_end(bool rcu_locked) { + if (!rcu_locked) + return; + rcu_read_unlock(); } From 6308ece8f68b091cdcb75777bcbb6e4b42ab42af Mon Sep 17 00:00:00 2001 From: Hui Zhu Date: Fri, 11 Sep 2026 16:00:51 +0800 Subject: [PATCH 0489/1012] mm: memcg: reparent non-hierarchical lruvec stats on cgroup v2 On cgroup v2, reparent_state_local() returns early and never moves the dying memcg's non-hierarchical state_local base counts to its parent. Meanwhile memcg_reparent_objcgs() rewrites objcg->memcg to the parent, so when the reparented folios are freed later, the negative deltas land on the parent's lruvec. The parent therefore receives the uncharges without ever having received the matching charges, and its state_local (NR_LRU_BASE + lru, MEMCG_SOCK, NR_SLAB_RECLAIMABLE_B, NR_SLAB_UNRECLAIMABLE_B) permanently underflows. Since lruvec_page_state_local() clamps negative values to zero, the underflow masks the parent's own legitimate pages. count_shadow_nodes() is the only reader of these non-hierarchical state_locals on cgroup v2, so the underflow directly distorts the workingset shadow node budget. Fix this by reparenting the lruvec state_locals on cgroup v2 as well, mirroring what cgroup v1 already does. Only the lruvec stats consumed by count_shadow_nodes() are moved; the memcg-level stats are left alone because on v2 they are exposed through the rstat hierarchical tree and are not read from state_local. Link: https://lore.kernel.org/4a7a64eed2b145ad535fedaea3624f8310c29d5b.1789096175.git.zhuhui@kylinos.cn Fixes: 8285917d6f38 ("mm: memcontrol: prepare for reparenting non-hierarchical stats") Signed-off-by: Hui Zhu Signed-off-by: Andrew Morton Cc: Axel Rasmussen Cc: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie Cc: --- mm/memcontrol-v1.h | 5 +++-- mm/memcontrol.c | 42 ++++++++++++++++++++++++++++-------------- 2 files changed, 31 insertions(+), 16 deletions(-) diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index b9a21f0fd2c3ac..2cd37e1792d79e 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -25,6 +25,9 @@ int memory_stat_show(struct seq_file *m, void *v); struct mem_cgroup *mem_cgroup_private_id_get_online(struct mem_cgroup *memcg, unsigned int n); +void reparent_memcg_lruvec_state_local(struct mem_cgroup *memcg, + struct mem_cgroup *parent, int idx); + /* Cgroup v1-specific declarations */ #ifdef CONFIG_MEMCG_V1 @@ -67,8 +70,6 @@ void reparent_memcg1_lruvec_state_local(struct mem_cgroup *memcg, struct mem_cgr void reparent_memcg_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent, int idx); -void reparent_memcg_lruvec_state_local(struct mem_cgroup *memcg, - struct mem_cgroup *parent, int idx); void memcg1_account_kmem(struct mem_cgroup *memcg, int nr_pages); static inline bool memcg1_tcpmem_active(struct mem_cgroup *memcg) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 307203301a2a56..cf53d4ac7ecc18 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -233,14 +233,29 @@ static inline struct obj_cgroup *__memcg_reparent_objcgs(struct mem_cgroup *memc return objcg; } -#ifdef CONFIG_MEMCG_V1 static void __mem_cgroup_flush_stats(struct mem_cgroup *memcg, bool force); -static inline void reparent_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent) +/* + * Reparent the non-hierarchical lruvec stats that count_shadow_nodes() reads + * to approximate the shadow node budget. They are not exposed to userspace + * on cgroup v2, but they must follow the reparented folios; otherwise the + * ancestor would only receive the negative deltas when the folios are freed + * without ever having received the positive base, and its local stats would + * permanently underflow. + */ +static void reparent_v2_lruvec_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent) { - if (cgroup_subsys_on_dfl(memory_cgrp_subsys)) - return; + int i; + + for (i = 0; i < NR_LRU_LISTS; i++) + reparent_memcg_lruvec_state_local(memcg, parent, NR_LRU_BASE + i); + + reparent_memcg_lruvec_state_local(memcg, parent, NR_SLAB_RECLAIMABLE_B); + reparent_memcg_lruvec_state_local(memcg, parent, NR_SLAB_UNRECLAIMABLE_B); +} +static inline void reparent_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent) +{ /* * Reparent stats exposed non-hierarchically. Flush @memcg's stats first * to read its stats accurately , and conservatively flush @parent's @@ -249,17 +264,18 @@ static inline void reparent_state_local(struct mem_cgroup *memcg, struct mem_cgr */ __mem_cgroup_flush_stats(memcg, true); - /* The following counts are all non-hierarchical and need to be reparented. */ - reparent_memcg1_state_local(memcg, parent); - reparent_memcg1_lruvec_state_local(memcg, parent); + if (cgroup_subsys_on_dfl(memory_cgrp_subsys)) { + reparent_v2_lruvec_state_local(memcg, parent); + } else { +#ifdef CONFIG_MEMCG_V1 + /* The following counts are all non-hierarchical and need to be reparented. */ + reparent_memcg1_state_local(memcg, parent); + reparent_memcg1_lruvec_state_local(memcg, parent); +#endif + } __mem_cgroup_flush_stats(parent, true); } -#else -static inline void reparent_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent) -{ -} -#endif static inline void reparent_locks(struct mem_cgroup *memcg, struct mem_cgroup *parent, int nid) { @@ -571,7 +587,6 @@ unsigned long lruvec_page_state_local(struct lruvec *lruvec, return x; } -#ifdef CONFIG_MEMCG_V1 static void __mod_memcg_lruvec_state(struct mem_cgroup_per_node *pn, enum node_stat_item idx, long val); @@ -593,7 +608,6 @@ void reparent_memcg_lruvec_state_local(struct mem_cgroup *memcg, __mod_memcg_lruvec_state(parent_pn, idx, value); } } -#endif /* Subset of vm_event_item to report for memcg event stats */ static const unsigned int memcg_vm_event_stat[] = { From ea0986dca3ada99c2a4bdfaccf0725f949f5ec70 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 18:49:58 +0300 Subject: [PATCH 0490/1012] mm/execmem: free ROX cache chunks only when they span an entire vm area Patch series "mm/execmem: fixes and cleanups for the ROX cache". Sashiko review of the ROX cache refill path asked what happens when the PMD_SIZE allocation fails there. Pulling that thread uncovered a bit more than the fallback. Patch 1 fixes a real bug, although rare bug: execmem_cache_clean() can free a chunk that is still partially in use, because a PMD sized and PMD aligned free range in the middle of a larger chunk looks exactly like a chunk that nobody uses. Patch 2 handles maple tree allocation failures in the cache. They are unlikely, but silently dropping an area from the tree is not a great way to deal with them. Patch 3 deals with the fallback that started all this. The cache exists to keep the direct map free of unnecessary splits, and the fallback happily filled it with base page mapped areas that could never be freed from the cache again. Ask vmalloc for a huge mapping or nothing, and serve whatever does not fit outside the cache. Patches 4 and 5 convert the ROX cache to scope based cleanup. The series was tested on x86 with module load/unload cycles, with PTDUMP confirming that the cache is using 2M mappings, and with nohugevmalloc to exercise the uncached fallback. This patch (of 5): When execmem refills the ROX cache, it vmalloc()s multiples of PMD_SIZE aligned to PMD_SIZE. For every such allocation vmalloc creates a vm area. The first part of the vmalloc()ed chunk is returned to the allocation that triggered the cache refill and the remaining part is added to the cache and handed out for subsequent allocations with execmem_alloc(). When only the first part is freed, the entire vm area remains in the ROX cache and can be handed out again. In the case when the first allocation is larger than PMD_SIZE and the second allocation from the freed first part of the chunk is exactly PMD_SIZE, execmem_cache_clean() will free the entire chunk while part of it is still in use. For example: /* * vmalloc(4M), return p0 to the caller * add [p0 + 3M, p0 + 4M) to the cache */ p0 = execmem_alloc(3M); /* return p0 + 3M from the cache to the caller */ p1 = execmem_alloc(1M); /* put [p0, p0 + 3M) back into the cache */ execmem_free(p0); /* return p0 from the cache to the caller */ p2 = execmem_alloc(2M); /* return p0 + 2M from the cache to the caller */ p3 = execmem_alloc(1M); /* bah! execmem_cache_clean() frees the entire 4M chunk */ execmem_free(p2); Make sure that the ranges that execmem_cache_clean() frees cover the entire vm area. Link: https://lore.kernel.org/20260903-execmem-rox-cache-pmd-v1-v1-0-11beb2a3d249@kernel.org Link: https://lore.kernel.org/20260903-execmem-rox-cache-pmd-v1-v1-1-11beb2a3d249@kernel.org Fixes: 2e45474ab14f ("execmem: add support for cache of large ROX pages") Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Assisted-by: copilot:claude-opus-5 Cc: Benjamin Tissoires Cc: Jiri Kosina Cc: Luis Chamberalin Cc: "Uladzislau Rezki (Sony)" Cc: --- mm/execmem.c | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/mm/execmem.c b/mm/execmem.c index ad07cae9ed5854..ba277790e31328 100644 --- a/mm/execmem.c +++ b/mm/execmem.c @@ -143,9 +143,11 @@ static void execmem_cache_clean(struct work_struct *work) mutex_lock(mutex); mas_for_each(&mas, area, ULONG_MAX) { + struct vm_struct *vm = find_vm_area(area); size_t size = mas_range_len(&mas); - if (IS_ALIGNED(size, PMD_SIZE) && + if (vm && get_vm_area_size(vm) == size && + IS_ALIGNED(size, PMD_SIZE) && IS_ALIGNED(mas.index, PMD_SIZE)) { mas_store_gfp(&mas, NULL, GFP_KERNEL); vfree(area); From 325587f0b051d32212bba795a5a01815259b9290 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 18:49:59 +0300 Subject: [PATCH 0491/1012] mm/execmem: handle potential allocation errors in the maple tree execmem_cache_clean() and execmem_cache_alloc_locked() ignore potential allocation failures in mas_store_gfp(). While in practice they are unlikely to happen, it's better to handle those errors and ensure the integrity of the ROX cache. Preallocate the maple tree nodes for the stores that must not fail and order the maple tree updates so that there won't be any failures once a tree has been modified. Link: https://lore.kernel.org/20260903-execmem-rox-cache-pmd-v1-v1-2-11beb2a3d249@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Assisted-by: copilot:claude-opus-5 Cc: Benjamin Tissoires Cc: Jiri Kosina Cc: Luis Chamberalin Cc: "Uladzislau Rezki (Sony)" --- mm/execmem.c | 32 ++++++++++++++++++++++---------- 1 file changed, 22 insertions(+), 10 deletions(-) diff --git a/mm/execmem.c b/mm/execmem.c index ba277790e31328..00dd6324cae01a 100644 --- a/mm/execmem.c +++ b/mm/execmem.c @@ -149,7 +149,15 @@ static void execmem_cache_clean(struct work_struct *work) if (vm && get_vm_area_size(vm) == size && IS_ALIGNED(size, PMD_SIZE) && IS_ALIGNED(mas.index, PMD_SIZE)) { - mas_store_gfp(&mas, NULL, GFP_KERNEL); + /* + * Preallocate to ensure mas_store does not fail + * If there is no memory for the tree update, bail out, + * next execmem_free() might be more lucky + */ + if (mas_preallocate(&mas, NULL, GFP_KERNEL)) + break; + + mas_store_prealloc(&mas, NULL); vfree(area); } } @@ -219,30 +227,34 @@ static void *execmem_cache_alloc_locked(struct execmem_range *range, size_t size addr = mas_free.index; last = mas_free.last; + mas_set_range(&mas_free, addr, addr + size - 1); + if (mas_preallocate(&mas_free, NULL, GFP_KERNEL)) + return NULL; + /* insert allocated size to busy_areas at range [addr, addr + size) */ mas_set_range(&mas_busy, addr, addr + size - 1); err = mas_store_gfp(&mas_busy, (void *)addr, GFP_KERNEL); if (err) - return NULL; + goto err_destroy_mas_free; - mas_store_gfp(&mas_free, NULL, GFP_KERNEL); + mas_store_prealloc(&mas_free, NULL); if (area_size > size) { - void *ptr = (void *)(addr + size); - /* * re-insert remaining free size to free_areas at range * [addr + size, last] + * the range matches an existing entry, so this cannot allocate */ + ptr = (void *)(addr + size); mas_set_range(&mas_free, addr + size, last); - err = mas_store_gfp(&mas_free, ptr, GFP_KERNEL); - if (err) { - mas_store_gfp(&mas_busy, NULL, GFP_KERNEL); - return NULL; - } + mas_store_gfp(&mas_free, ptr, GFP_KERNEL); } ptr = (void *)addr; return ptr; + +err_destroy_mas_free: + mas_destroy(&mas_free); + return NULL; } static void *__execmem_cache_alloc(struct execmem_range *range, size_t size) From 5e41d4d39e9bdd843edaaf56f01a6ade5b30d8a5 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 18:50:00 +0300 Subject: [PATCH 0492/1012] mm/execmem: make sure ROX cache always contains multiples of PMD_SIZE The ROX cache relies on its chunks being PMD mapped. For a PMD mapped chunk, set_memory_rox() updates the direct map alias one PMD at a time and the large mappings there survive. When execmem refills the cache, it rounds up the requested allocation size to PMD_SIZE and tries to allocate that with vmalloc(VM_ALLOW_HUGE_VMAP). If that allocation fails, execmem falls back to vmalloc() of the original size. There are two issues with this approach: * If huge pages are not available, __vmalloc_node_range() silently falls back to base pages. The area execmem gets is virtually contiguous, but it is backed by 512 scattered base pages. Permission updates on such areas split large mappings in the direct map that contain those base pages, up to 512 PMD splits in the worst case. * execmem's own fallback adds a small base page mapped area to the cache. This adds the overhead of cache management to these allocations with no benefit of reducing fragmentation either in the vmalloc/modules address space or in the direct map. Worse, these areas are never freed from the cache, because execmem_cache_clean() releases only chunks that are a multiple of PMD_SIZE and aligned to PMD_SIZE, exactly to minimize the number of base page mappings. Add VM_REQUIRE_HUGE_VMAP option to vmalloc that fails if the allocation of huge pages fails or if such an allocation is not possible because huge page allocations in vmalloc were disabled or the architecture does not support them. Use this option when populating the ROX cache. If vmalloc(VM_REQUIRE_HUGE_VMAP) fails or vmalloc of huge pages is unavailable, handle the memory allocation outside the ROX cache with plain vmalloc(). Since the fallback allocation has to return ROX memory, add an execmem_alloc_rox() helper and use it for both populating the ROX cache and dealing with a fallback allocation in a ROX execmem_range. With that, the cache only ever contains PMD aligned chunks sized as a multiple of PMD_SIZE, and the PMD checks in execmem_cache_clean() become a VM_WARN_ON_ONCE() to ensure that the PMD mapping invariant does not change. Link: https://lore.kernel.org/20260903-execmem-rox-cache-pmd-v1-v1-3-11beb2a3d249@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Assisted-by: copilot:claude-opus-5 Cc: Benjamin Tissoires Cc: Jiri Kosina Cc: Luis Chamberalin Cc: "Uladzislau Rezki (Sony)" --- include/linux/vmalloc.h | 1 + mm/execmem.c | 76 +++++++++++++++++++++++++---------------- mm/vmalloc.c | 15 +++++++- 3 files changed, 62 insertions(+), 30 deletions(-) diff --git a/include/linux/vmalloc.h b/include/linux/vmalloc.h index aed121d729b013..6e555e31e62251 100644 --- a/include/linux/vmalloc.h +++ b/include/linux/vmalloc.h @@ -38,6 +38,7 @@ struct iov_iter; /* in uio.h */ #define VM_DEFER_KMEMLEAK 0 #endif #define VM_SPARSE 0x00001000 /* sparse vm_area. not all pages are present. */ +#define VM_REQUIRE_HUGE_VMAP 0x00002000 /* huge page mapping or nothing */ /* bits [20..32] reserved for arch specific ioremap internals */ diff --git a/mm/execmem.c b/mm/execmem.c index 00dd6324cae01a..77653b7f163dcb 100644 --- a/mm/execmem.c +++ b/mm/execmem.c @@ -51,7 +51,8 @@ static void *execmem_vmalloc(struct execmem_range *range, size_t size, } if (!p) { - pr_warn_ratelimited("unable to allocate memory\n"); + if (!(vm_flags & VM_REQUIRE_HUGE_VMAP)) + pr_warn_ratelimited("unable to allocate memory\n"); return NULL; } @@ -146,9 +147,10 @@ static void execmem_cache_clean(struct work_struct *work) struct vm_struct *vm = find_vm_area(area); size_t size = mas_range_len(&mas); - if (vm && get_vm_area_size(vm) == size && - IS_ALIGNED(size, PMD_SIZE) && - IS_ALIGNED(mas.index, PMD_SIZE)) { + if (vm && get_vm_area_size(vm) == size) { + VM_WARN_ON_ONCE(!IS_ALIGNED(mas.index, PMD_SIZE) || + !IS_ALIGNED(size, PMD_SIZE)); + /* * Preallocate to ensure mas_store does not fail * If there is no memory for the tree update, bail out, @@ -264,38 +266,41 @@ static void *__execmem_cache_alloc(struct execmem_range *range, size_t size) return execmem_cache_alloc_locked(range, size); } -static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t size) +static void *execmem_vmalloc_rox(struct execmem_range *range, size_t size, + unsigned long vm_flags) { - unsigned long vm_flags = VM_ALLOW_HUGE_VMAP; - struct mutex *mutex = &execmem_cache.mutex; - struct vm_struct *vm; - size_t alloc_size; - int err = -ENOMEM; - void *p; - - alloc_size = round_up(size, PMD_SIZE); - p = execmem_vmalloc(range, alloc_size, PAGE_KERNEL, vm_flags); - if (!p) { - alloc_size = size; - p = execmem_vmalloc(range, alloc_size, PAGE_KERNEL, vm_flags); - } + void *p = execmem_vmalloc(range, size, PAGE_KERNEL, vm_flags); + int err; if (!p) return NULL; - vm = find_vm_area(p); - if (!vm) - goto err_free_mem; - /* fill memory with instructions that will trap */ - execmem_fill_trapping_insns(p, alloc_size); - + execmem_fill_trapping_insns(p, size); set_vm_flush_reset_perms(p); - - err = set_memory_rox((unsigned long)p, vm->nr_pages); + err = set_memory_rox((unsigned long)p, size >> PAGE_SHIFT); if (err) goto err_free_mem; + return p; + +err_free_mem: + vfree(p); + return NULL; +} + +static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t size) +{ + unsigned long vm_flags = VM_REQUIRE_HUGE_VMAP; + size_t alloc_size = round_up(size, PMD_SIZE); + struct mutex *mutex = &execmem_cache.mutex; + int err; + void *p; + + p = execmem_vmalloc_rox(range, alloc_size, vm_flags); + if (!p) + return NULL; + /* * New memory blocks must be allocated and added to the cache * as an atomic operation, otherwise they may be consumed @@ -317,6 +322,11 @@ static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t si return NULL; } +static void *execmem_alloc_rox(struct execmem_range *range, size_t size) +{ + return execmem_vmalloc_rox(range, size, 0); +} + static void *execmem_cache_alloc(struct execmem_range *range, size_t size) { void *p; @@ -444,6 +454,11 @@ static void *execmem_cache_alloc(struct execmem_range *range, size_t size) return NULL; } +static void *execmem_alloc_rox(struct execmem_range *range, size_t size) +{ + return NULL; +} + static bool execmem_cache_free(void *ptr) { return false; @@ -453,17 +468,20 @@ static bool execmem_cache_free(void *ptr) void *execmem_alloc(enum execmem_type type, size_t size) { struct execmem_range *range = &execmem_info->ranges[type]; - bool use_cache = range->flags & EXECMEM_ROX_CACHE; + bool use_rox_cache = range->flags & EXECMEM_ROX_CACHE; unsigned long vm_flags = VM_FLUSH_RESET_PERMS; pgprot_t pgprot = range->pgprot; void *p = NULL; size = PAGE_ALIGN(size); - if (use_cache) + if (use_rox_cache) { p = execmem_cache_alloc(range, size); - else + if (!p) + p = execmem_alloc_rox(range, size); + } else { p = execmem_vmalloc(range, size, pgprot, vm_flags); + } return kasan_reset_tag(p); } diff --git a/mm/vmalloc.c b/mm/vmalloc.c index db357a9bdd1250..aed70e4f8e4e4e 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -4034,6 +4034,12 @@ static gfp_t vmalloc_fix_flags(gfp_t flags) * %__GFP_SKIP_KASAN can be used to skip unpoisoning of mapped pages * (when prot=%PAGE_KERNEL). * + * %VM_ALLOW_HUGE_VMAP allocates huge pages when possible and falls back to + * base pages if huge page allocation fails. + * + * %VM_REQUIRE_HUGE_VMAP implies %VM_ALLOW_HUGE_VMAP and fails instead of + * silently falling back to base pages. + * * Can not be called from interrupt nor NMI contexts. * Return: the address of the area or %NULL on failure */ @@ -4059,6 +4065,10 @@ void *__vmalloc_node_range_noprof(unsigned long size, unsigned long align, return NULL; } + /* VM_REQUIRE_HUGE_VMAP implies VM_ALLOW_HUGE_VMAP */ + if (vm_flags & VM_REQUIRE_HUGE_VMAP) + vm_flags |= VM_ALLOW_HUGE_VMAP; + if (vmap_allow_huge && (vm_flags & VM_ALLOW_HUGE_VMAP)) { /* * Try huge pages. Only try for PAGE_KERNEL allocations, @@ -4075,6 +4085,9 @@ void *__vmalloc_node_range_noprof(unsigned long size, unsigned long align, align = max(original_align, 1UL << shift); } + if ((vm_flags & VM_REQUIRE_HUGE_VMAP) && shift == PAGE_SHIFT) + return NULL; + again: area = __get_vm_area_node(size, align, shift, VM_ALLOC | VM_UNINITIALIZED | vm_flags, start, end, node, @@ -4149,7 +4162,7 @@ void *__vmalloc_node_range_noprof(unsigned long size, unsigned long align, return area->addr; fail: - if (shift > PAGE_SHIFT) { + if (shift > PAGE_SHIFT && !(vm_flags & VM_REQUIRE_HUGE_VMAP)) { shift = PAGE_SHIFT; align = original_align; goto again; From 29f1f02f2f68835cf1376ab4b5342a552b54d097 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 18:50:01 +0300 Subject: [PATCH 0493/1012] mm/vmalloc: add DEFINE_FREE() for vfree() ... and use it in hid-core, the only place with a cleanup for a vmalloc() allocation. Link: https://lore.kernel.org/20260903-execmem-rox-cache-pmd-v1-v1-4-11beb2a3d249@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Cc: Benjamin Tissoires Cc: Jiri Kosina Cc: Luis Chamberalin Cc: "Uladzislau Rezki (Sony)" --- drivers/hid/hid-core.c | 4 ++-- include/linux/vmalloc.h | 3 +++ 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/drivers/hid/hid-core.c b/drivers/hid/hid-core.c index a3ff0514f9cdf8..ec7c2860c93efb 100644 --- a/drivers/hid/hid-core.c +++ b/drivers/hid/hid-core.c @@ -944,7 +944,7 @@ static int hid_scan_report(struct hid_device *hid) hid_parser_reserved }; - struct hid_parser *parser __free(kvfree) = vzalloc(sizeof(*parser)); + struct hid_parser *parser __free(vfree) = vzalloc(sizeof(*parser)); if (!parser) return -ENOMEM; @@ -1265,7 +1265,7 @@ static int hid_parse_collections(struct hid_device *device) hid_parser_reserved }; - struct hid_parser *parser __free(kvfree) = vzalloc(sizeof(*parser)); + struct hid_parser *parser __free(vfree) = vzalloc(sizeof(*parser)); if (!parser) return -ENOMEM; diff --git a/include/linux/vmalloc.h b/include/linux/vmalloc.h index 6e555e31e62251..034a693777ca05 100644 --- a/include/linux/vmalloc.h +++ b/include/linux/vmalloc.h @@ -3,6 +3,7 @@ #define _LINUX_VMALLOC_H #include +#include #include #include #include @@ -215,6 +216,8 @@ void *__must_check vrealloc_node_align_noprof(const void *p, size_t size, extern void vfree(const void *addr); extern void vfree_atomic(const void *addr); +DEFINE_FREE(vfree, void *, if (!IS_ERR_OR_NULL(_T)) vfree(_T)) + extern void *vmap(struct page **pages, unsigned int count, unsigned long flags, pgprot_t prot); void *vmap_pfn(unsigned long *pfns, unsigned int count, pgprot_t prot); From 14b9a10476e16e7b60db1fabb4e1b8e298e37dfe Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Thu, 3 Sep 2026 18:50:02 +0300 Subject: [PATCH 0494/1012] mm/execmem: use cleanup infrastructure in ROX cache functions After splitting out execmem_alloc_rox() from execmem_cache_populate_alloc(), the error paths of both functions became less complex and can be easily switched to use the cleanup infrastructure. Use __free(vfree) to free allocated memory on the error paths and guard(mutex) for synchronization in ROX cache functions. Link: https://lore.kernel.org/20260903-execmem-rox-cache-pmd-v1-v1-5-11beb2a3d249@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Cc: Benjamin Tissoires Cc: Jiri Kosina Cc: Luis Chamberalin Cc: "Uladzislau Rezki (Sony)" --- mm/execmem.c | 32 ++++++++++---------------------- 1 file changed, 10 insertions(+), 22 deletions(-) diff --git a/mm/execmem.c b/mm/execmem.c index 77653b7f163dcb..349cadd874863a 100644 --- a/mm/execmem.c +++ b/mm/execmem.c @@ -138,11 +138,10 @@ int execmem_restore_rox(void *ptr, size_t size) static void execmem_cache_clean(struct work_struct *work) { struct maple_tree *free_areas = &execmem_cache.free_areas; - struct mutex *mutex = &execmem_cache.mutex; MA_STATE(mas, free_areas, 0, ULONG_MAX); void *area; - mutex_lock(mutex); + guard(mutex)(&execmem_cache.mutex); mas_for_each(&mas, area, ULONG_MAX) { struct vm_struct *vm = find_vm_area(area); size_t size = mas_range_len(&mas); @@ -163,7 +162,6 @@ static void execmem_cache_clean(struct work_struct *work) vfree(area); } } - mutex_unlock(mutex); } static DECLARE_WORK(execmem_cache_clean_work, execmem_cache_clean); @@ -269,7 +267,7 @@ static void *__execmem_cache_alloc(struct execmem_range *range, size_t size) static void *execmem_vmalloc_rox(struct execmem_range *range, size_t size, unsigned long vm_flags) { - void *p = execmem_vmalloc(range, size, PAGE_KERNEL, vm_flags); + void *p __free(vfree) = execmem_vmalloc(range, size, PAGE_KERNEL, vm_flags); int err; if (!p) @@ -280,22 +278,17 @@ static void *execmem_vmalloc_rox(struct execmem_range *range, size_t size, set_vm_flush_reset_perms(p); err = set_memory_rox((unsigned long)p, size >> PAGE_SHIFT); if (err) - goto err_free_mem; - - return p; + return NULL; -err_free_mem: - vfree(p); - return NULL; + return no_free_ptr(p); } static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t size) { unsigned long vm_flags = VM_REQUIRE_HUGE_VMAP; size_t alloc_size = round_up(size, PMD_SIZE); - struct mutex *mutex = &execmem_cache.mutex; + void *p __free(vfree) = NULL; int err; - void *p; p = execmem_vmalloc_rox(range, alloc_size, vm_flags); if (!p) @@ -306,20 +299,15 @@ static void *execmem_cache_populate_alloc(struct execmem_range *range, size_t si * as an atomic operation, otherwise they may be consumed * by a parallel call to the execmem_cache_alloc function. */ - mutex_lock(mutex); + guard(mutex)(&execmem_cache.mutex); err = execmem_cache_add_locked(p, alloc_size, GFP_KERNEL); - if (!err) - p = execmem_cache_alloc_locked(range, size); - mutex_unlock(mutex); - if (err) - goto err_free_mem; + return NULL; - return p; + /* the chunk belongs to the cache now */ + retain_and_null_ptr(p); -err_free_mem: - vfree(p); - return NULL; + return execmem_cache_alloc_locked(range, size); } static void *execmem_alloc_rox(struct execmem_range *range, size_t size) From acd9d62e00f95fdc880b2cf8e6479b741c47080c Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:31 -0400 Subject: [PATCH 0495/1012] mm/swap: add folio_swap_entry() and folio_page_swap_entry() Patch series "mm: remove page_swap_entry()", v2. A folio in the swap cache occupies folio_nr_pages() contiguous swap entries starting at folio->swap, so a page's swap entry is just folio->swap plus the page's index in the folio. The swap entry is folio state, but the only helper for it is page-based: page_swap_entry() takes a page and recomputes the folio its callers already hold. Several callers avoid it by open-coding the arithmetic on folio->swap instead. Add folio_swap_entry() and folio_page_swap_entry(), convert all users, and remove page_swap_entry(), with a few other cleanups along the way. This patch (of 8): A folio in the swap cache occupies folio_nr_pages() contiguous swap entries starting at folio->swap, so a page's swap entry is just folio->swap plus the page's index in the folio. page_swap_entry() hides this behind a compound_head() call, and callers that already have the folio sometimes open-code the arithmetic instead. Add folio_swap_entry(), which takes a folio and a page index, and folio_page_swap_entry() for callers that have the page. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-0-ee6d01dfa5e1@columbia.edu Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-1-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- include/linux/swap.h | 33 +++++++++++++++++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/include/linux/swap.h b/include/linux/swap.h index 74ce794042474c..3da4e5843c0105 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -272,6 +272,39 @@ struct swap_info_struct { const struct swap_ops *ops; }; +/** + * folio_swap_entry - Return the swap entry for a page within a folio. + * @folio: The folio. + * @idx: The index of the page within the folio. + * + * A folio in the swap cache occupies folio_nr_pages() contiguous swap + * entries starting at folio->swap. The caller must ensure the folio is + * in the swap cache and that @idx is within the folio. + */ +static inline +swp_entry_t folio_swap_entry(const struct folio *folio, unsigned long idx) +{ + swp_entry_t entry = folio->swap; + + VM_WARN_ON_ONCE_FOLIO(idx >= folio_nr_pages(folio), folio); + entry.val += idx; + return entry; +} + +/** + * folio_page_swap_entry - Return the swap entry of a page in a folio. + * @folio: The folio containing @page. + * @page: A page within @folio. + * + * The caller must ensure the folio is in the swap cache and that @page + * is part of @folio. + */ +static inline swp_entry_t folio_page_swap_entry(const struct folio *folio, + const struct page *page) +{ + return folio_swap_entry(folio, folio_page_idx(folio, page)); +} + static inline swp_entry_t page_swap_entry(struct page *page) { struct folio *folio = page_folio(page); From a9936ea9eb5b342d210d36460dbeb19f0c2f2034 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Sun, 13 Sep 2026 17:52:28 -0400 Subject: [PATCH 0496/1012] mm-swap-add-folio_swap_entry-and-folio_page_swap_entry-fix adjust the folio_swap_entry() kerneldoc summary, per David Link: https://lore.kernel.org/20260913-folio_swap_entry-doc-fix-1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Cc: David Hildenbrand --- include/linux/swap.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index 3da4e5843c0105..a59737b7268173 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -273,7 +273,7 @@ struct swap_info_struct { }; /** - * folio_swap_entry - Return the swap entry for a page within a folio. + * folio_swap_entry - Return the swap entry at a page index within a folio. * @folio: The folio. * @idx: The index of the page within the folio. * From f21dc4575ecc7139e289c5985ca2a6d53e533496 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:32 -0400 Subject: [PATCH 0497/1012] mm/huge_memory: add a comment to the open-coded swap entry The swap entry of each new folio in __split_folio_to_order() is computed by hand from folio->swap rather than with folio_swap_entry(), because the folio's page count is not valid while it is being split. Add a comment explaining this. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-2-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon --- mm/huge_memory.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index dd66c6ad5af13c..009eb3adc2b783 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3762,6 +3762,10 @@ static void __split_folio_to_order(struct folio *folio, int old_order, */ VM_WARN_ON_ONCE_PAGE(new_folio->private, new_head); + /* + * Not all folio fields are valid during a split, so open-code + * the swap entry rather than using folio_swap_entry(). + */ if (folio_test_swapcache(folio)) new_folio->swap.val = folio->swap.val + i; From d22c6f8aaa82c928a7ea695829fd754969824a92 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:33 -0400 Subject: [PATCH 0498/1012] mm/rmap: use folio_page_swap_entry() in ttu_anon_swapbacked_folio() We already have the folio here, so use folio_page_swap_entry() instead of going through page_swap_entry(). This saves a call to compound_head(). No functional change. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-3-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- mm/rmap.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/rmap.c b/mm/rmap.c index 0a3952706faf5c..5332c52909be18 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -2147,7 +2147,7 @@ static bool ttu_anon_swapbacked_folio(struct vm_area_struct *vma, { const bool anon_exclusive = folio_test_anon(folio) && PageAnonExclusive(page); - swp_entry_t entry = page_swap_entry(page); + swp_entry_t entry = folio_page_swap_entry(folio, page); struct mm_struct *mm = vma->vm_mm; if (folio_dup_swap(folio, page) < 0) From 09e0fc1b8372ded2b3e581e788aab09176077a51 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:34 -0400 Subject: [PATCH 0499/1012] mm/zswap: use folio_swap_entry() in zswap_store_page() zswap_store_page() open-codes the swap entry computation from folio->swap and the page index. Use folio_swap_entry() instead. No functional change. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-4-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- mm/zswap.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index b894de1786fd3f..3a6f8901764641 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1395,8 +1395,7 @@ static bool zswap_store_page(struct folio *folio, long index, struct obj_cgroup *objcg, struct zswap_pool *pool) { - swp_entry_t page_swpentry = swp_entry(swp_type(folio->swap), - swp_offset(folio->swap) + index); + swp_entry_t page_swpentry = folio_swap_entry(folio, index); struct zswap_entry *entry, *old; /* allocate entry */ From 42579e72cb72eb045d3789349329e52a81465501 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:35 -0400 Subject: [PATCH 0500/1012] mm/swapfile: use folio_page_swap_entry() folio_dup_swap() and folio_put_swap() open-code the swap entry computation from folio->swap and folio_page_idx(). Use folio_page_swap_entry() instead. No functional change. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-5-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- mm/swapfile.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 01e7b6b046b67d..48d3cd40defdb4 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1822,7 +1822,7 @@ int folio_dup_swap(struct folio *folio, struct page *page) VM_WARN_ON_FOLIO(!folio_test_swapcache(folio), folio); if (page) { - entry.val += folio_page_idx(folio, page); + entry = folio_page_swap_entry(folio, page); nr_pages = 1; } @@ -1849,7 +1849,7 @@ void folio_put_swap(struct folio *folio, struct page *page) VM_WARN_ON_FOLIO(!folio_test_swapcache(folio), folio); if (page) { - entry.val += folio_page_idx(folio, page); + entry = folio_page_swap_entry(folio, page); nr_pages = 1; } From d64c44233ce76b7e8527b795d82360779c823182 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:36 -0400 Subject: [PATCH 0501/1012] arm64: mte: make mte_save_tags() and mte_restore_tags() static mte_save_tags() and mte_restore_tags() are only used by arch_prepare_to_swap() and arch_swap_restore() in mteswap.c. Make them static and remove their declarations from mte.h. No functional change. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-6-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- arch/arm64/include/asm/mte.h | 2 -- arch/arm64/mm/mteswap.c | 4 ++-- 2 files changed, 2 insertions(+), 4 deletions(-) diff --git a/arch/arm64/include/asm/mte.h b/arch/arm64/include/asm/mte.h index 7f7b97e099968e..83f3b05fc78490 100644 --- a/arch/arm64/include/asm/mte.h +++ b/arch/arm64/include/asm/mte.h @@ -23,9 +23,7 @@ unsigned long mte_copy_tags_from_user(void *to, const void __user *from, unsigned long n); unsigned long mte_copy_tags_to_user(void __user *to, void *from, unsigned long n); -int mte_save_tags(struct page *page); void mte_save_page_tags(const void *page_addr, void *tag_storage); -void mte_restore_tags(swp_entry_t entry, struct page *page); void mte_restore_page_tags(void *page_addr, const void *tag_storage); void mte_invalidate_tags(int type, pgoff_t offset); void mte_invalidate_tags_area(int type); diff --git a/arch/arm64/mm/mteswap.c b/arch/arm64/mm/mteswap.c index 63e8d72f202a3b..3c54afc8b7585b 100644 --- a/arch/arm64/mm/mteswap.c +++ b/arch/arm64/mm/mteswap.c @@ -20,7 +20,7 @@ void mte_free_tag_storage(char *storage) kfree(storage); } -int mte_save_tags(struct page *page) +static int mte_save_tags(struct page *page) { void *tag_storage, *ret; @@ -47,7 +47,7 @@ int mte_save_tags(struct page *page) return 0; } -void mte_restore_tags(swp_entry_t entry, struct page *page) +static void mte_restore_tags(swp_entry_t entry, struct page *page) { void *tags = xa_load(&mte_pages, entry.val); From 3e3c22d76f060fc831c1020111f963eea63a6e64 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:37 -0400 Subject: [PATCH 0502/1012] arm64: mte: pass the swap entry to mte_save_tags() arch_prepare_to_swap() derives a page from the folio only for mte_save_tags() to recompute the folio's swap entry from that page. Pass the entry in directly with folio_swap_entry(), matching mte_restore_tags(), and move the page_mte_tagged() check into the caller so we only compute the entry for tagged pages. __mte_invalidate_tags() loses its only user, so remove it and call mte_invalidate_tags() directly in the error path. This removes two calls to compound_head(). No functional change. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-7-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- arch/arm64/mm/mteswap.c | 30 +++++++++++++----------------- 1 file changed, 13 insertions(+), 17 deletions(-) diff --git a/arch/arm64/mm/mteswap.c b/arch/arm64/mm/mteswap.c index 3c54afc8b7585b..f64202b69309a7 100644 --- a/arch/arm64/mm/mteswap.c +++ b/arch/arm64/mm/mteswap.c @@ -20,22 +20,17 @@ void mte_free_tag_storage(char *storage) kfree(storage); } -static int mte_save_tags(struct page *page) +static int mte_save_tags(swp_entry_t entry, struct page *page) { void *tag_storage, *ret; - if (!page_mte_tagged(page)) - return 0; - tag_storage = mte_allocate_tag_storage(); if (!tag_storage) return -ENOMEM; mte_save_page_tags(page_address(page), tag_storage); - /* lookup the swap entry.val from the page */ - ret = xa_store(&mte_pages, page_swap_entry(page).val, tag_storage, - GFP_KERNEL); + ret = xa_store(&mte_pages, entry.val, tag_storage, GFP_KERNEL); if (WARN(xa_is_err(ret), "Failed to store MTE tags")) { mte_free_tag_storage(tag_storage); return xa_err(ret); @@ -68,13 +63,6 @@ void mte_invalidate_tags(int type, pgoff_t offset) mte_free_tag_storage(tags); } -static inline void __mte_invalidate_tags(struct page *page) -{ - swp_entry_t entry = page_swap_entry(page); - - mte_invalidate_tags(swp_type(entry), swp_offset(entry)); -} - void mte_invalidate_tags_area(int type) { swp_entry_t entry = swp_entry(type, 0); @@ -102,15 +90,23 @@ int arch_prepare_to_swap(struct folio *folio) nr = folio_nr_pages(folio); for (i = 0; i < nr; i++) { - err = mte_save_tags(folio_page(folio, i)); + struct page *page = folio_page(folio, i); + + if (!page_mte_tagged(page)) + continue; + + err = mte_save_tags(folio_swap_entry(folio, i), page); if (err) goto out; } return 0; out: - while (i--) - __mte_invalidate_tags(folio_page(folio, i)); + while (i--) { + swp_entry_t swap = folio_swap_entry(folio, i); + + mte_invalidate_tags(swp_type(swap), swp_offset(swap)); + } return err; } From 5c42dd043fdcbd610983fd29234f6d37a5505964 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 8 Sep 2026 11:16:38 -0400 Subject: [PATCH 0503/1012] mm/swap: remove page_swap_entry() All callers have been converted to folio_swap_entry() and folio_page_swap_entry(), so remove it. Link: https://lore.kernel.org/20260908-folio_swap_entry-v2-8-ee6d01dfa5e1@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Harry Yoo Cc: Jann Horn Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Lance Yang Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Marc Rutland Cc: Nhat Pham Cc: Rik van Riel Cc: Ryan Roberts Cc: Vlastimil Babka Cc: Will Deacon Cc: Zi Yan --- include/linux/swap.h | 9 --------- 1 file changed, 9 deletions(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index a59737b7268173..a37ad8375e011e 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -305,15 +305,6 @@ static inline swp_entry_t folio_page_swap_entry(const struct folio *folio, return folio_swap_entry(folio, folio_page_idx(folio, page)); } -static inline swp_entry_t page_swap_entry(struct page *page) -{ - struct folio *folio = page_folio(page); - swp_entry_t entry = folio->swap; - - entry.val += folio_page_idx(folio, page); - return entry; -} - /* linux/mm/page_alloc.c */ extern unsigned long totalreserve_pages; From 112244d547b4ca484df3a12bfd6ec9f06a915ff4 Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Tue, 8 Sep 2026 06:52:55 -0700 Subject: [PATCH 0504/1012] Docs/mm/damon/design: fix broken :ref: usage and a typo The Statistics section refers readers to the DAMON sysfs interface documentation for how to read the statistics, using a ':ref:' role. However, the role name has a superfluous 's' appended (':ref:s' instead of ':ref:'). Moreover, its target label 'sysfs_stats' does not exist; the correct label is 'sysfs_schemes_stats'. Fix both. Also fix a typo, 'frequenceis' -> 'frequencies', in the Adaptive Regions Adjustment section. Link: https://lore.kernel.org/20260908135257.97523-1-sj@kernel.org Signed-off-by: Zenghui Yu (Huawei) Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton --- Documentation/mm/damon/design.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index d036340dae8afb..aac84de261aa8a 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -249,7 +249,7 @@ For each ``aggregation interval``, it compares the access frequencies sum of the two regions' sizes is smaller than the size of total regions divided by the ``minimum number of regions``, DAMON merges the two regions. If the resulting number of total regions is still higher than ``maximum number of -regions``, it repeats the merging with increasing access frequenceis difference +regions``, it repeats the merging with increasing access frequencies difference threshold until the upper-limit of the number of regions is met, or the threshold becomes higher than possible maximum value (``aggregation interval`` divided by ``sampling interval``). Then, after it reports and clears the @@ -889,7 +889,7 @@ the scheme is deactivated. Note that, unlike watermarks, even if a scheme's ``nr_snapshots`` reaches ``max_nr_snapshots``, monitoring will not stop. To know how user-space can read the stats via :ref:`DAMON sysfs interface -`, refer to :ref:s`stats ` part of the +`, refer to :ref:`stats ` part of the documentation. Regions Walking From 143f57f606cc65711f4c48e02d522fdb25cefbb6 Mon Sep 17 00:00:00 2001 From: Krishna Iyer Date: Tue, 8 Sep 2026 06:51:53 -0700 Subject: [PATCH 0505/1012] mm/damon: move damon_hugetlb_mkold() from vaddr to ops-common Patch series "mm/damon: support access monitoring of hugetlb-backed memory", v3. On virtualization hosts, most system memory is often backed by hugetlbfs. On our production hosts, for example, ~95% of RAM is 1 GiB hugetlb pages backing guest memory. DAMON's physical address space monitoring is blind to such memory: every access check starts at damon_get_folio(), which rejects folios that are not on the LRU lists, and hugetlb folios are managed outside of the LRU by design. As a result, all hugetlb-backed memory is silently reported as never accessed. In testing on a 1 TiB host, an hour of 4-thread random access over 842 GiB inside a guest was statistically indistinguishable from an idle host. The first patch moves damon_hugetlb_mkold() from vaddr to ops-common as a preparation. The second patch teaches the folio mkold/young rmap walkers to handle hugetlb folios, aging the huge PTE and notifying secondary MMUs across the whole huge page size; the secondary MMU notification is what surfaces guest-side (e.g., KVM/EPT) accessed bits. The third patch adds damon_get_monitor_folio() and uses it from the paddr monitoring primitives only. DAMOS action appliers such as DAMON_RECLAIM and DAMON_LRU_SORT keep the LRU-only lookup and are behaviorally unchanged. This series is the first half of an earlier six-patch series [1], split out as SJ suggested [2]. The second half (the 'aging_flush' TLB-flush-assisted aging) is deferred: we will gather more quantitative data on the gap it addresses, including the workload-side impact of the flushes and the working set measurement details SJ asked about, and post it separately once the data is in hand, aligned with the ongoing monitoring preparation actions work. Per Documentation/process/generated-content.rst, this series was developed with the assistance of an AI coding assistant (Anthropic Claude, via Claude Code). The assistant helped draft the code and changelogs, and applied the v1 review feedback. All changes were reviewed by the human submitter, who takes full responsibility for the contribution. The series as posted here was regression-tested on its base commit with a full x86_64 kernel build (no W=1 warnings in mm/damon), the DAMON kunit suite (41/41 passing) and the DAMON selftests (15/15 passing) on a kernel booted with virtme-ng. This patch (of 3): damon_hugetlb_mkold() clears the accessed bit of a hugetlb-mapping huge PTE and propagates the aging to secondary MMUs via mmu_notifier_clear_young(), spanning the whole huge page size. It currently lives in vaddr.c, and is thus usable only by the virtual address space monitoring operations set. The physical address space monitoring operations set will need the same logic, to support access monitoring of hugetlb-backed memory. Move the function to ops-common as-is, with no behavioral change. A follow-up change will use it from the folio-granular rmap walkers. Link: https://lore.kernel.org/20260908135156.97481-1-sj@kernel.org Link: https://lore.kernel.org/20260902025700.17975-2-kiyer@crusoe.ai Link: https://lore.kernel.org/20260908135156.97481-2-sj@kernel.org Signed-off-by: Krishna Iyer Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Assisted-by: Claude:claude-fable-5 --- mm/damon/ops-common.c | 37 +++++++++++++++++++++++++++++++++++++ mm/damon/ops-common.h | 9 +++++++++ mm/damon/vaddr.c | 34 ---------------------------------- 3 files changed, 46 insertions(+), 34 deletions(-) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index 7219c608b1952b..995cc1f3b9f32e 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -3,6 +3,7 @@ * Common Code for Data Access Monitoring */ +#include #include #include #include @@ -103,6 +104,42 @@ void damon_pmdp_mkold(pmd_t *pmd, struct vm_area_struct *vma, unsigned long addr #endif /* CONFIG_TRANSPARENT_HUGEPAGE */ } +#ifdef CONFIG_HUGETLB_PAGE +static bool damon_hugetlb_ptep_mkold(pte_t *pte, struct mm_struct *mm, + struct vm_area_struct *vma, unsigned long addr, pte_t *entry) +{ + unsigned long psize = huge_page_size(hstate_vma(vma)); + + if (!pte_young(*entry)) + return false; + *entry = huge_ptep_get_and_clear(mm, addr, pte, psize); + *entry = pte_mkold(*entry); + set_huge_pte_at(mm, addr, pte, *entry, psize); + return true; +} + +void damon_hugetlb_mkold(pte_t *pte, struct mm_struct *mm, + struct vm_area_struct *vma, unsigned long addr) +{ + bool referenced = false; + pte_t entry = huge_ptep_get(mm, addr, pte); + struct folio *folio = pfn_folio(pte_pfn(entry)); + + folio_get(folio); + + referenced = damon_hugetlb_ptep_mkold(pte, mm, vma, addr, &entry); + if (mmu_notifier_clear_young(mm, addr, + addr + huge_page_size(hstate_vma(vma)))) + referenced = true; + + if (referenced) + folio_set_young(folio); + + folio_set_idle(folio); + folio_put(folio); +} +#endif /* CONFIG_HUGETLB_PAGE */ + #define DAMON_MAX_SUBSCORE (100) #define DAMON_MAX_AGE_IN_LOG (32) diff --git a/mm/damon/ops-common.h b/mm/damon/ops-common.h index 38d295488fa181..f7811c9c7a024b 100644 --- a/mm/damon/ops-common.h +++ b/mm/damon/ops-common.h @@ -9,6 +9,15 @@ struct folio *damon_get_folio(unsigned long pfn); void damon_ptep_mkold(pte_t *pte, struct vm_area_struct *vma, unsigned long addr); void damon_pmdp_mkold(pmd_t *pmd, struct vm_area_struct *vma, unsigned long addr); +#ifdef CONFIG_HUGETLB_PAGE +void damon_hugetlb_mkold(pte_t *pte, struct mm_struct *mm, + struct vm_area_struct *vma, unsigned long addr); +#else +static inline void damon_hugetlb_mkold(pte_t *pte, struct mm_struct *mm, + struct vm_area_struct *vma, unsigned long addr) +{ +} +#endif /* CONFIG_HUGETLB_PAGE */ void damon_folio_mkold(struct folio *folio); bool damon_folio_young(struct folio *folio); diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 91a0d441c1f940..af9e1b82454cc2 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -283,40 +283,6 @@ static int damon_mkold_pmd_entry(pmd_t *pmd, unsigned long addr, } #ifdef CONFIG_HUGETLB_PAGE -static bool damon_hugetlb_ptep_mkold(pte_t *pte, struct mm_struct *mm, - struct vm_area_struct *vma, unsigned long addr, pte_t *entry) -{ - unsigned long psize = huge_page_size(hstate_vma(vma)); - - if (!pte_young(*entry)) - return false; - *entry = huge_ptep_get_and_clear(mm, addr, pte, psize); - *entry = pte_mkold(*entry); - set_huge_pte_at(mm, addr, pte, *entry, psize); - return true; -} - -static void damon_hugetlb_mkold(pte_t *pte, struct mm_struct *mm, - struct vm_area_struct *vma, unsigned long addr) -{ - bool referenced = false; - pte_t entry = huge_ptep_get(mm, addr, pte); - struct folio *folio = pfn_folio(pte_pfn(entry)); - - folio_get(folio); - - referenced = damon_hugetlb_ptep_mkold(pte, mm, vma, addr, &entry); - if (mmu_notifier_clear_young(mm, addr, - addr + huge_page_size(hstate_vma(vma)))) - referenced = true; - - if (referenced) - folio_set_young(folio); - - folio_set_idle(folio); - folio_put(folio); -} - static int damon_mkold_hugetlb_entry(pte_t *pte, unsigned long hmask, unsigned long addr, unsigned long end, struct mm_walk *walk) From 92c5ca0f56096429bc279495a4fceb6d9abe9338 Mon Sep 17 00:00:00 2001 From: Krishna Iyer Date: Tue, 8 Sep 2026 06:51:54 -0700 Subject: [PATCH 0506/1012] mm/damon/ops-common: handle hugetlb folios in folio mkold/young rmap walkers damon_folio_mkold_one() and damon_folio_young_one() assume the folios they walk are mapped by normal PTEs or THP PMDs. When the folio is a hugetlb folio, page_vma_mapped_walk() returns the huge PTE in pvmw.pte with its page table lock held, but the walkers treat it as a normal PTE: they read and age it with PAGE_SIZE-granularity helpers, which is wrong for huge PTEs (up to PUD level), and notify secondary MMUs for only PAGE_SIZE of the mapping. Add hugetlb branches to both walkers. The mkold walker reuses damon_hugetlb_mkold(), which the virtual address space operations set has been using for hugetlb aging: it clears the young bit of the huge PTE via set_huge_pte_at() and calls mmu_notifier_clear_young() spanning the whole huge page size. The young walker gets an equivalent new helper, damon_hugetlb_young(), which reads the huge PTE with huge_ptep_get() and consults the page idle flag and mmu_notifier_test_young() like the existing PTE branch. Locking mirrors what page_vma_mapped_walk() provides: the huge PTE's page table lock is held inside the walk, and for shared hugetlb mappings (the only ones subject to huge PMD sharing), rmap_walk_file() already holds i_mmap_rwsem, satisfying hugetlb_walk()'s locking requirements. This is currently dead code: both rmap walkers are only reachable through damon_get_folio(), which rejects hugetlb folios since they are not on the LRU lists. A following commit will let the physical address space monitoring primitives opt in to hugetlb folios. Link: https://lore.kernel.org/20260902025700.17975-3-kiyer@crusoe.ai Link: https://lore.kernel.org/20260908135156.97481-3-sj@kernel.org Signed-off-by: Krishna Iyer Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Assisted-by: Claude:claude-fable-5 --- mm/damon/ops-common.c | 61 +++++++++++++++++++++++++++++++++---------- 1 file changed, 47 insertions(+), 14 deletions(-) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index 995cc1f3b9f32e..349e1604cc1b1f 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -205,10 +205,15 @@ static bool damon_folio_mkold_one(struct folio *folio, while (page_vma_mapped_walk(&pvmw)) { addr = pvmw.address; - if (pvmw.pte) - damon_ptep_mkold(pvmw.pte, vma, addr); - else + if (pvmw.pte) { + if (folio_test_hugetlb(folio)) + damon_hugetlb_mkold(pvmw.pte, vma->vm_mm, vma, + addr); + else + damon_ptep_mkold(pvmw.pte, vma, addr); + } else { damon_pmdp_mkold(pvmw.pmd, vma, addr); + } } return true; } @@ -233,27 +238,55 @@ void damon_folio_mkold(struct folio *folio) } +#ifdef CONFIG_HUGETLB_PAGE +static bool damon_hugetlb_young(pte_t *pte, struct vm_area_struct *vma, + unsigned long addr, struct folio *folio) +{ + pte_t entry = huge_ptep_get(vma->vm_mm, addr, pte); + + return (pte_present(entry) && pte_young(entry)) || + !folio_test_idle(folio) || + mmu_notifier_test_young(vma->vm_mm, addr); +} +#else +static bool damon_hugetlb_young(pte_t *pte, struct vm_area_struct *vma, + unsigned long addr, struct folio *folio) +{ + return false; +} +#endif /* CONFIG_HUGETLB_PAGE */ + +static bool damon_pte_young(pte_t *pte, struct vm_area_struct *vma, + unsigned long addr, struct folio *folio) +{ + pte_t entry = ptep_get(pte); + + /* + * PFN swap PTEs, such as device-exclusive ones, that actually map + * pages are "old" from a CPU perspective. The MMU notifier takes care + * of any device aspects. + */ + return (pte_present(entry) && pte_young(entry)) || + !folio_test_idle(folio) || + mmu_notifier_test_young(vma->vm_mm, addr); +} + static bool damon_folio_young_one(struct folio *folio, struct vm_area_struct *vma, unsigned long addr, void *arg) { bool *accessed = arg; DEFINE_FOLIO_VMA_WALK(pvmw, folio, vma, addr, 0); - pte_t pte; *accessed = false; while (page_vma_mapped_walk(&pvmw)) { addr = pvmw.address; if (pvmw.pte) { - pte = ptep_get(pvmw.pte); - - /* - * PFN swap PTEs, such as device-exclusive ones, that - * actually map pages are "old" from a CPU perspective. - * The MMU notifier takes care of any device aspects. - */ - *accessed = (pte_present(pte) && pte_young(pte)) || - !folio_test_idle(folio) || - mmu_notifier_test_young(vma->vm_mm, addr); + if (folio_test_hugetlb(folio)) + *accessed = damon_hugetlb_young(pvmw.pte, vma, + addr, folio); + else + *accessed = damon_pte_young(pvmw.pte, vma, + addr, folio); } else { #ifdef CONFIG_TRANSPARENT_HUGEPAGE pmd_t pmd = pmdp_get(pvmw.pmd); From 519626270b78f3f163eef142af27043bf428e6e5 Mon Sep 17 00:00:00 2001 From: Krishna Iyer Date: Tue, 8 Sep 2026 06:51:55 -0700 Subject: [PATCH 0507/1012] mm/damon/paddr: support hugetlb folios in access monitoring DAMON's physical address space monitoring is blind to hugetlb-backed memory. Every access check starts at damon_get_folio(), which rejects folios that are not on the LRU lists. Hugetlb folios are managed outside of the LRU by design, so every sampling attempt on hugetlb-backed memory silently fails and the pages are reported as never accessed. This is a significant blind spot on virtualization hosts. Cloud hypervisor hosts commonly back guest memory with 1 GiB hugetlbfs pages, covering the vast majority of the machine's memory. On such hosts, modules like DAMON_STAT observe only the host-side remainder (page cache, daemons) and report all guest working sets as permanently idle, defeating the purpose of host-level access monitoring. In testing on a 1 TiB host, an hour of 4-thread random access over 842 GiB inside a guest was statistically indistinguishable from an idle host, while a 40x smaller host-side workload produced a quantitatively correct response. Add damon_get_monitor_folio(), which additionally accepts hugetlb folios, and use it in the two paddr access monitoring primitives, damon_pa_mkold() and damon_pa_young(). With the previous commit teaching the folio-granular rmap walkers to age huge PTEs and to call the mmu notifiers spanning the whole huge page, this makes guest accesses visible through secondary MMU (e.g. KVM/EPT) young bits. Free hugetlb pool folios have a zero refcount, so folio_try_get() naturally keeps rejecting them. The DAMOS action appliers (damon_pa_pageout(), damon_pa_mark_accessed_or_deactivate(), damon_pa_migrate(), damon_pa_stat()) keep using damon_get_folio(): reclaim, LRU manipulation and migration cannot act on hugetlb folios, so their behavior is unchanged. Note that the access check granularity for hugetlb-backed memory is the huge page size: one touched byte reports the whole (up to 1 GiB) page as accessed. Also, DAMON now consumes secondary MMU young bits that KVM's own aging uses; at DAMON's sampling rate (one page per region per sampling interval) the interference is negligible. Link: https://lore.kernel.org/20260902025700.17975-4-kiyer@crusoe.ai Link: https://lore.kernel.org/20260908135156.97481-4-sj@kernel.org Signed-off-by: Krishna Iyer Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Assisted-by: Claude:claude-fable-5 --- mm/damon/ops-common.c | 25 +++++++++++++++++++++---- mm/damon/ops-common.h | 1 + mm/damon/paddr.c | 4 ++-- 3 files changed, 24 insertions(+), 6 deletions(-) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index 349e1604cc1b1f..acf8f216c51cc8 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -15,14 +15,20 @@ #include "../internal.h" #include "ops-common.h" +static bool damon_folio_acceptable(struct folio *folio, bool monitor) +{ + return folio_test_lru(folio) || + (monitor && folio_test_hugetlb(folio)); +} + /* - * Get an online page for a pfn if it's in the LRU list. Otherwise, returns - * NULL. + * Get an online page for a pfn if it's in the LRU list, or a hugetlb folio if + * @monitor is set. Otherwise, returns NULL. * * The body of this function is stolen from the 'page_idle_get_folio()'. We * steal rather than reuse it because the code is quite simple. */ -struct folio *damon_get_folio(unsigned long pfn) +static struct folio *__damon_get_folio(unsigned long pfn, bool monitor) { struct page *page = pfn_to_online_page(pfn); struct folio *folio; @@ -33,13 +39,24 @@ struct folio *damon_get_folio(unsigned long pfn) folio = page_folio(page); if (!folio_try_get(folio)) return NULL; - if (unlikely(page_folio(page) != folio) || !folio_test_lru(folio)) { + if (unlikely(page_folio(page) != folio) || + !damon_folio_acceptable(folio, monitor)) { folio_put(folio); folio = NULL; } return folio; } +struct folio *damon_get_folio(unsigned long pfn) +{ + return __damon_get_folio(pfn, false); +} + +struct folio *damon_get_monitor_folio(unsigned long pfn) +{ + return __damon_get_folio(pfn, true); +} + void damon_ptep_mkold(pte_t *pte, struct vm_area_struct *vma, unsigned long addr) { pte_t pteval = ptep_get(pte); diff --git a/mm/damon/ops-common.h b/mm/damon/ops-common.h index f7811c9c7a024b..172f0f17c4a84d 100644 --- a/mm/damon/ops-common.h +++ b/mm/damon/ops-common.h @@ -6,6 +6,7 @@ #include struct folio *damon_get_folio(unsigned long pfn); +struct folio *damon_get_monitor_folio(unsigned long pfn); void damon_ptep_mkold(pte_t *pte, struct vm_area_struct *vma, unsigned long addr); void damon_pmdp_mkold(pmd_t *pmd, struct vm_area_struct *vma, unsigned long addr); diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index c1e7d7a4f40df3..d2173a448d0b0c 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -37,7 +37,7 @@ static unsigned long damon_pa_core_addr( static void damon_pa_mkold(phys_addr_t paddr) { - struct folio *folio = damon_get_folio(PHYS_PFN(paddr)); + struct folio *folio = damon_get_monitor_folio(PHYS_PFN(paddr)); if (!folio) return; @@ -67,7 +67,7 @@ static void damon_pa_prepare_access_checks(struct damon_ctx *ctx) static bool damon_pa_young(phys_addr_t paddr) { - struct folio *folio = damon_get_folio(PHYS_PFN(paddr)); + struct folio *folio = damon_get_monitor_folio(PHYS_PFN(paddr)); bool accessed; if (!folio) From bd3c2399affaffaaa18683b0d03bc1e19cee590d Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Tue, 8 Sep 2026 21:41:15 +0800 Subject: [PATCH 0508/1012] selftests/mm: fix size truncation in pagemap_ioctl test Patch series "selftests/mm: pagemap_ioctl test fixes and cleanups", v2. This series fixes a size truncation bug in the pagemap_ioctl test that breaks it on arm64 systems with 64K base pages, and applies two small cleanups suggested during the review of the fix. This patch (of 3): On arm64 with 64K base pages, the huge page size is 512 MiB, and hpage_unit_tests() builds a 5 GiB range (10 * 512 MiB) for its tests. This exceeds the range of the int size parameters of gethugepage(), wp_addr_range() and pagemap_ioctl(). The implicit truncation to 1 GiB makes gethugepage() allocate a too small buffer, while the callers keep operating on the original 5 GiB range, resulting in spurious failures or SIGSEGV. Fix the truncation by changing those size parameters to size_t, and for consistency, also convert the remaining size-related parameters and variables that use int, long or unsigned long long to size_t. Link: https://lore.kernel.org/20260908134117.84405-1-zenghui.yu@linux.dev Link: https://lore.kernel.org/20260908134117.84405-2-zenghui.yu@linux.dev Fixes: 46fd75d4a3c9 ("selftests: mm: add pagemap ioctl tests") Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: David Hildenbrand (Arm) Assisted-by: GLM-5.3 OpenCode Cc: Gregory Price Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/pagemap_ioctl.c | 55 ++++++++++++---------- 1 file changed, 29 insertions(+), 26 deletions(-) diff --git a/tools/testing/selftests/mm/pagemap_ioctl.c b/tools/testing/selftests/mm/pagemap_ioctl.c index eadc7159ca5b90..3665530eda7688 100644 --- a/tools/testing/selftests/mm/pagemap_ioctl.c +++ b/tools/testing/selftests/mm/pagemap_ioctl.c @@ -44,7 +44,7 @@ const char *progname; #define LEN(region) ((region.end - region.start)/page_size) -static long pagemap_ioctl(void *start, int len, void *vec, int vec_len, int flag, +static long pagemap_ioctl(void *start, size_t len, void *vec, size_t vec_len, int flag, int max_pages, long required_mask, long anyof_mask, long excluded_mask, long return_mask) { @@ -65,7 +65,7 @@ static long pagemap_ioctl(void *start, int len, void *vec, int vec_len, int flag return ioctl(pagemap_fd, PAGEMAP_SCAN, &arg); } -static long pagemap_ioc(void *start, int len, void *vec, int vec_len, int flag, +static long pagemap_ioc(void *start, size_t len, void *vec, size_t vec_len, int flag, int max_pages, long required_mask, long anyof_mask, long excluded_mask, long return_mask, long *walk_end) { @@ -116,7 +116,7 @@ int init_uffd(void) return 0; } -int wp_init(void *addr, long size) +int wp_init(void *addr, size_t size) { struct uffdio_register uffdio_register; struct uffdio_writeprotect wp; @@ -140,7 +140,7 @@ int wp_init(void *addr, long size) return 0; } -int wp_free(void *addr, long size) +int wp_free(void *addr, size_t size) { struct uffdio_register uffdio_register; @@ -152,7 +152,7 @@ int wp_free(void *addr, long size) return 0; } -int wp_addr_range(void *addr, int size) +int wp_addr_range(void *addr, size_t size) { if (pagemap_ioctl(addr, size, NULL, 0, PM_SCAN_WP_MATCHING | PM_SCAN_CHECK_WPASYNC, @@ -162,7 +162,7 @@ int wp_addr_range(void *addr, int size) return 0; } -void *gethugetlb_mem(int size, int *shmid) +void *gethugetlb_mem(size_t size, int *shmid) { char *mem; @@ -188,7 +188,8 @@ void *gethugetlb_mem(int size, int *shmid) int userfaultfd_tests(void) { - long mem_size, vec_size, written, num_pages = 16; + size_t mem_size, vec_size, num_pages = 16; + long written; char *mem, *vec; mem_size = num_pages * page_size; @@ -229,9 +230,10 @@ int userfaultfd_tests(void) return 0; } -int get_reads(struct page_region *vec, int vec_size) +int get_reads(struct page_region *vec, size_t vec_size) { - int i, sum = 0; + size_t i; + int sum = 0; for (i = 0; i < vec_size; i++) sum += LEN(vec[i]); @@ -241,7 +243,7 @@ int get_reads(struct page_region *vec, int vec_size) int sanity_tests_sd(void) { - unsigned long long mem_size, vec_size, i, total_pages = 0; + size_t mem_size, vec_size, i, total_pages = 0; long ret, ret2, ret3; int num_pages = 1000; int total_writes, total_reads, reads, count; @@ -331,7 +333,7 @@ int sanity_tests_sd(void) if (ret < 0) ksft_exit_fail_msg("error %ld %d %s\n", ret, errno, strerror(errno)); - ksft_test_result((unsigned long long)ret == mem_size/(page_size * 2), + ksft_test_result((size_t)ret == mem_size/(page_size * 2), "%s Repeated pattern of written and non-written pages\n", __func__); /* 4. Repeated pattern of written and non-written pages in parts */ @@ -682,9 +684,9 @@ int sanity_tests_sd(void) return 0; } -int base_tests(char *prefix, char *mem, unsigned long long mem_size, int skip) +int base_tests(char *prefix, char *mem, size_t mem_size, int skip) { - unsigned long long vec_size; + size_t vec_size; int written; struct page_region *vec, *vec2; @@ -787,7 +789,7 @@ int base_tests(char *prefix, char *mem, unsigned long long mem_size, int skip) return 0; } -void *gethugepage(int map_size) +void *gethugepage(size_t map_size) { int ret; char *map; @@ -810,8 +812,8 @@ int hpage_unit_tests(void) char *map; int ret, ret2; size_t num_pages = 10; - unsigned long long map_size = hpage_size * num_pages; - unsigned long long vec_size = map_size/page_size; + size_t map_size = hpage_size * num_pages; + size_t vec_size = map_size/page_size; struct page_region *vec, *vec2; vec = calloc(vec_size, sizeof(struct page_region)); @@ -1002,8 +1004,9 @@ int hpage_unit_tests(void) int unmapped_region_tests(void) { void *start = (void *)0x10000000; - int written, len = 0x00040000; - long vec_size = len / page_size; + int written; + size_t len = 0x00040000; + size_t vec_size = len / page_size; struct page_region *vec = calloc(vec_size, sizeof(struct page_region)); if (!vec) ksft_exit_fail_msg("error nomem\n"); @@ -1072,7 +1075,7 @@ static void test_simple(void) * with no page table, exercising pagemap_scan_pte_hole(); a base-page range * leaves pte_none entries. */ -static void unpopulated_written_test(const char *name, char *mem, long size, +static void unpopulated_written_test(const char *name, char *mem, size_t size, bool use_thp) { long npages = size / page_size, fast = 0, slow = 0, ret; @@ -1115,7 +1118,7 @@ static void unpopulated_written_test(const char *name, char *mem, long size, static void unpopulated_scan_test(void) { - long mem_size = 16 * page_size; + size_t mem_size = 16 * page_size; char *mem; mem = mmap(NULL, mem_size, PROT_READ | PROT_WRITE, @@ -1157,8 +1160,8 @@ static void unpopulated_thp_scan_test(void) int sanity_tests(void) { - unsigned long long mem_size, vec_size; - long ret, fd, i, buf_size, nr_pages; + size_t mem_size, vec_size, i, buf_size; + long ret, fd, nr_pages; struct page_region *vec; char *mem, *fmem; struct stat sbuf; @@ -1582,9 +1585,9 @@ static void transact_test(int page_size) void zeropfn_tests(void) { - unsigned long long mem_size; + size_t mem_size, i; struct page_region vec; - int i, ret; + int ret; char *mmap_mem, *mem; /* Test with normal memory */ @@ -1642,8 +1645,8 @@ void zeropfn_tests(void) int main(int __attribute__((unused)) argc, char *argv[]) { - int shmid, buf_size, fd, i, ret; - unsigned long long mem_size; + int shmid, fd, ret; + size_t mem_size, buf_size, i; char *mem, *map, *fmem; struct stat sbuf; From bcb8f48c3d04e42f86c62b7a89c7619fcdfc95ee Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Tue, 8 Sep 2026 21:43:14 +0800 Subject: [PATCH 0509/1012] selftests/mm: mark file-local symbols of pagemap_ioctl.c static The file-scope variables (pagemap_fd, uffd, page_size, hpage_size and progname) and most functions of the pagemap_ioctl test are only used locally, but lack the static storage class. Mark them static so that the compiler can catch accidental outer references. Link: https://lore.kernel.org/20260908134315.84431-1-zenghui.yu@linux.dev Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: David Hildenbrand (Arm) Cc: Gregory Price Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/pagemap_ioctl.c | 43 +++++++++++----------- 1 file changed, 21 insertions(+), 22 deletions(-) diff --git a/tools/testing/selftests/mm/pagemap_ioctl.c b/tools/testing/selftests/mm/pagemap_ioctl.c index 3665530eda7688..1a87b7483316d8 100644 --- a/tools/testing/selftests/mm/pagemap_ioctl.c +++ b/tools/testing/selftests/mm/pagemap_ioctl.c @@ -36,11 +36,11 @@ #define TEST_ITERATIONS 100 #define PAGEMAP "/proc/self/pagemap" -int pagemap_fd; -int uffd; -size_t page_size; -size_t hpage_size; -const char *progname; +static int pagemap_fd; +static int uffd; +static size_t page_size; +static size_t hpage_size; +static const char *progname; #define LEN(region) ((region.end - region.start)/page_size) @@ -92,8 +92,7 @@ static long pagemap_ioc(void *start, size_t len, void *vec, size_t vec_len, int return ret; } - -int init_uffd(void) +static int init_uffd(void) { struct uffdio_api uffdio_api; @@ -116,7 +115,7 @@ int init_uffd(void) return 0; } -int wp_init(void *addr, size_t size) +static int wp_init(void *addr, size_t size) { struct uffdio_register uffdio_register; struct uffdio_writeprotect wp; @@ -140,7 +139,7 @@ int wp_init(void *addr, size_t size) return 0; } -int wp_free(void *addr, size_t size) +static int wp_free(void *addr, size_t size) { struct uffdio_register uffdio_register; @@ -152,7 +151,7 @@ int wp_free(void *addr, size_t size) return 0; } -int wp_addr_range(void *addr, size_t size) +static int wp_addr_range(void *addr, size_t size) { if (pagemap_ioctl(addr, size, NULL, 0, PM_SCAN_WP_MATCHING | PM_SCAN_CHECK_WPASYNC, @@ -162,7 +161,7 @@ int wp_addr_range(void *addr, size_t size) return 0; } -void *gethugetlb_mem(size_t size, int *shmid) +static void *gethugetlb_mem(size_t size, int *shmid) { char *mem; @@ -186,7 +185,7 @@ void *gethugetlb_mem(size_t size, int *shmid) return mem; } -int userfaultfd_tests(void) +static int userfaultfd_tests(void) { size_t mem_size, vec_size, num_pages = 16; long written; @@ -230,7 +229,7 @@ int userfaultfd_tests(void) return 0; } -int get_reads(struct page_region *vec, size_t vec_size) +static int get_reads(struct page_region *vec, size_t vec_size) { size_t i; int sum = 0; @@ -241,7 +240,7 @@ int get_reads(struct page_region *vec, size_t vec_size) return sum; } -int sanity_tests_sd(void) +static int sanity_tests_sd(void) { size_t mem_size, vec_size, i, total_pages = 0; long ret, ret2, ret3; @@ -684,7 +683,7 @@ int sanity_tests_sd(void) return 0; } -int base_tests(char *prefix, char *mem, size_t mem_size, int skip) +static int base_tests(char *prefix, char *mem, size_t mem_size, int skip) { size_t vec_size; int written; @@ -789,7 +788,7 @@ int base_tests(char *prefix, char *mem, size_t mem_size, int skip) return 0; } -void *gethugepage(size_t map_size) +static void *gethugepage(size_t map_size) { int ret; char *map; @@ -807,7 +806,7 @@ void *gethugepage(size_t map_size) return map; } -int hpage_unit_tests(void) +static int hpage_unit_tests(void) { char *map; int ret, ret2; @@ -1001,7 +1000,7 @@ int hpage_unit_tests(void) return 0; } -int unmapped_region_tests(void) +static int unmapped_region_tests(void) { void *start = (void *)0x10000000; int written; @@ -1158,7 +1157,7 @@ static void unpopulated_thp_scan_test(void) munmap(area, 2 * hpage_size); } -int sanity_tests(void) +static int sanity_tests(void) { size_t mem_size, vec_size, i, buf_size; long ret, fd, nr_pages; @@ -1330,7 +1329,7 @@ int sanity_tests(void) return 0; } -int mprotect_tests(void) +static int mprotect_tests(void) { int ret; char *mem, *mem2; @@ -1450,7 +1449,7 @@ static ssize_t get_dirty_pages_reset(char *mem, unsigned int count, return cnt; } -void *thread_proc(void *mem) +static void *thread_proc(void *mem) { int *m = mem; long curr_faults, faults; @@ -1583,7 +1582,7 @@ static void transact_test(int page_size) extra_thread_faults); } -void zeropfn_tests(void) +static void zeropfn_tests(void) { size_t mem_size, i; struct page_region vec; From acdca9cfbedf5bad6185c1653cbac3cf69eb7f38 Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Tue, 8 Sep 2026 21:44:05 +0800 Subject: [PATCH 0510/1012] selftests/mm: init page sizes early in pagemap_ioctl test Initialize page_size and hpage_size before calling init_uffd(), hugetlb_setup_default(), etc. That won't fix anything, but it is safer and saner to get these globals set up before doing other things. While at it, drop the page_size parameter of transact_test(), which is actually unnecessary. Link: https://lore.kernel.org/20260908134405.84448-1-zenghui.yu@linux.dev Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: Andrew Morton Suggested-by: Andrew Morton Link: https://lore.kernel.org/20260628111329.9cfcd9c67925869307020aba@linux-foundation.org/ Cc: David Hildenbrand (Arm) Cc: Gregory Price Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/pagemap_ioctl.c | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/tools/testing/selftests/mm/pagemap_ioctl.c b/tools/testing/selftests/mm/pagemap_ioctl.c index 1a87b7483316d8..d9a4fb782ecfe7 100644 --- a/tools/testing/selftests/mm/pagemap_ioctl.c +++ b/tools/testing/selftests/mm/pagemap_ioctl.c @@ -1489,7 +1489,7 @@ static void *thread_proc(void *mem) return NULL; } -static void transact_test(int page_size) +static void transact_test(void) { unsigned int i, count, extra_pages; unsigned int c; @@ -1653,6 +1653,9 @@ int main(int __attribute__((unused)) argc, char *argv[]) ksft_print_header(); + page_size = getpagesize(); + hpage_size = read_pmd_pagesize(); + if (init_uffd()) ksft_exit_skip("Failed to initialize userfaultfd\n"); @@ -1661,9 +1664,6 @@ int main(int __attribute__((unused)) argc, char *argv[]) ksft_set_plan(119); - page_size = getpagesize(); - hpage_size = read_pmd_pagesize(); - pagemap_fd = open(PAGEMAP, O_RDONLY); if (pagemap_fd < 0) ksft_exit_fail_msg("Failed to open " PAGEMAP "\n"); @@ -1823,7 +1823,7 @@ int main(int __attribute__((unused)) argc, char *argv[]) mprotect_tests(); /* 13. Transact test */ - transact_test(page_size); + transact_test(); /* 14. Sanity testing */ sanity_tests(); From 9ec5ce4bcc8992f040a8b798202a11e243a57dbb Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Tue, 8 Sep 2026 14:28:21 +0100 Subject: [PATCH 0511/1012] mm/huge_memory: add folio_reset_partially_mapped() __folio_unqueue_deferred_split() and __folio_freeze_and_split_unmapped() both clear PG_partially_mapped and take the folio out of MTHP_STAT_NR_ANON_PARTIALLY_MAPPED with the same five lines. Move the block into folio_reset_partially_mapped() and call it from both places. The helper asserts what both callers rely on: the folio is frozen, so deferred_split_folio() cannot set the flag again under it, and the folio is already off the deferred split queue. The list check sits behind the flag test because order-1 folios have no _deferred_list. folio_order() is safe to use at this point in the split process: it still shows the pre-split order. Link: https://lore.kernel.org/20260908132821.1517475-1-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: David Hildenbrand (Arm) Acked-by: Balbir Singh Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Reviewed-by: Lorenzo Stoakes (ARM) Assisted-by: LLM --- mm/huge_memory.c | 32 +++++++++++++++++++++----------- 1 file changed, 21 insertions(+), 11 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 009eb3adc2b783..194188c292af3d 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3976,6 +3976,25 @@ static unsigned int folio_cache_ref_count(const struct folio *folio) return folio_nr_pages(folio); } +static void folio_reset_partially_mapped(struct folio *folio) +{ + /* Folio must be frozen. */ + VM_WARN_ON_FOLIO(folio_ref_count(folio), folio); + + if (!folio_test_partially_mapped(folio)) + return; + + /* + * Order-1 folios have no _deferred_list. The flag is only ever set + * on folios that do, so the list can be checked after the flag. + */ + VM_WARN_ON_FOLIO(!list_empty(&folio->_deferred_list), folio); + + folio_clear_partially_mapped(folio); + mod_mthp_stat(folio_order(folio), + MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); +} + static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int new_order, struct page *split_at, struct xa_state *xas, struct address_space *mapping, bool do_lru, @@ -3984,7 +4003,6 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n { struct folio *end_folio = folio_next(folio); struct folio *new_folio, *next; - int old_order = folio_order(folio); int ret = 0; VM_WARN_ON_ONCE(!mapping && end); @@ -4002,11 +4020,7 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n * leaves PG_partially_mapped set. * Clear it here: the flag does not survive the split. */ - if (folio_test_partially_mapped(folio)) { - folio_clear_partially_mapped(folio); - mod_mthp_stat(old_order, - MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); - } + folio_reset_partially_mapped(folio); if (mapping) { int nr = folio_nr_pages(folio); @@ -4520,11 +4534,7 @@ bool __folio_unqueue_deferred_split(struct folio *folio) memcg = folio_memcg(folio); lru = list_lru_lock_irqsave(&deferred_split_lru, nid, &memcg, &flags); if (__list_lru_del(&deferred_split_lru, lru, &folio->_deferred_list, nid)) { - if (folio_test_partially_mapped(folio)) { - folio_clear_partially_mapped(folio); - mod_mthp_stat(folio_order(folio), - MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); - } + folio_reset_partially_mapped(folio); unqueued = true; } list_lru_unlock_irqrestore(lru, &flags); From 60be6ae48dd359cb887557074a8d3660a2e5acaf Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:36 +0100 Subject: [PATCH 0512/1012] mm/khugepaged: deposit a newly allocated page table on collapse Patch series "mm: make userland page table freeing RCU-safe", v5. The majority of architectures in the kernel defer page table freeing until an RCU grace period has elapsed, this series converts all remaining architectures to do so too and eliminates CONFIG_MMU_GATHER_RCU_TABLE_FREE altogether. This is important because it enables safe lockless page table walking under RCU alone. Doing so allows for reduced lock contention, avoids lock ordering concerns and enables fast, efficient and correct page table walking as a result. Additionally it removes a bunch of code and architecture-specific behaviour which is always a beneficial thing to do. There has been much recent work on this: * In 2023 Hugh Dickins RCU-deferred khugepaged page table retraction in commit 13cf577e6b66 ("mm/pgtable: add pte_free_defer() for pgtable as page"). * Qi Zheng has done most of the work that made this possible starting with the critical commit 718b13861d22 ("x86: mm: free page table pages by RCU instead of semi RCU"). * Qi then went on to convert a large number of architectures in commit e3ecf7c7d082 ("mm: pgtable: convert some architectures to use tlb_remove_ptdesc()"), commit 44b079583f7d ("alpha: mm: enable MMU_GATHER_RCU_TABLE_FREE") and the series to which it belongs. * Qi then introduced the important CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE option in commit 086498aed3f6 ("mm: convert __HAVE_ARCH_TLB_REMOVE_TABLE to CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE config"). * Finally, and critically, Lance Yang then converted the batch allocation fallback case to be RCU-safe in commit 1fb3d8c20bfa ("mm/mmu_gather: replace IPI with synchronize_rcu() when batch allocation fails"). The work I do here is only possible due to the work Hugh, Qi, Lance and others have done previously. An initial task this series addresses is to deposit a freshly allocated PTE page table and RCU-free the existing PTE page table. Not doing so is currently safe, but for page table walkers relying on RCU alone, it would not be. Additionally, it makes it possible to implement lockless RCU-only page table walkers which otherwise would have required the PTE PTL. The changes are largely mechanical - the majority of arches already have the machinery required to support CONFIG_MMU_GATHER_RCU_TABLE_FREE and simply needed configuration changes or small implementation changes to switch over. However some arches required extra attention - sh-X2, m68k-motorola and sparc32. sh-X2 allocates PMDs from the slab allocator and PTEs as normal. Therefore CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE is set to customise page table freeing and the LSB is used to encode which page table level is used, with __tlb_remove_table() doing the right thing depending on this. This pattern is repeated for m68k-motorola and sparc32 to account for different page table levels. In each case, the page tables are aligned such that sufficient bits are available in each case for encoding this information. m68k-motorola required the biggest change - since RCU page table freeing uses call_rcu(), this means page table freeing can arise from softirq context. This was fixed with a BH-disabling spin lock used in both get_pointer_table() and free_pointer_table() (softirq being the only asynchronous context in which the lock is taken). As part of this change, the logic for allocation of a new pointer table was separated out into add_pointer_table() to make the locking more obviously correct. Finally, sparc32 was similar to m68k-motorola in that locking was required, however this was already implemented via a spinlock, and only had to be updated to be IRQ-safe. Additionally, the nocache pool's bit_map lock was updated to be IRQ-safe for softirq frees. Separately, the PTE path can't take mm->page_table_lock from softirq (no mm there), which is fine because the page reference count transitions are atomic and fully ordered. The series finally removes CONFIG_MMU_GATHER_RCU_TABLE_FREE and all related configurations and code that supported !CONFIG_MMU_GATHER_RCU_TABLE_FREE. As a result, page table walks can now be performed safely under RCU without any risk of page tables being freed underneath a walker. However, this is the only guarantee that this work provides - page table walkers must still ensure that page table entries are as expected throughout. All changes have been build tested. As most of the conversions are simply utilising existing mechanics that are known to work, this suffices for most cases. However those arches where significant changes have been made - m68k-motorola, sparc32 and sh-X2 - have been tested further. For each of these a boot test and stress test has been performed - fork 400 children, each mmap()'ing 2 MiB and touching every page then partially munmap()'ing then exiting to trigger as much page table freeing as possible. All were found to be working correctly. (Note that sparc32 LEON SMP is not possible to emulate.) This patch (of 12): collapse_huge_page() deposits a PTE page table on PMD collapse in order that it can be utilised for subsequent split operations, meaning that those operations do not need to perform an allocation (as they are in a context where it might be unwise). However the PTE page table which is deposited is the one which is currently mapped by the PMD entry that is in the process of being collapsed. Once deposited, the PTE page table may be used in a split of any other unrelated PMD entry. This is currently not an issue as this operation is performed with VMA/mmap write lock + anon rmap locks held, so ordinary page table walkers will never accidentally end up walking the wrong thing, and GUP-fast is protected by an IPI via tlb_remove_table_sync_one(). However, the series to which this commit belongs implements RCU-safe page table traversal, at which point this becomes problematic. This can be resolved by using pte_offset_map_lock() which gates on a PTE PTL and a pmd_same() check, but lockless walks are unsafe as things stand. Resolve this by simply allocating a new, zeroed, PTE page table to deposit at the point of collapse. This path is already costly and an allocation has already been performed for the huge folio, so this allocation is statistical noise in terms of performance and memory usage at this point. With this PTE page table deposited, RCU-free the existing PTE page table so it is safe for page table walkers to traverse within a grace period. This also brings this deposit case in line with all other page table deposit logic which deposit a fresh page table. Additionally, this was the only place in the kernel that displaced a page table like this, so eliminating it also helps consistency. Since khugepaged runs as a kernel thread, do a little dance in alloc_deposit_pte_table() to correctly charge the allocation. This is already done for the folio allocation via alloc_charge_folio() but no such wrapper exists for a page table allocation. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-0-31e91065fea4@kernel.org Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-1-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Cc: Zi Yan Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Kiryl Shutsemau Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- mm/khugepaged.c | 34 ++++++++++++++++++++++++++++++++-- 1 file changed, 32 insertions(+), 2 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 792166950bba81..85ea095906fb70 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1278,6 +1278,23 @@ static enum scan_result alloc_charge_folio(struct folio **foliop, struct mm_stru return SCAN_SUCCEED; } +static pgtable_t alloc_deposit_pte_table(struct mm_struct *mm) +{ + /* + * khugepaged is run from a kernel thread, so need to manually set the + * correct memcg so the allocation gets charged correctly. + */ + struct mem_cgroup *memcg = get_mem_cgroup_from_mm(mm); + struct mem_cgroup *old_memcg = set_active_memcg(memcg); + pgtable_t pgtable; + + pgtable = pte_alloc_one(mm); + + set_active_memcg(old_memcg); + mem_cgroup_put(memcg); + return pgtable; +} + /* * collapse_huge_page() expects the mmap_lock to be unlocked before entering and * will always return with the lock unlocked, to avoid holding the mmap_lock @@ -1293,7 +1310,7 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s LIST_HEAD(compound_pagelist); pmd_t *pmd, _pmd; pte_t *pte = NULL; - pgtable_t pgtable; + pgtable_t pgtable = NULL; struct folio *folio; spinlock_t *pmd_ptl, *pte_ptl; enum scan_result result = SCAN_FAIL; @@ -1310,6 +1327,14 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s goto out_nolock; } + if (is_pmd_order(order)) { + pgtable = alloc_deposit_pte_table(mm); + if (!pgtable) { + result = SCAN_ALLOC_HUGE_PAGE_FAIL; + goto out_nolock; + } + } + mmap_read_lock(mm); result = hugepage_vma_revalidate(mm, pmd_addr, /*expect_anon=*/ true, &vma, cc, order); @@ -1433,8 +1458,8 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s spin_lock(pmd_ptl); VM_WARN_ON_ONCE(!pmd_none(*pmd)); if (is_pmd_order(order)) { - pgtable = pmd_pgtable(_pmd); pgtable_trans_huge_deposit(mm, pmd, pgtable); + pgtable = NULL; map_anon_folio_pmd_nopf(folio, pmd, vma, pmd_addr); } else { /* @@ -1453,6 +1478,9 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s } spin_unlock(pmd_ptl); + if (is_pmd_order(order)) + pte_free_defer(mm, pmd_pgtable(_pmd)); + folio = NULL; result = SCAN_SUCCEED; @@ -1463,6 +1491,8 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s anon_vma_unlock_write(vma->anon_vma); mmap_write_unlock(mm); out_nolock: + if (pgtable) + pte_free(mm, pgtable); if (folio) folio_put(folio); trace_mm_collapse_huge_page(mm, result == SCAN_SUCCEED, result, order); From 0b2d80dc30e09f678c1066bcddb4b2d1a0413d64 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:37 +0100 Subject: [PATCH 0513/1012] mm: enable MMU_GATHER_RCU_TABLE_FREE for most 2-level architectures Commit e3ecf7c7d082 ("mm: pgtable: convert some architectures to use tlb_remove_ptdesc()") updated a number of architectures from using pagetable_dtor() + tlb_remove_page_ptdesc() to using tlb_remove_ptdesc() in __pte_free_tlb(). This is meaningful as tlb_remove_ptdesc() allows for RCU page table freeing if CONFIG_MMU_GATHER_RCU_TABLE_FREE is specified. The csky, hexagon, nios2, openrisc, sh (except X2) and m68k-sun3 architectures all have 2 levels of page tables, so the only page tables ever freed by mmu_gather are PTEs, so this update suffices to ensure that every page table freed by the mmu_gather mechanism is freed under RCU. Therefore, update all of these architectures to select CONFIG_MMU_GATHER_RCU_TABLE_FREE. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-2-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Reviewed-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/csky/Kconfig | 1 + arch/hexagon/Kconfig | 1 + arch/m68k/Kconfig | 1 + arch/nios2/Kconfig | 1 + arch/openrisc/Kconfig | 1 + arch/sh/Kconfig | 1 + 6 files changed, 6 insertions(+) diff --git a/arch/csky/Kconfig b/arch/csky/Kconfig index 4331313a42ff30..80f89ef1d9622c 100644 --- a/arch/csky/Kconfig +++ b/arch/csky/Kconfig @@ -96,6 +96,7 @@ config CSKY select HAVE_SYSCALL_TRACEPOINTS select HOTPLUG_CORE_SYNC_DEAD if HOTPLUG_CPU select LOCK_MM_AND_FIND_VMA + select MMU_GATHER_RCU_TABLE_FREE select MAY_HAVE_SPARSE_IRQ select MODULES_USE_ELF_RELA if MODULES select OF diff --git a/arch/hexagon/Kconfig b/arch/hexagon/Kconfig index b4849114001335..d9b3fb86556be3 100644 --- a/arch/hexagon/Kconfig +++ b/arch/hexagon/Kconfig @@ -23,6 +23,7 @@ config HEXAGON # select HAVE_CLK select GENERIC_ATOMIC64 select HAVE_PERF_EVENTS + select MMU_GATHER_RCU_TABLE_FREE # GENERIC_ALLOCATOR is used by dma_alloc_coherent() select GENERIC_ALLOCATOR select GENERIC_IRQ_PROBE diff --git a/arch/m68k/Kconfig b/arch/m68k/Kconfig index 11835eb59d94db..e29610fd1240f6 100644 --- a/arch/m68k/Kconfig +++ b/arch/m68k/Kconfig @@ -36,6 +36,7 @@ config M68K select HAVE_MOD_ARCH_SPECIFIC select HAVE_UID16 select MMU_GATHER_NO_RANGE if MMU + select MMU_GATHER_RCU_TABLE_FREE if MMU && SUN3 select MODULES_USE_ELF_REL select MODULES_USE_ELF_RELA select NO_DMA if !MMU && !COLDFIRE diff --git a/arch/nios2/Kconfig b/arch/nios2/Kconfig index 9c0e6eaeb005cd..b0ccfc3b7a7e1b 100644 --- a/arch/nios2/Kconfig +++ b/arch/nios2/Kconfig @@ -19,6 +19,7 @@ config NIOS2 select HAVE_PAGE_SIZE_4KB select IRQ_DOMAIN select LOCK_MM_AND_FIND_VMA + select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select OF select OF_EARLY_FLATTREE diff --git a/arch/openrisc/Kconfig b/arch/openrisc/Kconfig index 5eb995c13074c0..d90b24dd3bce3c 100644 --- a/arch/openrisc/Kconfig +++ b/arch/openrisc/Kconfig @@ -35,6 +35,7 @@ config OPENRISC select GENERIC_ATOMIC64 select GENERIC_CLOCKEVENTS_BROADCAST select GENERIC_SMP_IDLE_THREAD + select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select HAVE_DEBUG_STACKOVERFLOW select OR1K_PIC diff --git a/arch/sh/Kconfig b/arch/sh/Kconfig index d60f1d5a94c0f4..204f64912f0e42 100644 --- a/arch/sh/Kconfig +++ b/arch/sh/Kconfig @@ -61,6 +61,7 @@ config SUPERH select HAVE_SYSCALL_TRACEPOINTS select IRQ_FORCED_THREADING select LOCK_MM_AND_FIND_VMA + select MMU_GATHER_RCU_TABLE_FREE if MMU && !X2TLB select MODULES_USE_ELF_RELA select NEED_SG_DMA_LENGTH select NO_DMA if !MMU && !DMA_COHERENT From ceaa630959d75ce9c4dac5b9eea32ac572b9a639 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:38 +0100 Subject: [PATCH 0514/1012] mm: enable MMU_GATHER_RCU_TABLE_FREE for MMU riscv Currently riscv gates MMU_GATHER_RCU_TABLE_FREE on CONFIG_SMP and CONFIG_MMU. Commit 69be3fb111e7 ("riscv: enable MMU_GATHER_RCU_TABLE_FREE for SMP && MMU") enabled CONFIG_MMU_GATHER_RCU_TABLE_FREE for CONFIG_SMP, CONFIG_MMU riscv builds. This is expressly for the safety of GUP-fast walkers (CONFIG_HAVE_GUP_FAST is enabled if CONFIG_MMU is enabled). Naturally a single core system does not encounter issues with software page table walkers being correctly synchronised across cores, as there is only a single core. However, CONFIG_PREEMPT_RCU is still available on a riscv UP system, so for a future RCU-only page table walker, this guarantee is required to prevent concurrent page table teardown. All page table freeing is already done via tlb_remove_ptdesc() so the conditions of CONFIG_MMU_GATHER_RCU_TABLE_FREE are already met. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-3-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Acked-by: Lance Yang Tested-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/riscv/Kconfig | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/riscv/Kconfig b/arch/riscv/Kconfig index 5965666194b0f0..c13ef9cd688d33 100644 --- a/arch/riscv/Kconfig +++ b/arch/riscv/Kconfig @@ -208,7 +208,7 @@ config RISCV select IRQ_FORCED_THREADING select KASAN_VMALLOC if KASAN select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE if SMP && MMU + select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_RELA if MODULES select OF select OF_EARLY_FLATTREE From 9861256e47f53f9b9db94a85fcd0c0862f152709 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:39 +0100 Subject: [PATCH 0515/1012] mm: enable MMU_GATHER_RCU_TABLE_FREE for MMU arm Commit a0ad5496b2b3 ("arm: mm: enable HAVE_RCU_TABLE_FREE logic") enabled CONFIG_MMU_GATHER_RCU_TABLE_FREE (then named HAVE_RCU_TABLE_FREE) for SMP arm architectures with LPAE enabled. Regardless of whether CONFIG_ARM_LPAE is enabled or not, the same page table freeing functions __pte_free_tlb() and __pmd_free_tlb() are used. Non-LPAE PMD page tables are folded into the PGD and freed by pgd_free() (PGD freeing is not part of mmu_gather page table freeing in any case), so this is a noop in this case. Since commit 358d1c39c82a ("arm: convert various functions to use ptdescs") both LPAE and non-LPAE PTE page table freeing uses tlb_remove_ptdesc(). Thus all page table freeing is performed under RCU with CONFIG_MMU_GATHER_RCU_TABLE_FREE enabled for LPAE and non-LPAE and thus it need not be gated on LPAE. A UP arm system can set CONFIG_PREEMPT_RCU, so a future pure RCU page table walker requires MMU_GATHER_RCU_TABLE_FREE to be enabled on UP as well, even if concurrent GUP fast is not possible there. Therefore, it is both safe and desirable to set CONFIG_MMU_GATHER_RCU_TABLE_FREE for all MMU arm architectures (nommu does not perform mmu_gather operations). This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-4-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Tested-by: Lance Yang Acked-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/arm/Kconfig | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm/Kconfig b/arch/arm/Kconfig index 408aa58a2a5bbc..72b9afc6ae1058 100644 --- a/arch/arm/Kconfig +++ b/arch/arm/Kconfig @@ -134,7 +134,7 @@ config ARM select HAVE_PERF_REGS select HAVE_PERF_USER_STACK_DUMP select HAVE_POSIX_CPU_TIMERS_TASK_WORK - select MMU_GATHER_RCU_TABLE_FREE if SMP && ARM_LPAE + select MMU_GATHER_RCU_TABLE_FREE if MMU select HAVE_REGS_AND_STACK_ACCESS_API select HAVE_RSEQ select HAVE_RUST if CPU_LITTLE_ENDIAN && CPU_32v7 && !KASAN From 2294a54e48009984400c6d30be20252f2b867d37 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:40 +0100 Subject: [PATCH 0516/1012] mm: enable MMU_GATHER_RCU_TABLE_FREE for arc, microblaze, xtensa Each of these architectures directly free page tables without routing these changes through tlb_remove_ptdesc(). The use of tlb_remove_ptdesc() is required for CONFIG_MMU_GATHER_RCU_TABLE_FREE to correctly free page tables under RCU, so simply update these architectures to use these functions. Since none of the architectures share page tables or do anything unusual, nothing complicated is required here. Therefore this is simply a mechanical change - convert __pud_free_tlb(), __pmd_free_tlb() and __pte_free_tlb() to use tlb_remove_ptdesc() as required. At the point this is in place, all mmu_gather page table freeing is performed under RCU, and thus MMU_GATHER_RCU_TABLE_FREE is selected for each architecture. Note that CONFIG_MMU_GATHER_RCU_TABLE_FREE is dependent on CONFIG_MMU for xtensa to reflect the fact that nommu does not implement page table gathering. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-5-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Acked-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/arc/Kconfig | 1 + arch/arc/include/asm/pgalloc.h | 6 +++--- arch/microblaze/Kconfig | 1 + arch/microblaze/include/asm/pgalloc.h | 2 +- arch/xtensa/Kconfig | 1 + arch/xtensa/include/asm/tlb.h | 2 +- 6 files changed, 8 insertions(+), 5 deletions(-) diff --git a/arch/arc/Kconfig b/arch/arc/Kconfig index 80e61175bf526b..0c1679f4d386f8 100644 --- a/arch/arc/Kconfig +++ b/arch/arc/Kconfig @@ -47,6 +47,7 @@ config ARC select HAVE_SYSCALL_TRACEPOINTS select IRQ_DOMAIN select LOCK_MM_AND_FIND_VMA + select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select OF select OF_EARLY_FLATTREE diff --git a/arch/arc/include/asm/pgalloc.h b/arch/arc/include/asm/pgalloc.h index dfae070fe8d556..9b6c37f92e97f3 100644 --- a/arch/arc/include/asm/pgalloc.h +++ b/arch/arc/include/asm/pgalloc.h @@ -72,7 +72,7 @@ static inline void p4d_populate(struct mm_struct *mm, p4d_t *p4dp, pud_t *pudp) set_p4d(p4dp, __p4d((unsigned long)pudp)); } -#define __pud_free_tlb(tlb, pmd, addr) pud_free((tlb)->mm, pmd) +#define __pud_free_tlb(tlb, pmd, addr) tlb_remove_ptdesc((tlb), virt_to_ptdesc(pmd)) #endif @@ -83,10 +83,10 @@ static inline void pud_populate(struct mm_struct *mm, pud_t *pudp, pmd_t *pmdp) set_pud(pudp, __pud((unsigned long)pmdp)); } -#define __pmd_free_tlb(tlb, pmd, addr) pmd_free((tlb)->mm, pmd) +#define __pmd_free_tlb(tlb, pmd, addr) tlb_remove_ptdesc((tlb), virt_to_ptdesc(pmd)) #endif -#define __pte_free_tlb(tlb, pte, addr) pte_free((tlb)->mm, pte) +#define __pte_free_tlb(tlb, pte, addr) tlb_remove_ptdesc((tlb), page_ptdesc(pte)) #endif /* _ASM_ARC_PGALLOC_H */ diff --git a/arch/microblaze/Kconfig b/arch/microblaze/Kconfig index 484ebb3baedf15..af7e821e96c1da 100644 --- a/arch/microblaze/Kconfig +++ b/arch/microblaze/Kconfig @@ -41,6 +41,7 @@ config MICROBLAZE select PCI_SYSCALL if PCI select CPU_NO_EFFICIENT_FFS select MMU_GATHER_NO_RANGE + select MMU_GATHER_RCU_TABLE_FREE select SPARSE_IRQ select ZONE_DMA select TRACE_IRQFLAGS_SUPPORT diff --git a/arch/microblaze/include/asm/pgalloc.h b/arch/microblaze/include/asm/pgalloc.h index 084a8a0dc23952..ffee6a009219ac 100644 --- a/arch/microblaze/include/asm/pgalloc.h +++ b/arch/microblaze/include/asm/pgalloc.h @@ -25,7 +25,7 @@ extern void __bad_pte(pmd_t *pmd); extern pte_t *pte_alloc_one_kernel(struct mm_struct *mm); -#define __pte_free_tlb(tlb, pte, addr) pte_free((tlb)->mm, (pte)) +#define __pte_free_tlb(tlb, pte, addr) tlb_remove_ptdesc((tlb), page_ptdesc(pte)) #define pmd_populate(mm, pmd, pte) \ (pmd_val(*(pmd)) = (unsigned long)page_address(pte)) diff --git a/arch/xtensa/Kconfig b/arch/xtensa/Kconfig index f2f9cd9cde505d..33c4caee30e27b 100644 --- a/arch/xtensa/Kconfig +++ b/arch/xtensa/Kconfig @@ -55,6 +55,7 @@ config XTENSA select HAVE_VIRT_CPU_ACCOUNTING_GEN select IRQ_DOMAIN select LOCK_MM_AND_FIND_VMA + select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_RELA select PERF_USE_VMALLOC select TRACE_IRQFLAGS_SUPPORT diff --git a/arch/xtensa/include/asm/tlb.h b/arch/xtensa/include/asm/tlb.h index 8c3ceb4270180b..6fb7b78154f62a 100644 --- a/arch/xtensa/include/asm/tlb.h +++ b/arch/xtensa/include/asm/tlb.h @@ -16,7 +16,7 @@ #include -#define __pte_free_tlb(tlb, pte, address) pte_free((tlb)->mm, pte) +#define __pte_free_tlb(tlb, pte, address) tlb_remove_ptdesc((tlb), page_ptdesc(pte)) void check_tlb_sanity(void); From 8065a71e973287f2959644b7f8c46cb0f7e11d50 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:41 +0100 Subject: [PATCH 0517/1012] mm: enable MMU_GATHER_RCU_TABLE_FREE for sparc64 Commit 4a0100f7546f ("sparc64: use RCU page table freeing") enabled CONFIG_MMU_GATHER_RCU_TABLE_FREE for SMP sparc64 architectures, expressly for GUP-fast page table walkers. Naturally, UP systems do not have to worry about concurrent GUP fast operations. However, CONFIG_PREEMPT_RCU is also available even on a UP system, so a future pure-RCU page table walker requires MMU_GATHER_RCU_TABLE_FREE to be enabled on UP, even if concurrent GUP fast is not possible there. To enable future pure-RCU page table walkers, enable MMU_GATHER_RCU_TABLE_FREE unconditionally. With this change, it is no longer necessary to have !CONFIG_SMP pgtable_free_tlb(), so also remove this now dead code. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-6-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Tested-by: Lance Yang Acked-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/sparc/Kconfig | 4 ++-- arch/sparc/include/asm/pgalloc_64.h | 8 -------- 2 files changed, 2 insertions(+), 10 deletions(-) diff --git a/arch/sparc/Kconfig b/arch/sparc/Kconfig index ab77d3f2536e1a..8d42ebc6d3029c 100644 --- a/arch/sparc/Kconfig +++ b/arch/sparc/Kconfig @@ -75,8 +75,8 @@ config SPARC64 select HAVE_FUNCTION_GRAPH_TRACER select HAVE_KRETPROBES select HAVE_KPROBES - select MMU_GATHER_RCU_TABLE_FREE if SMP - select HAVE_ARCH_TLB_REMOVE_TABLE if SMP + select MMU_GATHER_RCU_TABLE_FREE + select HAVE_ARCH_TLB_REMOVE_TABLE select MMU_GATHER_MERGE_VMAS select MMU_GATHER_NO_FLUSH_CACHE select HAVE_ARCH_TRANSPARENT_HUGEPAGE diff --git a/arch/sparc/include/asm/pgalloc_64.h b/arch/sparc/include/asm/pgalloc_64.h index caa7632be4c2ae..b5055d259b74d6 100644 --- a/arch/sparc/include/asm/pgalloc_64.h +++ b/arch/sparc/include/asm/pgalloc_64.h @@ -74,8 +74,6 @@ void pte_free_defer(struct mm_struct *mm, pgtable_t pgtable); void pgtable_free(void *table, bool is_page); -#ifdef CONFIG_SMP - struct mmu_gather; void tlb_remove_table(struct mmu_gather *, void *); @@ -96,12 +94,6 @@ static inline void __tlb_remove_table(void *_table) is_page = true; pgtable_free(table, is_page); } -#else /* CONFIG_SMP */ -static inline void pgtable_free_tlb(struct mmu_gather *tlb, void *table, bool is_page) -{ - pgtable_free(table, is_page); -} -#endif /* !CONFIG_SMP */ static inline void __pte_free_tlb(struct mmu_gather *tlb, pte_t *pte, unsigned long address) From c8715f11d220cc22641e201b1f91b9d758c6be36 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:42 +0100 Subject: [PATCH 0518/1012] mm: enable MMU_GATHER_RCU_TABLE_FREE for m68k-coldfire Similar to sun3, the coldfire variant of m68k uses 2-level page tables. Update its __pte_free_tlb() function to use tlb_remove_ptdesc() in order that, with CONFIG_MMU_GATHER_RCU_TABLE_FREE, page tables are freed under RCU. The page tables occupy a page each and have no odd semantics, so this change suffices to allow enabling of CONFIG_MMU_GATHER_RCU_TABLE_FREE for m68k-coldfire, so do so. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-7-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Acked-by: Greg Ungerer Tested-by: Greg Ungerer Acked-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/m68k/Kconfig | 2 +- arch/m68k/include/asm/mcf_pgalloc.h | 5 +---- 2 files changed, 2 insertions(+), 5 deletions(-) diff --git a/arch/m68k/Kconfig b/arch/m68k/Kconfig index e29610fd1240f6..6b8ec67c86fde5 100644 --- a/arch/m68k/Kconfig +++ b/arch/m68k/Kconfig @@ -36,7 +36,7 @@ config M68K select HAVE_MOD_ARCH_SPECIFIC select HAVE_UID16 select MMU_GATHER_NO_RANGE if MMU - select MMU_GATHER_RCU_TABLE_FREE if MMU && SUN3 + select MMU_GATHER_RCU_TABLE_FREE if MMU && (SUN3 || COLDFIRE) select MODULES_USE_ELF_REL select MODULES_USE_ELF_RELA select NO_DMA if !MMU && !COLDFIRE diff --git a/arch/m68k/include/asm/mcf_pgalloc.h b/arch/m68k/include/asm/mcf_pgalloc.h index fc5454d37da318..b53ff0950db2e3 100644 --- a/arch/m68k/include/asm/mcf_pgalloc.h +++ b/arch/m68k/include/asm/mcf_pgalloc.h @@ -39,10 +39,7 @@ extern inline pmd_t *pmd_alloc_kernel(pgd_t *pgd, unsigned long address) static inline void __pte_free_tlb(struct mmu_gather *tlb, pgtable_t pgtable, unsigned long address) { - struct ptdesc *ptdesc = virt_to_ptdesc(pgtable); - - pagetable_dtor(ptdesc); - pagetable_free(ptdesc); + tlb_remove_ptdesc(tlb, virt_to_ptdesc(pgtable)); } static inline pgtable_t pte_alloc_one(struct mm_struct *mm) From 0baa6c28f74dd981f4c35a40d70f6ea108142d2b Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:43 +0100 Subject: [PATCH 0519/1012] mm: enable MMU_GATHER_RCU_TABLE_FREE for sh-X2 Currently, non-x2 sh specifies CONFIG_MMU_GATHER_RCU_TABLE_FREE allowing RCU page table freeing. sh-X2 is problematic because it utilises slab-allocated PMD page tables, and thus tlb_remove_ptdesc() cannot be used in these cases. All other sh variants are fine as commit e3ecf7c7d082 ("mm: pgtable: convert some architectures to use tlb_remove_ptdesc()") already converted page table freeing to use tlb_remove_ptdesc(), which does so after an RCU grace period when CONFIG_MMU_GATHER_RCU_TABLE_FREE is specified. Resolve this issue by firstly specifying CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE for sh-X2, so the arch can provide its own __tlb_remove_table() implementation (called after the RCU grace period). Then, convert __pmd_free_tlb() to tag the pointer to the PMD, and have __tlb_remove_table() check this tag to determine whether to free via the slab or to use pagetable_dtor_free(). This follows the pattern used by sparc64 as implemented in commit 4a0100f7546f ("sparc64: use RCU page table freeing"). Previously __pmd_free_tlb() freed PMD page tables immediately, before any TLB flush IPI. This seems to be a pre-existing bug, which this change also resolves. CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE is only specified for sh-X2, as setting it disables CONFIG_PT_RECLAIM and causes __tlb_remove_table_one() to call tlb_remove_table_sync_rcu() and synchronize_rcu() in turn, and this is not necessary for other sh variants. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-8-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Acked-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/sh/Kconfig | 3 ++- arch/sh/include/asm/pgalloc.h | 6 +++++- arch/sh/mm/pgtable.c | 20 ++++++++++++++++++++ 3 files changed, 27 insertions(+), 2 deletions(-) diff --git a/arch/sh/Kconfig b/arch/sh/Kconfig index 204f64912f0e42..75236bef6f16ee 100644 --- a/arch/sh/Kconfig +++ b/arch/sh/Kconfig @@ -33,6 +33,7 @@ config SUPERH select HAVE_ARCH_AUDITSYSCALL select HAVE_ARCH_KGDB select HAVE_ARCH_SECCOMP_FILTER + select HAVE_ARCH_TLB_REMOVE_TABLE if X2TLB select HAVE_ARCH_TRACEHOOK select HAVE_DEBUG_BUGVERBOSE select HAVE_DEBUG_KMEMLEAK @@ -61,7 +62,7 @@ config SUPERH select HAVE_SYSCALL_TRACEPOINTS select IRQ_FORCED_THREADING select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE if MMU && !X2TLB + select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_RELA select NEED_SG_DMA_LENGTH select NO_DMA if !MMU && !DMA_COHERENT diff --git a/arch/sh/include/asm/pgalloc.h b/arch/sh/include/asm/pgalloc.h index 6fe7123d38fa9e..67ce7fa23fa128 100644 --- a/arch/sh/include/asm/pgalloc.h +++ b/arch/sh/include/asm/pgalloc.h @@ -17,7 +17,11 @@ extern void pgd_free(struct mm_struct *mm, pgd_t *pgd); extern void pud_populate(struct mm_struct *mm, pud_t *pudp, pmd_t *pmd); extern pmd_t *pmd_alloc_one(struct mm_struct *mm, unsigned long address); extern void pmd_free(struct mm_struct *mm, pmd_t *pmd); -#define __pmd_free_tlb(tlb, pmdp, addr) pmd_free((tlb)->mm, (pmdp)) +extern void __tlb_remove_table(void *table); + +/* PMDs are slab-allocated, tag so they are freed correctly. */ +#define __pmd_free_tlb(tlb, pmdp, addr) \ + tlb_remove_table((tlb), (void *)((unsigned long)(pmdp) | 1)) #endif static inline void pmd_populate_kernel(struct mm_struct *mm, pmd_t *pmd, diff --git a/arch/sh/mm/pgtable.c b/arch/sh/mm/pgtable.c index 3a4085ea0161fe..f6184b86b89c6c 100644 --- a/arch/sh/mm/pgtable.c +++ b/arch/sh/mm/pgtable.c @@ -56,4 +56,24 @@ void pmd_free(struct mm_struct *mm, pmd_t *pmd) { kmem_cache_free(pmd_cachep, pmd); } + +static void __tlb_remove_table_slab(void *table) +{ + kmem_cache_free(pmd_cachep, table); +} + +static void __tlb_remove_table_pgtable(void *table) +{ + pagetable_dtor_free(table); +} + +void __tlb_remove_table(void *table) +{ + const unsigned long addr = (unsigned long)table; + + if (addr & 1) + __tlb_remove_table_slab((void *)(addr & ~1UL)); + else + __tlb_remove_table_pgtable(table); +} #endif /* PAGETABLE_LEVELS > 2 */ From 4a026ca0fa7aedf32b7ee340da2d0d5677a47329 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:44 +0100 Subject: [PATCH 0520/1012] mm: enable MMU_GATHER_RCU_TABLE_FREE for m68k-motorola sun3 and coldfire are already supported, however motorola requires a little more care. Here, custom table removal logic is required, so CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE is enabled for m68k-motorola. Firstly as part of this change, the page table level must be communicated to the underlying __tlb_remove_table() implementation. Take advantage of the fact that page tables are aligned by more than enough to permit setting TABLE_PTE or TABLE_PMD in the low bits of the pointer, and store this there. Then update __pte_free_tlb() and __pmd_free_tlb() to pass this through, then have __tlb_remove_table() decode this and pass it to free_pointer_table(). The page table freeing is performed via call_rcu(), so free_pointer_table() now will be invoked from softirq context, and as such may be re-entrant. Introduce a spinlock to handle this and hold it over the time a given ptable entry is being referenced in both get_pointer_table() and free_pointer_table(). As softirq is the only asynchronous context in which the lock is taken, it suffices to disable bottom halves while holding it. In order to make things a little easier in this respect, separate out the logic for adding a new ptable entry into add_pointer_table() and only hold the lock during ptable entry insertion in this case. Note that original list_add_tail(new, dp) added new prior to dp, which is ptable_list[type].next, i.e. after ptable_list[type]. The equivalent therefore is list_add(new, &ptable_list[type]), which adds new after ptable_list[type], only without needing to make reference to dp. Note that, as m68k-motorola specifies CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE, it does not enable CONFIG_PT_RECLAIM. This isn't meaningfully impactful. With this applied, all of m68k implements CONFIG_MMU_GATHER_RCU_TABLE_FREE. This forms part of an overall effort to switch every architecture to this mode. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-9-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Acked-by: Lance Yang Tested-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/m68k/Kconfig | 3 +- arch/m68k/include/asm/motorola_pgalloc.h | 9 +- arch/m68k/mm/motorola.c | 119 +++++++++++++++-------- 3 files changed, 84 insertions(+), 47 deletions(-) diff --git a/arch/m68k/Kconfig b/arch/m68k/Kconfig index 6b8ec67c86fde5..fa5d39549da96a 100644 --- a/arch/m68k/Kconfig +++ b/arch/m68k/Kconfig @@ -29,6 +29,7 @@ config M68K select HAVE_ARCH_LIBGCC_H select HAVE_ARCH_SECCOMP select HAVE_ARCH_SECCOMP_FILTER + select HAVE_ARCH_TLB_REMOVE_TABLE if MMU_MOTOROLA select HAVE_ASM_MODVERSIONS select HAVE_DEBUG_BUGVERBOSE select HAVE_EFFICIENT_UNALIGNED_ACCESS if !CPU_HAS_NO_UNALIGNED @@ -36,7 +37,7 @@ config M68K select HAVE_MOD_ARCH_SPECIFIC select HAVE_UID16 select MMU_GATHER_NO_RANGE if MMU - select MMU_GATHER_RCU_TABLE_FREE if MMU && (SUN3 || COLDFIRE) + select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_REL select MODULES_USE_ELF_RELA select NO_DMA if !MMU && !COLDFIRE diff --git a/arch/m68k/include/asm/motorola_pgalloc.h b/arch/m68k/include/asm/motorola_pgalloc.h index 1091fb0affbee4..dcde40e8b5c6a1 100644 --- a/arch/m68k/include/asm/motorola_pgalloc.h +++ b/arch/m68k/include/asm/motorola_pgalloc.h @@ -17,6 +17,7 @@ enum m68k_table_types { extern void init_pointer_table(void *table, int type); extern void *get_pointer_table(struct mm_struct *mm, int type); extern int free_pointer_table(void *table, int type); +extern void __tlb_remove_table(void *table); /* * Allocate and free page tables. The xxx_kernel() versions are @@ -47,7 +48,7 @@ static inline void pte_free(struct mm_struct *mm, pgtable_t pgtable) static inline void __pte_free_tlb(struct mmu_gather *tlb, pgtable_t pgtable, unsigned long address) { - free_pointer_table(pgtable, TABLE_PTE); + tlb_remove_table(tlb, (void *)((unsigned long)pgtable | TABLE_PTE)); } @@ -61,10 +62,10 @@ static inline int pmd_free(struct mm_struct *mm, pmd_t *pmd) return free_pointer_table(pmd, TABLE_PMD); } -static inline int __pmd_free_tlb(struct mmu_gather *tlb, pmd_t *pmd, - unsigned long address) +static inline void __pmd_free_tlb(struct mmu_gather *tlb, pmd_t *pmd, + unsigned long address) { - return free_pointer_table(pmd, TABLE_PMD); + tlb_remove_table(tlb, (void *)((unsigned long)pmd | TABLE_PMD)); } diff --git a/arch/m68k/mm/motorola.c b/arch/m68k/mm/motorola.c index b30aa69a73a6ad..f3efa0d139634c 100644 --- a/arch/m68k/mm/motorola.c +++ b/arch/m68k/mm/motorola.c @@ -20,6 +20,7 @@ #include #include #include +#include #include #include @@ -103,6 +104,8 @@ static struct list_head ptable_list[3] = { LIST_HEAD_INIT(ptable_list[2]), }; +static DEFINE_SPINLOCK(ptable_lock); + #define PD_PTABLE(ptdesc) ((ptable_desc *)&(virt_to_ptdesc((void *)(ptdesc))->pt_list)) #define PD_PTDESC(ptable) (list_entry(ptable, struct ptdesc, pt_list)) #define PD_MARKBITS(dp) (*(unsigned int *)&PD_PTDESC(dp)->pt_index) @@ -139,52 +142,65 @@ void __init init_pointer_table(void *table, int type) return; } -void *get_pointer_table(struct mm_struct *mm, int type) +/* + * For a pointer table for a user process address space, a + * table is taken from a ptdesc allocated for the purpose. Each + * ptdesc can hold 8 pointer tables. The ptdesc is remapped in + * virtual address space to be noncacheable. + */ +static void *add_pointer_table(struct mm_struct *mm, int type) { - ptable_desc *dp = ptable_list[type].next; - unsigned int mask = list_empty(&ptable_list[type]) ? 0 : PD_MARKBITS(dp); - unsigned int tmp, off; + struct ptdesc *ptdesc; + ptable_desc *new; + void *pt_addr; - /* - * For a pointer table for a user process address space, a - * table is taken from a ptdesc allocated for the purpose. Each - * ptdesc can hold 8 pointer tables. The ptdesc is remapped in - * virtual address space to be noncacheable. - */ - if (mask == 0) { - struct ptdesc *ptdesc; - ptable_desc *new; - void *pt_addr; - - ptdesc = pagetable_alloc(GFP_KERNEL | __GFP_ZERO, 0); - if (!ptdesc) - return NULL; - - pt_addr = ptdesc_address(ptdesc); - - switch (type) { - case TABLE_PTE: - /* - * m68k doesn't have SPLIT_PTE_PTLOCKS for not having - * SMP. - */ - pagetable_pte_ctor(mm, ptdesc); - break; - case TABLE_PMD: - pagetable_pmd_ctor(mm, ptdesc); - break; - case TABLE_PGD: - pagetable_pgd_ctor(ptdesc); - break; - } + ptdesc = pagetable_alloc(GFP_KERNEL | __GFP_ZERO, 0); + if (!ptdesc) + return NULL; + + pt_addr = ptdesc_address(ptdesc); + + switch (type) { + case TABLE_PTE: + /* + * m68k doesn't have SPLIT_PTE_PTLOCKS for not having + * SMP. + */ + pagetable_pte_ctor(mm, ptdesc); + break; + case TABLE_PMD: + pagetable_pmd_ctor(mm, ptdesc); + break; + case TABLE_PGD: + pagetable_pgd_ctor(ptdesc); + break; + } + + mmu_page_ctor(pt_addr); + + new = PD_PTABLE(pt_addr); - mmu_page_ctor(pt_addr); + PD_MARKBITS(new) = ptable_mask(type) - 1; + scoped_guard(spinlock_bh, &ptable_lock) + list_add(new, &ptable_list[type]); - new = PD_PTABLE(pt_addr); - PD_MARKBITS(new) = ptable_mask(type) - 1; - list_add_tail(new, dp); + return (pmd_t *)pt_addr; +} + +void *get_pointer_table(struct mm_struct *mm, int type) +{ + unsigned int tmp, off; + unsigned long mask; + ptable_desc *dp; + void *ret; - return (pmd_t *)pt_addr; + spin_lock_bh(&ptable_lock); + dp = ptable_list[type].next; + mask = list_empty(&ptable_list[type]) ? 0 : PD_MARKBITS(dp); + + if (mask == 0) { + spin_unlock_bh(&ptable_lock); + return add_pointer_table(mm, type); } for (tmp = 1, off = 0; (mask & tmp) == 0; tmp <<= 1, off += ptable_size(type)) @@ -194,7 +210,10 @@ void *get_pointer_table(struct mm_struct *mm, int type) /* move to end of list */ list_move_tail(dp, &ptable_list[type]); } - return ptdesc_address(PD_PTDESC(dp)) + off; + + ret = ptdesc_address(PD_PTDESC(dp)) + off; + spin_unlock_bh(&ptable_lock); + return ret; } int free_pointer_table(void *table, int type) @@ -204,6 +223,8 @@ int free_pointer_table(void *table, int type) unsigned long pt_addr = ptable & PAGE_MASK; unsigned int mask = 1U << ((ptable - pt_addr)/ptable_size(type)); + spin_lock_bh(&ptable_lock); + dp = PD_PTABLE(pt_addr); if (PD_MARKBITS (dp) & mask) panic ("table already free!"); @@ -213,6 +234,8 @@ int free_pointer_table(void *table, int type) if (PD_MARKBITS(dp) == ptable_mask(type)) { /* all tables in ptdesc are free, free ptdesc */ list_del(dp); + spin_unlock_bh(&ptable_lock); + mmu_page_dtor((void *)pt_addr); pagetable_dtor_free(virt_to_ptdesc((void *)pt_addr)); return 1; @@ -223,9 +246,21 @@ int free_pointer_table(void *table, int type) */ list_move(dp, &ptable_list[type]); } + + spin_unlock_bh(&ptable_lock); return 0; } +void __tlb_remove_table(void *table) +{ + /* The bottom 2 bits are used to encode page table type. */ + const unsigned long encoded = (unsigned long)table; + void *addr = (void *)(encoded & ~3UL); + const int type = encoded & 3; + + free_pointer_table(addr, type); +} + /* size of memory already mapped in head.S */ extern __initdata unsigned long m68k_init_mapped_size; From 8381f742a108ede70aef8d02ecc58dac8d98e5da Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:45 +0100 Subject: [PATCH 0521/1012] mm: enable MMU_GATHER_RCU_TABLE_FREE for sparc32 Careful handling is required for sparc32 which implements page tables as part of a shared backing page. To support this, a custom __tlb_remove_table() function is required, as specified by CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE. This allows __pte_free_tlb() and __pmd_free_tlb() to specify which page table level is being freed, which is transmitted to __tlb_remove_table() through setting the lowest bit of the page table to 1 for a PMD and 0 for a PTE (the page tables are 256-byte aligned so this is safe to do). Next, since the page table freeing is done via RCU callback, and thus might be executed in softirq context, update the spin locks to IRQ save/restore. Then, in __tlb_remove_table(), figure out whether to free a PMD page table via free_pmd_fast() or a PTE via the newly introduced __pte_free() function, using the lower bit encoded in __pte_free_tlb() or __pmd_free_tlb() to determine which to call. As part of this change use this spin lock rather than mm->page_table_lock for all shared page table exclusion, as RCU freeing means that page tables can be freed from soft IRQ context so both don't have an mm and also mm->page_table_lock is not IRQ-safe. Note that the specification of CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE disables CONFIG_PT_RECLAIM for sparc32, which mirrors sparc64. This forms part of an overall effort to switch every architecture to this mode, and with it complete, means every architecture now supports CONFIG_MMU_GATHER_RCU_TABLE_FREE. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-10-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Tested-by: Lance Yang Acked-by: Lance Yang Cc: David Hildenbrand Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/sparc/Kconfig | 2 ++ arch/sparc/include/asm/pgalloc_32.h | 7 +++++-- arch/sparc/lib/bitext.c | 14 ++++++------- arch/sparc/mm/srmmu.c | 32 ++++++++++++++++++++++++----- 4 files changed, 41 insertions(+), 14 deletions(-) diff --git a/arch/sparc/Kconfig b/arch/sparc/Kconfig index 8d42ebc6d3029c..79c09d6ee466c4 100644 --- a/arch/sparc/Kconfig +++ b/arch/sparc/Kconfig @@ -64,6 +64,8 @@ config SPARC32 select HAVE_UID16 select HAVE_PAGE_SIZE_4KB select LOCK_MM_AND_FIND_VMA + select MMU_GATHER_RCU_TABLE_FREE + select HAVE_ARCH_TLB_REMOVE_TABLE select OLD_SIGACTION select ZONE_DMA diff --git a/arch/sparc/include/asm/pgalloc_32.h b/arch/sparc/include/asm/pgalloc_32.h index 4f73e87b22a32b..36010852ba0c04 100644 --- a/arch/sparc/include/asm/pgalloc_32.h +++ b/arch/sparc/include/asm/pgalloc_32.h @@ -48,7 +48,9 @@ static inline void free_pmd_fast(pmd_t * pmd) } #define pmd_free(mm, pmd) free_pmd_fast(pmd) -#define __pmd_free_tlb(tlb, pmd, addr) pmd_free((tlb)->mm, pmd) + +#define __pmd_free_tlb(tlb, pmd, addr) \ + tlb_remove_table((tlb), (void *)((unsigned long)(pmd) | 1UL)) #define pmd_populate(mm, pmd, pte) pmd_set(pmd, pte) @@ -72,6 +74,7 @@ static inline void free_pte_fast(pte_t *pte) #define pte_free_kernel(mm, pte) free_pte_fast(pte) void pte_free(struct mm_struct * mm, pgtable_t pte); -#define __pte_free_tlb(tlb, pte, addr) pte_free((tlb)->mm, pte) +void __tlb_remove_table(void *table); +#define __pte_free_tlb(tlb, pte, addr) tlb_remove_table((tlb), (void *)(pte)) #endif /* _SPARC_PGALLOC_H */ diff --git a/arch/sparc/lib/bitext.c b/arch/sparc/lib/bitext.c index 32a5c1d9459cde..c309e27973ce6f 100644 --- a/arch/sparc/lib/bitext.c +++ b/arch/sparc/lib/bitext.c @@ -22,8 +22,6 @@ * @align: requested alignment * * Returns offset in the map or -1 if out of space. - * - * Not safe to call from an interrupt (uses spin_lock). */ int bit_map_string_get(struct bit_map *t, int len, int align) { @@ -31,6 +29,7 @@ int bit_map_string_get(struct bit_map *t, int len, int align) int off_new; int align1; int i, color; + unsigned long flags; if (t->num_colors) { /* align is overloaded to be the page color */ @@ -50,7 +49,7 @@ int bit_map_string_get(struct bit_map *t, int len, int align) BUG(); color &= align1; - spin_lock(&t->lock); + spin_lock_irqsave(&t->lock, flags); if (len < t->last_size) offset = t->first_free; else @@ -64,7 +63,7 @@ int bit_map_string_get(struct bit_map *t, int len, int align) if (offset >= t->size) offset = 0; if (count + len > t->size) { - spin_unlock(&t->lock); + spin_unlock_irqrestore(&t->lock, flags); /* P3 */ printk(KERN_ERR "bitmap out: size %d used %d off %d len %d align %d count %d\n", t->size, t->used, offset, len, align, count); @@ -90,7 +89,7 @@ int bit_map_string_get(struct bit_map *t, int len, int align) t->last_off = 0; t->used += len; t->last_size = len; - spin_unlock(&t->lock); + spin_unlock_irqrestore(&t->lock, flags); return offset; } } @@ -103,10 +102,11 @@ int bit_map_string_get(struct bit_map *t, int len, int align) void bit_map_clear(struct bit_map *t, int offset, int len) { int i; + unsigned long flags; if (t->used < len) BUG(); /* Much too late to do any good, but alas... */ - spin_lock(&t->lock); + spin_lock_irqsave(&t->lock, flags); for (i = 0; i < len; i++) { if (test_bit(offset + i, t->map) == 0) BUG(); @@ -115,7 +115,7 @@ void bit_map_clear(struct bit_map *t, int offset, int len) if (offset < t->first_free) t->first_free = offset; t->used -= len; - spin_unlock(&t->lock); + spin_unlock_irqrestore(&t->lock, flags); } void bit_map_init(struct bit_map *t, unsigned long *map, int size) diff --git a/arch/sparc/mm/srmmu.c b/arch/sparc/mm/srmmu.c index 9a74902ad18147..1c277ab3cdb848 100644 --- a/arch/sparc/mm/srmmu.c +++ b/arch/sparc/mm/srmmu.c @@ -340,38 +340,60 @@ pgd_t *get_pgd_fast(void) * Alignments up to the page size are the same for physical and virtual * addresses of the nocache area. */ + +static DEFINE_SPINLOCK(pte_page_lock); + pgtable_t pte_alloc_one(struct mm_struct *mm) { + unsigned long flags; pte_t *ptep; struct page *page; if (!(ptep = pte_alloc_one_kernel(mm))) return NULL; page = pfn_to_page(__nocache_pa((unsigned long)ptep) >> PAGE_SHIFT); - spin_lock(&mm->page_table_lock); + spin_lock_irqsave(&pte_page_lock, flags); if (page_ref_inc_return(page) == 2 && !pagetable_pte_ctor(mm, page_ptdesc(page))) { page_ref_dec(page); ptep = NULL; } - spin_unlock(&mm->page_table_lock); + spin_unlock_irqrestore(&pte_page_lock, flags); return ptep; } -void pte_free(struct mm_struct *mm, pgtable_t ptep) +static void __pte_free(pgtable_t ptep) { struct page *page; + unsigned long flags; page = pfn_to_page(__nocache_pa((unsigned long)ptep) >> PAGE_SHIFT); - spin_lock(&mm->page_table_lock); + spin_lock_irqsave(&pte_page_lock, flags); if (page_ref_dec_return(page) == 1) pagetable_dtor(page_ptdesc(page)); - spin_unlock(&mm->page_table_lock); + spin_unlock_irqrestore(&pte_page_lock, flags); srmmu_free_nocache(ptep, SRMMU_PTE_TABLE_SIZE); } +void pte_free(struct mm_struct *mm, pgtable_t ptep) +{ + __pte_free(ptep); +} + +void __tlb_remove_table(void *table) +{ + const unsigned long encoded = (unsigned long)table; + const unsigned long addr = encoded & ~1UL; + const bool is_pmd = encoded & 1; + + if (is_pmd) + free_pmd_fast((pmd_t *)addr); + else + __pte_free((pgtable_t)addr); +} + /* context handling - a dynamically sized pool is used */ #define NO_CONTEXT -1 From 116614e9f2f160a57fd8025bdb134593adb64423 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:46 +0100 Subject: [PATCH 0522/1012] mm: userland pgtable freeing is RCU-safe now, remove leftover bits Now every architecture has been converted to support CONFIG_MMU_GATHER_RCU_TABLE_FREE, this configuration option no longer makes any sense to keep around. Therefore remove it, and remove all the dead code that existed for !CONFIG_MMU_GATHER_RCU_TABLE_FREE architectures previously. Additionally, CONFIG_MMU_GATHER_TABLE_FREE is no longer necessary, as all architectures instead use CONFIG_HAVE_ARCH_TLB_REMOVE_TABLE when a custom __tlb_remove_table() is required, so remove this too. A number of architectures only enabled CONFIG_MMU_GATHER_RCU_TABLE_FREE if CONFIG_MMU was set, however the mmu_gather logic only actually does something meaningful if CONFIG_MMU is set (mmu_gather.c is only compiled in this case, for instance). As a result, there's no need to gate any of this logic on CONFIG_MMU explicitly. CONFIG_PT_RECLAIM however does have a strict dependency on CONFIG_MMU, so make this dependency explicit. Additionally, correct comments to remove references to non-RCU page table gathering and make it clear that this is not 'semi-RCU', nor has it been since commit 1fb3d8c20bfa ("mm/mmu_gather: replace IPI with synchronize_rcu() when batch allocation fails"). With this change in place the kernel policy is now that userspace page tables are freed after an RCU grace period, and thus it is now safe to unconditionally perform page table walks under RCU, safe in the knowledge that page tables will not be freed underneath the walker. This is all that is guaranteed, however, so naturally it is still incumbent upon page table walkers to ensure that the page table entries are as expected. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-11-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Reviewed-by: Lance Yang Acked-by: David Hildenbrand (Arm) Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- arch/Kconfig | 8 ---- arch/alpha/Kconfig | 1 - arch/arc/Kconfig | 1 - arch/arm/Kconfig | 1 - arch/arm64/Kconfig | 1 - arch/csky/Kconfig | 1 - arch/hexagon/Kconfig | 1 - arch/loongarch/Kconfig | 1 - arch/m68k/Kconfig | 1 - arch/microblaze/Kconfig | 1 - arch/mips/Kconfig | 1 - arch/nios2/Kconfig | 1 - arch/openrisc/Kconfig | 1 - arch/parisc/Kconfig | 1 - arch/powerpc/Kconfig | 1 - arch/riscv/Kconfig | 1 - arch/s390/Kconfig | 1 - arch/sh/Kconfig | 1 - arch/sparc/Kconfig | 2 - arch/sparc/include/asm/tlb_64.h | 2 - arch/um/Kconfig | 1 - arch/x86/Kconfig | 1 - arch/xtensa/Kconfig | 1 - include/asm-generic/tlb.h | 66 +++++---------------------------- mm/Kconfig | 2 +- mm/gup.c | 5 ++- mm/mmu_gather.c | 30 +++------------ 27 files changed, 18 insertions(+), 117 deletions(-) diff --git a/arch/Kconfig b/arch/Kconfig index 45c65777236231..6f7516916797eb 100644 --- a/arch/Kconfig +++ b/arch/Kconfig @@ -526,13 +526,6 @@ config HAVE_ARCH_JUMP_LABEL config HAVE_ARCH_JUMP_LABEL_RELATIVE bool -config MMU_GATHER_TABLE_FREE - bool - -config MMU_GATHER_RCU_TABLE_FREE - bool - select MMU_GATHER_TABLE_FREE - config MMU_GATHER_PAGE_SIZE bool @@ -548,7 +541,6 @@ config MMU_GATHER_MERGE_VMAS config MMU_GATHER_NO_GATHER bool - depends on MMU_GATHER_TABLE_FREE config ARCH_WANT_IRQS_OFF_ACTIVATE_MM bool diff --git a/arch/alpha/Kconfig b/arch/alpha/Kconfig index e53ef2d8846360..9063c7bda4e41e 100644 --- a/arch/alpha/Kconfig +++ b/arch/alpha/Kconfig @@ -42,7 +42,6 @@ config ALPHA select ARCH_STACKWALK select CPU_NO_EFFICIENT_FFS if !ALPHA_EV67 select MMU_GATHER_NO_RANGE - select MMU_GATHER_RCU_TABLE_FREE select SPARSEMEM_EXTREME if SPARSEMEM select ZONE_DMA select TRACE_IRQFLAGS_SUPPORT diff --git a/arch/arc/Kconfig b/arch/arc/Kconfig index 0c1679f4d386f8..80e61175bf526b 100644 --- a/arch/arc/Kconfig +++ b/arch/arc/Kconfig @@ -47,7 +47,6 @@ config ARC select HAVE_SYSCALL_TRACEPOINTS select IRQ_DOMAIN select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select OF select OF_EARLY_FLATTREE diff --git a/arch/arm/Kconfig b/arch/arm/Kconfig index 72b9afc6ae1058..0cc289a7184ab1 100644 --- a/arch/arm/Kconfig +++ b/arch/arm/Kconfig @@ -134,7 +134,6 @@ config ARM select HAVE_PERF_REGS select HAVE_PERF_USER_STACK_DUMP select HAVE_POSIX_CPU_TIMERS_TASK_WORK - select MMU_GATHER_RCU_TABLE_FREE if MMU select HAVE_REGS_AND_STACK_ACCESS_API select HAVE_RSEQ select HAVE_RUST if CPU_LITTLE_ENDIAN && CPU_32v7 && !KASAN diff --git a/arch/arm64/Kconfig b/arch/arm64/Kconfig index 2bbeded33da0da..b6c2dd8b26124d 100644 --- a/arch/arm64/Kconfig +++ b/arch/arm64/Kconfig @@ -221,7 +221,6 @@ config ARM64 select HAVE_RELIABLE_STACKTRACE select HAVE_POSIX_CPU_TIMERS_TASK_WORK select HAVE_FUNCTION_ARG_ACCESS_API - select MMU_GATHER_RCU_TABLE_FREE select HAVE_RSEQ select HAVE_RUST if RUSTC_SUPPORTS_ARM64 select HAVE_STACKPROTECTOR diff --git a/arch/csky/Kconfig b/arch/csky/Kconfig index 80f89ef1d9622c..4331313a42ff30 100644 --- a/arch/csky/Kconfig +++ b/arch/csky/Kconfig @@ -96,7 +96,6 @@ config CSKY select HAVE_SYSCALL_TRACEPOINTS select HOTPLUG_CORE_SYNC_DEAD if HOTPLUG_CPU select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE select MAY_HAVE_SPARSE_IRQ select MODULES_USE_ELF_RELA if MODULES select OF diff --git a/arch/hexagon/Kconfig b/arch/hexagon/Kconfig index d9b3fb86556be3..b4849114001335 100644 --- a/arch/hexagon/Kconfig +++ b/arch/hexagon/Kconfig @@ -23,7 +23,6 @@ config HEXAGON # select HAVE_CLK select GENERIC_ATOMIC64 select HAVE_PERF_EVENTS - select MMU_GATHER_RCU_TABLE_FREE # GENERIC_ALLOCATOR is used by dma_alloc_coherent() select GENERIC_ALLOCATOR select GENERIC_IRQ_PROBE diff --git a/arch/loongarch/Kconfig b/arch/loongarch/Kconfig index 1d8fb1e456d6d8..0bd8503fd5c4c2 100644 --- a/arch/loongarch/Kconfig +++ b/arch/loongarch/Kconfig @@ -188,7 +188,6 @@ config LOONGARCH select IRQ_LOONGARCH_CPU select LOCK_MM_AND_FIND_VMA select MMU_GATHER_MERGE_VMAS if MMU - select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA if MODULES select NEED_PER_CPU_EMBED_FIRST_CHUNK select NEED_PER_CPU_PAGE_FIRST_CHUNK diff --git a/arch/m68k/Kconfig b/arch/m68k/Kconfig index fa5d39549da96a..eb84c3af92c02c 100644 --- a/arch/m68k/Kconfig +++ b/arch/m68k/Kconfig @@ -37,7 +37,6 @@ config M68K select HAVE_MOD_ARCH_SPECIFIC select HAVE_UID16 select MMU_GATHER_NO_RANGE if MMU - select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_REL select MODULES_USE_ELF_RELA select NO_DMA if !MMU && !COLDFIRE diff --git a/arch/microblaze/Kconfig b/arch/microblaze/Kconfig index af7e821e96c1da..484ebb3baedf15 100644 --- a/arch/microblaze/Kconfig +++ b/arch/microblaze/Kconfig @@ -41,7 +41,6 @@ config MICROBLAZE select PCI_SYSCALL if PCI select CPU_NO_EFFICIENT_FFS select MMU_GATHER_NO_RANGE - select MMU_GATHER_RCU_TABLE_FREE select SPARSE_IRQ select ZONE_DMA select TRACE_IRQFLAGS_SUPPORT diff --git a/arch/mips/Kconfig b/arch/mips/Kconfig index d7c67cebe065f9..ed17c9b1624cac 100644 --- a/arch/mips/Kconfig +++ b/arch/mips/Kconfig @@ -97,7 +97,6 @@ config MIPS select IRQ_FORCED_THREADING select ISA if EISA select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_REL if MODULES select MODULES_USE_ELF_RELA if MODULES && 64BIT select PERF_USE_VMALLOC diff --git a/arch/nios2/Kconfig b/arch/nios2/Kconfig index b0ccfc3b7a7e1b..9c0e6eaeb005cd 100644 --- a/arch/nios2/Kconfig +++ b/arch/nios2/Kconfig @@ -19,7 +19,6 @@ config NIOS2 select HAVE_PAGE_SIZE_4KB select IRQ_DOMAIN select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select OF select OF_EARLY_FLATTREE diff --git a/arch/openrisc/Kconfig b/arch/openrisc/Kconfig index d90b24dd3bce3c..5eb995c13074c0 100644 --- a/arch/openrisc/Kconfig +++ b/arch/openrisc/Kconfig @@ -35,7 +35,6 @@ config OPENRISC select GENERIC_ATOMIC64 select GENERIC_CLOCKEVENTS_BROADCAST select GENERIC_SMP_IDLE_THREAD - select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select HAVE_DEBUG_STACKOVERFLOW select OR1K_PIC diff --git a/arch/parisc/Kconfig b/arch/parisc/Kconfig index d3afac2f0d9be9..77f67028ad89ce 100644 --- a/arch/parisc/Kconfig +++ b/arch/parisc/Kconfig @@ -80,7 +80,6 @@ config PARISC select GENERIC_CLOCKEVENTS select CPU_NO_EFFICIENT_FFS select THREAD_INFO_IN_TASK - select MMU_GATHER_RCU_TABLE_FREE select NEED_DMA_MAP_STATE select NEED_SG_DMA_LENGTH select HAVE_ARCH_KGDB diff --git a/arch/powerpc/Kconfig b/arch/powerpc/Kconfig index 2580e27e432874..0767cfcbaa422b 100644 --- a/arch/powerpc/Kconfig +++ b/arch/powerpc/Kconfig @@ -307,7 +307,6 @@ config PPC select KASAN_VMALLOC if KASAN && EXECMEM select LOCK_MM_AND_FIND_VMA select MMU_GATHER_PAGE_SIZE - select MMU_GATHER_RCU_TABLE_FREE select HAVE_ARCH_TLB_REMOVE_TABLE select MMU_GATHER_MERGE_VMAS select MMU_LAZY_TLB_SHOOTDOWN if PPC_BOOK3S_64 diff --git a/arch/riscv/Kconfig b/arch/riscv/Kconfig index c13ef9cd688d33..0db108ea146626 100644 --- a/arch/riscv/Kconfig +++ b/arch/riscv/Kconfig @@ -208,7 +208,6 @@ config RISCV select IRQ_FORCED_THREADING select KASAN_VMALLOC if KASAN select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_RELA if MODULES select OF select OF_EARLY_FLATTREE diff --git a/arch/s390/Kconfig b/arch/s390/Kconfig index b88b8504213692..a34376c05f6e3c 100644 --- a/arch/s390/Kconfig +++ b/arch/s390/Kconfig @@ -267,7 +267,6 @@ config S390 select LOCK_MM_AND_FIND_VMA select MMU_GATHER_MERGE_VMAS select MMU_GATHER_NO_GATHER - select MMU_GATHER_RCU_TABLE_FREE select MODULES_USE_ELF_RELA select NEED_DMA_MAP_STATE if PCI select NEED_PER_CPU_EMBED_FIRST_CHUNK diff --git a/arch/sh/Kconfig b/arch/sh/Kconfig index 75236bef6f16ee..fe859def918cc3 100644 --- a/arch/sh/Kconfig +++ b/arch/sh/Kconfig @@ -62,7 +62,6 @@ config SUPERH select HAVE_SYSCALL_TRACEPOINTS select IRQ_FORCED_THREADING select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_RELA select NEED_SG_DMA_LENGTH select NO_DMA if !MMU && !DMA_COHERENT diff --git a/arch/sparc/Kconfig b/arch/sparc/Kconfig index 79c09d6ee466c4..742ffff8c37f21 100644 --- a/arch/sparc/Kconfig +++ b/arch/sparc/Kconfig @@ -64,7 +64,6 @@ config SPARC32 select HAVE_UID16 select HAVE_PAGE_SIZE_4KB select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE select HAVE_ARCH_TLB_REMOVE_TABLE select OLD_SIGACTION select ZONE_DMA @@ -77,7 +76,6 @@ config SPARC64 select HAVE_FUNCTION_GRAPH_TRACER select HAVE_KRETPROBES select HAVE_KPROBES - select MMU_GATHER_RCU_TABLE_FREE select HAVE_ARCH_TLB_REMOVE_TABLE select MMU_GATHER_MERGE_VMAS select MMU_GATHER_NO_FLUSH_CACHE diff --git a/arch/sparc/include/asm/tlb_64.h b/arch/sparc/include/asm/tlb_64.h index 3037187482db7e..f5f9631685d505 100644 --- a/arch/sparc/include/asm/tlb_64.h +++ b/arch/sparc/include/asm/tlb_64.h @@ -29,9 +29,7 @@ void flush_tlb_pending(void); * and therefore we don't need a TLBI when freeing page-table pages. */ -#ifdef CONFIG_MMU_GATHER_RCU_TABLE_FREE #define tlb_needs_table_invalidate() (false) -#endif #include diff --git a/arch/um/Kconfig b/arch/um/Kconfig index d9541d13d9eb06..94b8ff70f578b5 100644 --- a/arch/um/Kconfig +++ b/arch/um/Kconfig @@ -44,7 +44,6 @@ config UML select HAVE_SYSCALL_TRACEPOINTS select THREAD_INFO_IN_TASK select SPARSE_IRQ - select MMU_GATHER_RCU_TABLE_FREE config MMU bool diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig index a8c3b3d31a2761..6e5e462ec059a1 100644 --- a/arch/x86/Kconfig +++ b/arch/x86/Kconfig @@ -283,7 +283,6 @@ config X86 select HAVE_PERF_REGS select HAVE_PERF_USER_STACK_DUMP select ASYNC_KERNEL_PGTABLE_FREE if IOMMU_SVA - select MMU_GATHER_RCU_TABLE_FREE select MMU_GATHER_MERGE_VMAS select HAVE_POSIX_CPU_TIMERS_TASK_WORK select HAVE_REGS_AND_STACK_ACCESS_API diff --git a/arch/xtensa/Kconfig b/arch/xtensa/Kconfig index 33c4caee30e27b..f2f9cd9cde505d 100644 --- a/arch/xtensa/Kconfig +++ b/arch/xtensa/Kconfig @@ -55,7 +55,6 @@ config XTENSA select HAVE_VIRT_CPU_ACCOUNTING_GEN select IRQ_DOMAIN select LOCK_MM_AND_FIND_VMA - select MMU_GATHER_RCU_TABLE_FREE if MMU select MODULES_USE_ELF_RELA select PERF_USE_VMALLOC select TRACE_IRQFLAGS_SUPPORT diff --git a/include/asm-generic/tlb.h b/include/asm-generic/tlb.h index bdcc2778ac64f4..9d827076db1969 100644 --- a/include/asm-generic/tlb.h +++ b/include/asm-generic/tlb.h @@ -67,11 +67,8 @@ * - tlb_remove_table() * * tlb_remove_table() is the basic primitive to free page-table directories - * (__p*_free_tlb()). In it's most primitive form it is an alias for - * tlb_remove_page() below, for when page directories are pages and have no - * additional constraints. - * - * See also MMU_GATHER_TABLE_FREE and MMU_GATHER_RCU_TABLE_FREE. + * (__p*_free_tlb()). Page directories are freed after an RCU grace + * period - see the comment in mm/mmu_gather.c. * * - tlb_remove_page() / tlb_remove_page_size() * - __tlb_remove_folio_pages() / __tlb_remove_page_size() @@ -151,24 +148,15 @@ * This might be useful if your architecture has size specific TLB * invalidation instructions. * - * MMU_GATHER_TABLE_FREE - * - * This provides tlb_remove_table(), to be used instead of tlb_remove_page() - * for page directores (__p*_free_tlb()). - * - * Useful if your architecture has non-page page directories. + * Page directories (__p*_free_tlb()) are always freed via tlb_remove_table(), + * after an RCU grace period (see mm/mmu_gather.c). * - * When used, an architecture is expected to provide __tlb_remove_table() or - * use the generic __tlb_remove_table(), which does the actual freeing of these - * pages. + * This serialises against software page-table walkers, including architectures + * which do not use IPIs for remote TLB invalidates. * - * MMU_GATHER_RCU_TABLE_FREE - * - * Like MMU_GATHER_TABLE_FREE, and adds semi-RCU semantics to the free (see - * comment below). - * - * Useful if your architecture doesn't use IPIs for remote TLB invalidates - * and therefore doesn't naturally serialize with software page-table walkers. + * An architecture is expected to provide __tlb_remove_table() (see + * HAVE_ARCH_TLB_REMOVE_TABLE) or use the generic __tlb_remove_table(), which + * does the actual freeing of these pages. * * MMU_GATHER_NO_FLUSH_CACHE * @@ -200,12 +188,8 @@ * various ptep_get_and_clear() functions. */ -#ifdef CONFIG_MMU_GATHER_TABLE_FREE - struct mmu_table_batch { -#ifdef CONFIG_MMU_GATHER_RCU_TABLE_FREE struct rcu_head rcu; -#endif unsigned int nr; void *tables[]; }; @@ -224,23 +208,6 @@ static inline void __tlb_remove_table(void *table) extern void tlb_remove_table(struct mmu_gather *tlb, void *table); -#else /* !CONFIG_MMU_GATHER_TABLE_FREE */ - -static inline void tlb_remove_page(struct mmu_gather *tlb, struct page *page); -/* - * Without MMU_GATHER_TABLE_FREE the architecture is assumed to have page based - * page directories and we can use the normal page batching to free them. - */ -static inline void tlb_remove_table(struct mmu_gather *tlb, void *table) -{ - struct ptdesc *ptdesc = (struct ptdesc *)table; - - pagetable_dtor(ptdesc); - tlb_remove_page(tlb, ptdesc_page(ptdesc)); -} -#endif /* CONFIG_MMU_GATHER_TABLE_FREE */ - -#ifdef CONFIG_MMU_GATHER_RCU_TABLE_FREE /* * This allows an architecture that does not use the linux page-tables for * hardware to skip the TLBI when freeing page tables. @@ -253,19 +220,6 @@ void tlb_remove_table_sync_one(void); void tlb_remove_table_sync_rcu(void); -#else - -#ifdef tlb_needs_table_invalidate -#error tlb_needs_table_invalidate() requires MMU_GATHER_RCU_TABLE_FREE -#endif - -static inline void tlb_remove_table_sync_one(void) { } - -static inline void tlb_remove_table_sync_rcu(void) { } - -#endif /* CONFIG_MMU_GATHER_RCU_TABLE_FREE */ - - #ifndef CONFIG_MMU_GATHER_NO_GATHER /* * If we can't allocate a page to make a big batch of page pointers @@ -325,9 +279,7 @@ static inline void tlb_flush_rmaps(struct mmu_gather *tlb, struct vm_area_struct struct mmu_gather { struct mm_struct *mm; -#ifdef CONFIG_MMU_GATHER_TABLE_FREE struct mmu_table_batch *batch; -#endif unsigned long start; unsigned long end; diff --git a/mm/Kconfig b/mm/Kconfig index c1ddf59c0d71a8..bc7befafb47b57 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -1465,7 +1465,7 @@ config HAVE_ARCH_TLB_REMOVE_TABLE config PT_RECLAIM def_bool y - depends on MMU_GATHER_RCU_TABLE_FREE && !HAVE_ARCH_TLB_REMOVE_TABLE + depends on MMU && !HAVE_ARCH_TLB_REMOVE_TABLE help Try to reclaim empty user page table pages in paths other than munmap and exit_mmap path. diff --git a/mm/gup.c b/mm/gup.c index a4036c02e2137f..c2dfcb4744bc3f 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -2700,8 +2700,9 @@ EXPORT_SYMBOL(get_user_pages_unlocked); * Before activating this code, please be aware that the following assumptions * are currently made: * - * *) Either MMU_GATHER_RCU_TABLE_FREE is enabled, and tlb_remove_table() is used to - * free pages containing page tables or TLB flushing requires IPI broadcast. + * *) tlb_remove_table() is used to free pages containing page tables, with + * the free deferred until an RCU grace period has elapsed (see + * mm/mmu_gather.c). * * *) ptes can be read atomically by the architecture. * diff --git a/mm/mmu_gather.c b/mm/mmu_gather.c index 3985d856de7f9b..2a72a9686773a3 100644 --- a/mm/mmu_gather.c +++ b/mm/mmu_gather.c @@ -218,8 +218,6 @@ bool __tlb_remove_page_size(struct mmu_gather *tlb, struct page *page, int page_ #endif /* MMU_GATHER_NO_GATHER */ -#ifdef CONFIG_MMU_GATHER_TABLE_FREE - static void __tlb_remove_table_free(struct mmu_table_batch *batch) { int i; @@ -230,10 +228,8 @@ static void __tlb_remove_table_free(struct mmu_table_batch *batch) free_page((unsigned long)batch); } -#ifdef CONFIG_MMU_GATHER_RCU_TABLE_FREE - /* - * Semi RCU freeing of the page directories. + * RCU freeing of the page directories. * * This is needed by some architectures to implement software pagetable walkers. * @@ -259,13 +255,13 @@ static void __tlb_remove_table_free(struct mmu_table_batch *batch) * means. * * What we do is batch the freed directory pages (tables) and RCU free them. - * We use the sched RCU variant, as that guarantees that IRQ/preempt disabling - * holds off grace periods. + * Disabling IRQs or preemption holds off RCU grace periods, so this protects + * both rcu_read_lock() and IRQ-disabling walkers. * * However, in order to batch these pages we need to allocate storage, this * allocation is deep inside the MM code and can thus easily fail on memory - * pressure. To guarantee progress we fall back to single table freeing, see - * the implementation of tlb_remove_table_one(). + * pressure. To guarantee progress we fall back to single table freeing, which + * is also RCU-deferred - see the implementation of tlb_remove_table_one(). * */ @@ -315,15 +311,6 @@ void tlb_remove_table_sync_rcu(void) synchronize_rcu(); } -#else /* !CONFIG_MMU_GATHER_RCU_TABLE_FREE */ - -static void tlb_remove_table_free(struct mmu_table_batch *batch) -{ - __tlb_remove_table_free(batch); -} - -#endif /* CONFIG_MMU_GATHER_RCU_TABLE_FREE */ - /* * If we want tlb_remove_table() to imply TLB invalidates. */ @@ -403,13 +390,6 @@ static inline void tlb_table_init(struct mmu_gather *tlb) tlb->batch = NULL; } -#else /* !CONFIG_MMU_GATHER_TABLE_FREE */ - -static inline void tlb_table_flush(struct mmu_gather *tlb) { } -static inline void tlb_table_init(struct mmu_gather *tlb) { } - -#endif /* CONFIG_MMU_GATHER_TABLE_FREE */ - static void tlb_flush_mmu_free(struct mmu_gather *tlb) { tlb_table_flush(tlb); From 928c8e7ca57e606c092c5d7f28eb52423b3a4864 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 21:09:47 +0100 Subject: [PATCH 0523/1012] mm: change the contract for free_pgtables(), update docs Now that page tables are freed after an RCU grace period, it is safe for read-only page table walkers to walk page table ranges that are being concurrently torn down, provided the mm is kept alive via mmgrab(). It is however unsafe for writers to do so, as they must obtain an appropriate lock to do so safely. Update the pte_offset_map_lock()'s comment block to reflect this. Similarly update the process addresses documentation. Link: https://lore.kernel.org/20260925-rcu-pagetable-freeing-v5-12-31e91065fea4@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Zi Yan Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Guo Ren Cc: Brian Cain Cc: Geert Uytterhoeven Cc: Dinh Nguyen Cc: Simon Schuster Cc: Jonas Bonn Cc: Stefan Kristiansson Cc: Stafford Horne Cc: Rich Felker Cc: John Paul Adrian Glaubitz Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti Cc: Russell King Cc: Vineet Gupta Cc: Michal Simek Cc: Chris Zankel Cc: Max Filippov Cc: Will Deacon Cc: Aneesh Kumar K.V Cc: Nicholas Piggin Cc: Peter Zijlstra Cc: David S. Miller Cc: Andreas Larsson Cc: Richard Henderson Cc: Matt Turner Cc: Magnus Lindholm Cc: Catalin Marinas Cc: Mark Rutland Cc: Huacai Chen Cc: WANG Xuerui Cc: Thomas Bogendoerfer Cc: James Bottomley Cc: Helge Deller Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Christophe Leroy Cc: Heiko Carstens Cc: Vasily Gorbik Cc: Alexander Gordeev Cc: Christian Borntraeger Cc: Sven Schnelle Cc: Richard Weinberger Cc: Anton Ivanov Cc: Johannes Berg Cc: Thomas Gleixner Cc: Ingo Molnar Cc: Borislav Petkov Cc: Dave Hansen Cc: H. Peter Anvin Cc: Arnd Bergmann Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu Cc: Yoshinori Sato Cc: Shakeel Butt Cc: Jonathan Corbet Cc: Randy Dunlap Cc: Hugh Dickins Cc: Qi Zheng --- Documentation/mm/process_addrs.rst | 11 +++++++++++ mm/pgtable-generic.c | 15 +++++++++++---- 2 files changed, 22 insertions(+), 4 deletions(-) diff --git a/Documentation/mm/process_addrs.rst b/Documentation/mm/process_addrs.rst index a7296f251799cb..78231995e49012 100644 --- a/Documentation/mm/process_addrs.rst +++ b/Documentation/mm/process_addrs.rst @@ -537,6 +537,17 @@ We establish basic locking rules when interacting with page tables: * When changing a page table entry the page table lock for that page table **must** be held, except if you can safely assume nobody can access the page tables concurrently (such as on invocation of :c:func:`!free_pgtables`). +* Page tables may be *walked* under RCU alone, as page tables are freed only + after an RCU grace period has elapsed. However, any entry found must be + revalidated after the page table lock is taken (such as the + :c:func:`!pmd_same` recheck performed by :c:func:`!pte_offset_map_lock`) + before it is acted upon. Changing an entry requires the page table lock + and one of the locks that excludes teardown (any one of the mmap, VMA or + rmap locks). +* When traversing page tables under RCU alone it is important to take care + when operating upon leaf entries - if the value is operated upon (for + instance getting the folio associated with a PTE) an appropriate lock must + be taken to prevent concurrent modification. * Reads from and writes to page table entries must be *appropriately* atomic. See the section on atomicity below for details. * Populating previously empty entries requires that the mmap or VMA locks are diff --git a/mm/pgtable-generic.c b/mm/pgtable-generic.c index b45e891d1193fd..26643d76bfb00e 100644 --- a/mm/pgtable-generic.c +++ b/mm/pgtable-generic.c @@ -397,10 +397,17 @@ pte_t *pte_offset_map_rw_nolock(struct mm_struct *mm, pmd_t *pmd, * Note: "RO" / "RW" expresses the intended semantics, not that the *kmap* will * be read-only/read-write protected. * - * Note that free_pgtables(), used after unmapping detached vmas, or when - * exiting the whole mm, does not take page table lock before freeing a page - * table, and may not use RCU at all: "outsiders" like khugepaged should avoid - * pte_offset_map() and co once the vma is detached from mm or mm_users is zero. + * Note that free_pgtables(), used after unmapping detached vmas or when exiting + * the whole mm, does not take a page table lock before freeing a page table. + * + * As page table freeing itself is RCU-safe, page table readers can safely run + * concurrently with page table teardown. + * + * However, writers CANNOT as, without a lock being held, nothing prevents + * concurrent teardown. + * + * Also note that the PGD itself is freed at mmdrop() time, not under RCU - so + * the walker must keep the mm alive either by pinning the mm or the VMA. */ pte_t *pte_offset_map_lock(struct mm_struct *mm, pmd_t *pmd, unsigned long addr, spinlock_t **ptlp) From 7dc317fbd7d9d9216cab659092eff0b8dbc27426 Mon Sep 17 00:00:00 2001 From: Bo Zhang Date: Tue, 8 Sep 2026 14:26:49 +0800 Subject: [PATCH 0524/1012] mm: vmscan: avoid anon scanning for GFP_NOIO with low swapcache We have observed some cases where memory is allocated with GFP_NOIO, so we cannot reclaim any anon folios unless they are in swapcache. We can end up spending more than 150 ms looping in `shrink_folio_list()` scanning non-swapcache folios without reclaiming a single folio. This is pure overhead. This is particularly true on systems using zRAM, where swapcache is relatively rare. So let's check whether anon reclaim is allowed by GFP_IO and whether there is enough swapcache to make it worthwhile. If the swapcache is extremely low, we're essentially searching for a needle in a haystack, so let's avoid scanning anon in the first place. On Android this is triggered by dm-verity hash-block reads through dm-bufio, which legitimately use GFP_NOIO because they run underneath the IO path: verity_verify_io -> verity_hash_for_block -> verity_verify_level -> dm_bufio_read_with_ioprio -> new_read -> __bufio_new -> alloc_buffer gfp: GFP_NOIO | __GFP_NORETRY | __GFP_NOMEMALLOC | __GFP_NOWARN Such a reclaimer can land on a memcg with a large, unswapped anon LRU and a tiny file LRU (e.g. inactive_anon ~335 MB vs inactive_file ~4 MB, with negligible swapcache). shrink_lruvec() then keeps feeding that huge anon list into shrink_folio_list() - ~2400 shrink_folio_list() calls, ~93,000 anon folios scanned - where every folio is kept because it needs IO. The 150+ ms above is one such single shrink_lruvec() pass (not accumulated across a reclaim cycle), and it reclaims nothing; the actual progress comes entirely from the file side. Aging anon alongside file does have some value for a later __GFP_IO reclaimer, so it is not strictly pure overhead. But that aging is only deferred, not lost: kswapd and other __GFP_IO reclaimers still walk and age anon. Spending ~168 ms aging memory that this context cannot reclaim is not a worthwhile trade-off in a latency-sensitive path. To stay conservative, this only skips anon when the swapcache is really tiny - below 1/64 of the anon LRU - i.e. when essentially no anon on the list can be reclaimed without IO. Whenever there is a meaningful amount of swapcached anon, the normal path is used and anon is scanned and aged as before. Note this only addresses the traditional active/inactive LRU. MGLRU selects anon vs file scanning in its own path and is not covered here; fixing the MGLRU case is left as a TODO. Link: https://lore.kernel.org/20260908062649.1045883-1-zhangbo56@xiaomi.com Signed-off-by: Bo Zhang Signed-off-by: Andrew Morton Reviewed-by: Barry Song Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt --- mm/vmscan.c | 62 ++++++++++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 57 insertions(+), 5 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 4059d130078960..8cbf562e625203 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -362,20 +362,72 @@ static bool can_demote(int nid, struct scan_control *sc, return !nodes_empty(allowed_mask); } +#ifdef CONFIG_SWAP +static inline bool reclaimable_anon_is_low(struct mem_cgroup *memcg, + int nid, struct scan_control *sc) +{ + pg_data_t *pgdat = NODE_DATA(nid); + unsigned long anon_pages, swapcache; + + /* + * A GFP_NOIO reclaimer can only reclaim anon that is already in the + * swapcache (adding anon to the swapcache needs IO). When swapcache is + * far below the anon LRU, scanning anon reclaims nothing and only burns + * CPU. The 1/64 threshold keeps this to the case where anon is + * effectively unreclaimable. + */ + if (!sc || (sc->gfp_mask & __GFP_IO)) + return false; + + /* + * FIXME: MGLRU doesn't fully respect can_reclaim_anon_pages() for the + * scanning type, so only apply this to the traditional LRU for now. + */ + if (lru_gen_enabled()) + return false; + + if (memcg) { + struct lruvec *lruvec = mem_cgroup_lruvec(memcg, pgdat); + + anon_pages = lruvec_page_state(lruvec, NR_INACTIVE_ANON) + + lruvec_page_state(lruvec, NR_ACTIVE_ANON); + swapcache = lruvec_page_state(lruvec, NR_SWAPCACHE); + } else { + anon_pages = node_page_state(pgdat, NR_INACTIVE_ANON) + + node_page_state(pgdat, NR_ACTIVE_ANON); + swapcache = node_page_state(pgdat, NR_SWAPCACHE); + } + + return swapcache < (anon_pages >> 6); +} +#else +static inline bool reclaimable_anon_is_low(struct mem_cgroup *memcg, + int nid, struct scan_control *sc) +{ + return true; +} +#endif /* CONFIG_SWAP */ + static inline bool can_reclaim_anon_pages(struct mem_cgroup *memcg, int nid, struct scan_control *sc) { if (memcg == NULL) { /* - * For non-memcg reclaim, is there - * space in any swap device? + * For non-memcg reclaim, is there space in any swap device? + * And under GFP_NOIO, is there enough swapcached anon to make + * scanning anon worthwhile? */ - if (get_nr_swap_pages() > 0) + if (get_nr_swap_pages() > 0 && + !reclaimable_anon_is_low(memcg, nid, sc)) return true; } else { - /* Is the memcg below its swap limit? */ - if (mem_cgroup_get_nr_swap_pages(memcg) > 0) + /* + * Is the memcg below its swap limit, and under GFP_NOIO does + * it have enough swapcached anon to make scanning worthwhile? + */ + if (mem_cgroup_get_nr_swap_pages(memcg) > 0 && + !reclaimable_anon_is_low(memcg, nid, sc)) return true; } From 566bf5129dc47dab4a9bc0911edc7133c6432989 Mon Sep 17 00:00:00 2001 From: Baolin Wang Date: Mon, 14 Sep 2026 19:29:22 +0800 Subject: [PATCH 0525/1012] mm: mglru: clear the reference counter for rejected folios As per the comment on LRU_REFS_FLAGS, when accessed folios are promoted to a new generation, LRU_REFS_FLAGS should be cleared so that the reference counter can start over. For folios rejected by shrink_folio_list(), we clear LRU_REFS_FLAGS and set the PG_active flag when lru_gen_folio_seq() would place them in the oldest generation. That's fine. But for rejected folios where lru_gen_folio_seq() returns a generation other than the oldest one (which can be treated as a promotion), we do not clear LRU_REFS_FLAGS. This can violate the promotion mechanism. And this means the rejected folio enters the new generation with stale, inflated tier bits, which can inflate reference counts and distort eviction statistics for these rejected folios. Fix this by clearing LRU_REFS_FLAGS for rejected folios. Of course, I need to evaluate the impact of the changes, which mainly falls into 3 cases: 1. When lru_gen_folio_seq() returns the oldest generation for rejected folios, there are no logic changes, and they will be put back into the 2nd youngest generation. 2. For rejected folios with PG_active set by shrink_folio_list(), we only clear the LRU_REFS_FLAGS and do not change the generation. 3. For rejected folios with PG_referenced set, the original code would put them back into the 2nd oldest generation. After this patch, we will put them back into the 2nd youngest generation. I think case 3 is also reasonable, before commmit 6cbdd9726fb5 ("mm/mglru: use folio_mark_accessed to replace folio_set_active"), a rejected referenced folio was also put back to the 2nd youngest gen. Meanwhile, I didn't see any noticeable performance impact on my 32-core Arm machine when running 'make -j32' to build the kernel inside a 3G-limited memcg with either zram or NVMe swap. Link: https://lore.kernel.org/7384df363c12e4acdaa2e0428420cd8eed320ee7.1789384831.git.baolin.wang@linux.alibaba.com Signed-off-by: Baolin Wang Signed-off-by: Andrew Morton Reviewed-by: Baoquan He Reviewed-by: Kairui Song Reviewed-by: Barry Song Cc: Axel Rasmussen Cc: David Hildenbrand Cc: Johannes Weiner Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie --- mm/vmscan.c | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 8cbf562e625203..a49da82ed4797a 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -5075,11 +5075,15 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec, continue; } - /* don't add rejected folios to the oldest generation */ - if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) { - folio_set_lru_refs(folio, 0); + /* + * See the comments on LRU_REFS_FLAGS. + * + * The rejected folios are never added to the oldest generation, + * so this effectively promotes them by at least one generation. + */ + folio_set_lru_refs(folio, 0); + if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) folio_set_active(folio); - } } move_folios_to_lru(&list); From a82d3d85c508838c7f0d731aa1d08c28a93f09e5 Mon Sep 17 00:00:00 2001 From: Anastasios Papagiannis Date: Wed, 9 Sep 2026 09:42:31 +0300 Subject: [PATCH 0526/1012] mm/nommu: reject wrapping ranges in access_remote_vm() The NOMMU implementation of access_process_vm() rejects address ranges whose end wraps around, but access_remote_vm() bypasses this check even though both functions delegate to __access_remote_vm(). Move the wraparound check into __access_remote_vm() so it applies to both entry points. This is originally reported in [1]. Link: https://lore.kernel.org/20260909064231.18693-1-tasos.papagiannnis@gmail.com Fixes: f55f199b7d76 ("NOMMU: implement access_remote_vm") Signed-off-by: Anastasios Papagiannis Signed-off-by: Andrew Morton Closes: https://lore.kernel.org/bpf/4ef240a5bea36ff84df9589671367832860795159386a4c8fba546a0fa8b786f@mail.kernel.org/ [1] Reviewed-by: Lorenzo Stoakes (ARM) Cc: Liam R. Howlett Cc: Hajime Tazaki Cc: --- mm/nommu.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/nommu.c b/mm/nommu.c index 498e01ee40b056..ed44510e37707d 100644 --- a/mm/nommu.c +++ b/mm/nommu.c @@ -1674,6 +1674,9 @@ static int __access_remote_vm(struct mm_struct *mm, unsigned long addr, struct vm_area_struct *vma; int write = gup_flags & FOLL_WRITE; + if (addr + len < addr) + return 0; + if (mmap_read_lock_killable(mm)) return 0; @@ -1727,9 +1730,6 @@ int access_process_vm(struct task_struct *tsk, unsigned long addr, void *buf, in { struct mm_struct *mm; - if (addr + len < addr) - return 0; - mm = get_task_mm(tsk); if (!mm) return 0; From a50e56f91d34675b73ca65deebdfa8ed60cbfb83 Mon Sep 17 00:00:00 2001 From: Youngjun Park Date: Thu, 10 Sep 2026 01:15:51 +0900 Subject: [PATCH 0527/1012] mm/swap: fix stale comment on swap_info_struct::cluster_info Patch series "mm/swap: skip empty clusters in the swapoff scan", v4. Speed up swapoff and reduce scanning stalls on large, mostly empty swap devices by skipping empty clusters. Reduce swapoff time on a 1 TiB device from 158 ms to 94 ms after filling 128 GiB, and from 392 ms to 73 ms after filling 512 GiB; expect little benefit when substantial swap data remains. find_next_to_unuse() walks a swap device one offset at a time. Slot state now lives in a per cluster swap table, so patch 2 dismisses an empty cluster with one counter read instead of SWAPFILE_CLUSTER table reads. Patch 1 is an unrelated one line comment fix noticed on the way. A debug test confirmed the skip path runs, and swapoff completed under load with no DEBUG_VM or lockdep splats. This patch (of 2): setup_swap_clusters_info() allocates cluster_info for every swap area, not only for SSDs. Link: https://lore.kernel.org/20260909161552.2335971-1-youngjun.park@lge.com Link: https://lore.kernel.org/20260909161552.2335971-2-youngjun.park@lge.com Signed-off-by: Youngjun Park Signed-off-by: Andrew Morton Acked-by: Kairui Song Reviewed-by: Barry Song Cc: Baoquan He Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham --- include/linux/swap.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index a37ad8375e011e..61005501888c53 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -240,7 +240,7 @@ struct swap_info_struct { struct plist_node list; /* entry in swap_active_head */ signed char type; /* strange name for an index */ unsigned int max; /* size of this swap device */ - struct swap_cluster_info *cluster_info; /* cluster info. Only for SSD */ + struct swap_cluster_info *cluster_info; /* array, one entry per cluster */ struct list_head free_clusters; /* free clusters list */ struct list_head full_clusters; /* full clusters list */ struct list_head nonfull_clusters[SWAP_NR_ORDERS]; From 820c8bff56e8364004ecae3d5e6148009bdae762 Mon Sep 17 00:00:00 2001 From: Youngjun Park Date: Thu, 10 Sep 2026 01:15:52 +0900 Subject: [PATCH 0528/1012] mm/swap: scan by cluster in find_next_to_unuse() find_next_to_unuse() walks every offset from 0 to si->max, and swapoff restarts that walk on each retry, so the cost scales with the size of the device rather than with the few slots the shmem and mmlist passes could not free. It has caused stalls before. The flat walk predates the swap table. Slot state now lives in a per cluster table, and wait_for_allocation() stops all allocation before try_to_unuse() runs, so a cluster that holds no slot in use stays that way. Skip such a cluster instead of reading all of its entries. Fill a 1 TiB swap up to some amount, then swapoff. What is left sits at the top of what was filled, so every slot below it is free. Medians over 11 pairs at 32 and 128 GiB, 3 pairs at 256 and 512. filled swapoff old new 32 GiB 92.4ms 66.3ms 128 GiB 157.6ms 94.4ms 256 GiB 209.7ms 63.8ms 512 GiB 391.8ms 73.4ms old grows with how much was filled, new does not. In the ordinary case swap still holds real data and swapoff spends its time reading it back. it is tested 4 GiB on an 8 GiB device, where the scan is 1.4% of try_to_unuse(), and there is no difference either way. Commit dc644a073769 ("mm: add three more cond_resched() in swapoff") answered those stalls with a cond_resched() every 256 offsets. A walk bounded by one cluster no longer needs that counter. The loop now runs at most SWAPFILE_CLUSTER times before it returns or reschedules, the same bound swap_reclaim_full_clusters() already scans between cond_resched() calls. The scan end is clamped to si->max, so the walk stops there rather than running into the masked tail of the last cluster. ci->count is read without ci->lock, so READ_ONCE() marks the read for KCSAN. Allocation is already stopped, so the count can only drop, and a slot stops being counted only after its folio has left the swap cache. An empty cluster therefore holds nothing for try_to_unuse() to act on. Link: https://lore.kernel.org/20260909161552.2335971-3-youngjun.park@lge.com Signed-off-by: Youngjun Park Signed-off-by: Andrew Morton Reviewed-by: Barry Song Acked-by: Kairui Song Reviewed-by: Baoquan He Reviewed-by: Nhat Pham Cc: Chris Li Cc: Kemeng Shi --- mm/swapfile.c | 43 ++++++++++++++++++++++++++++++------------- 1 file changed, 30 insertions(+), 13 deletions(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 48d3cd40defdb4..a8118f095f4d9a 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -370,8 +370,6 @@ static void discard_swap_cluster(struct swap_info_struct *si, } } -#define LATENCY_LIMIT 256 - static inline bool cluster_is_empty(struct swap_cluster_info *info) { return info->count == 0; @@ -2726,7 +2724,9 @@ static int unuse_mm(struct mm_struct *mm, unsigned int type) static unsigned int find_next_to_unuse(struct swap_info_struct *si, unsigned int prev) { - unsigned int i; + struct swap_cluster_info *ci; + unsigned long i, end; + unsigned int ci_off; unsigned long swp_tb; /* @@ -2735,19 +2735,36 @@ static unsigned int find_next_to_unuse(struct swap_info_struct *si, * hits are okay, and sys_swapoff() has already prevented new * allocations from this area (while holding swap_lock). */ - for (i = prev + 1; i < si->max; i++) { - swp_tb = swap_table_get(__swap_offset_to_cluster(si, i), - i % SWAPFILE_CLUSTER); - if (!swp_tb_is_null(swp_tb) && !swp_tb_is_bad(swp_tb)) - break; - if ((i % LATENCY_LIMIT) == 0) + i = prev + 1; + while (i < si->max) { + ci = __swap_offset_to_cluster(si, i); + end = min_t(unsigned long, + ALIGN_DOWN(i, SWAPFILE_CLUSTER) + SWAPFILE_CLUSTER, + si->max); + + /* + * An empty cluster has no slot in use, so skip it whole. + * A slot is uncounted only after its folio left the swap + * cache, so there is nothing here for try_to_unuse() to act on. + * Count only drops here, so a READ_ONCE() without ci->lock is + * enough, unlike in every other cluster_is_empty() caller. + */ + if (!READ_ONCE(ci->count)) { + i = end; cond_resched(); - } + continue; + } - if (i == si->max) - i = 0; + ci_off = i % SWAPFILE_CLUSTER; + for (; i < end; ci_off++, i++) { + swp_tb = swap_table_get(ci, ci_off); + if (!swp_tb_is_null(swp_tb) && !swp_tb_is_bad(swp_tb)) + return i; + } + cond_resched(); + } - return i; + return 0; } static int try_to_unuse(unsigned int type) From 606a2da05824c835d70d9fb2a086e15e793db64d Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 9 Sep 2026 07:04:03 -0700 Subject: [PATCH 0529/1012] mm/damon/vaddr: support prep_probes Patch series "mm/damon/vaddr: support {prep,apply}_probes". DAMON supports data attributes monitoring. However, only the physical address space operation set (paddr) is supporting it. Add the support to the virtual address space operation set (vaddr). Patch 1 adds prep_probes support to vaddr. Patch 2 moves probe filter handling code in paddr.c that can be reused by vaddr to ops-common.c. Patch 3 adds minimum apply_probes support to vaddr. Patch 4 extends the support for hugetlb. Patch 5 extends the support for pgidle_unset filter. Test ==== I confirmed it can capture ~48 mb working set of masim in vaddr mode, like below. First, start masim [1] to access ~48 mb memory at a time, in the background. $ ./masim/masim.py run \ --config_file ./masim/configs/stairs-50mb.cfg \ --repeat 10 --quiet & Note that the config says the working set is 50mb. It is 50 million bytes, so ~48 MiB. Start traditional access monitoring of masim's virtual address space using damo [2]. $ sudo ./damo/damo start $(pidof masim) Confirm it can capture the ~48 MiB working set as the 4-th region on the snapshot. $ sudo ./damo/damo report access heatmap: 11111111334[...]3000000000000000000000000000000000000000489999997777777743333333555556[...]8 # min/max temperatures: -1,150,000,000, 80,009,499, column size: 6.963 MiB intervals: sample 5 ms aggr 100 ms (max access hz 200) 0 addr 85.355 TiB size 55.703 MiB access 0 hz age 9.500 s 1 addr 85.355 TiB size 18.984 MiB access 0 hz age 6.700 s 2 addr 127.183 TiB size 278.516 MiB access 0 hz age 11.500 s 3 addr 127.183 TiB size 7.570 MiB access 0 hz age 900 ms 4 addr 127.183 TiB size 48.133 MiB access 190 hz age 800 ms 5 addr 127.183 TiB size 54.977 MiB access 0 hz age 1.700 s 6 addr 127.183 TiB size 55.113 MiB access 0 hz age 6.700 s 7 addr 127.183 TiB size 37.902 MiB access 0 hz age 3.900 s 8 addr 127.990 TiB size 120.000 KiB access 0 hz age 11.400 s 9 addr 127.990 TiB size 8.000 KiB access 70 hz age 0 ns 10 addr 127.990 TiB size 4.000 KiB access 0 hz age 11.200 s memory bw estimate: 8.931 GiB per second total size: 557.027 MiB record DAMON intervals: sample 5 ms, aggr 100 ms Stop access monitoring and start probe-only mode access monitoring. $ sudo ./damo/damo stop $ sudo ./damo/damo start $(pidof masim) --probe_prep set_pgidle \ --probe_filter allow pgidle_unset --probe_weight 1 Confirm it can also capture the ~48 MiB working set as the 11-th region on the snapshot. $ sudo ./damo/damo report attrs heatmap: 00000000113[...]40000000000000000000000000000000000000000000000001111114999999533333336[...]6 # min/max temperatures: -840,000,000, 330,002,000, column size: 6.961 MiB probe prep: set_pgidle, filter: allow pgidle_unset (weight: 1) intervals: sample 5 ms aggr 100 ms (max probe hits 20) # size address age probe_hits 0 8.000 KiB 127.990 TiB 8.700 s 0 1 120.000 KiB 127.990 TiB 8.600 s 0 2 55.543 MiB 127.183 TiB 8.400 s 0 3 110.008 MiB 127.183 TiB 8.300 s 0 4 55.352 MiB 127.183 TiB 8.100 s 0 5 55.605 MiB 127.183 TiB 8 s 0 6 55.691 MiB 85.355 TiB 7.900 s 0 7 54.430 MiB 127.183 TiB 7.900 s 0 8 50.391 MiB 127.183 TiB 6.900 s 0 9 18.879 MiB 85.355 TiB 6 s 0 10 52.844 MiB 127.183 TiB 3.500 s 0 11 48.039 MiB 127.183 TiB 3.300 s 20 12 4.000 KiB 127.990 TiB 8.500 s 20 memory bw estimate: 0 B per second total size: 556.910 MiB record DAMON intervals: sample 5 ms, aggr 100 ms This patch (of 5): DAMON virtual address space operation set (vaddr) is not supporting prep_probes. Add the support. Link: https://lore.kernel.org/20260909140408.104699-2-sj@kernel.org Link: https://github.com/sjp38/masim [1] Link: https://github.com/damonitor/damo [2] Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Gutierrez Asier --- mm/damon/vaddr.c | 40 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 40 insertions(+) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index af9e1b82454cc2..20f4784f178173 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -483,6 +483,45 @@ static unsigned int damon_va_check_accesses(struct damon_ctx *ctx) return max_nr_accesses; } +static void damon_va_prep_probe_region(struct damon_ctx *ctx, + struct mm_struct *mm, struct damon_region *r, + struct damon_probe *probe) +{ + struct damon_prep *p; + + damon_for_each_prep(p, probe) { + switch (p->action) { + case DAMON_PREP_SET_PGIDLE: + damon_va_mkold(mm, r->sampling_addr); + break; + default: + break; + } + } +} + +static void damon_va_prep_probes(struct damon_ctx *ctx, bool set_samples) +{ + struct damon_target *t; + struct mm_struct *mm; + struct damon_region *r; + struct damon_probe *p; + + damon_for_each_target(t, ctx) { + mm = damon_get_mm(t); + if (!mm) + continue; + damon_for_each_region(r, t) { + if (set_samples) + r->sampling_addr = damon_rand(ctx, r->ar.start, + r->ar.end); + damon_for_each_probe(p, ctx) + damon_va_prep_probe_region(ctx, mm, r, p); + } + mmput(mm); + } +} + static bool damos_va_filter_young_match(struct damos_filter *filter, struct folio *folio, struct vm_area_struct *vma, unsigned long addr, pte_t *ptep, pmd_t *pmdp) @@ -911,6 +950,7 @@ static int __init damon_va_initcall(void) .update = damon_va_update, .prepare_access_checks = damon_va_prepare_access_checks, .check_accesses = damon_va_check_accesses, + .prep_probes = damon_va_prep_probes, .target_valid = damon_va_target_valid, .cleanup_target = damon_va_cleanup_target, .apply_scheme = damon_va_apply_scheme, From 27c9e03997d392e66d175943f2b26d6da1cac303 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 9 Sep 2026 07:04:04 -0700 Subject: [PATCH 0530/1012] mm/damon/paddr: move probe filter handling to ops-common Probe filter matching logic for anon and memcg type filters in DAMON's physical address space operation set (paddr) can be reused by virtual address space operation set (vaddr) in future. Prepare the future by moving the code to ops-common.c Link: https://lore.kernel.org/20260909140408.104699-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Gutierrez Asier --- mm/damon/ops-common.c | 32 ++++++++++++++++++++++++++++++++ mm/damon/ops-common.h | 2 ++ mm/damon/paddr.c | 23 +---------------------- 3 files changed, 35 insertions(+), 22 deletions(-) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index acf8f216c51cc8..c36cc39cd2c707 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -531,3 +531,35 @@ bool damos_ops_has_filter(struct damos *s) return true; return false; } + +bool damon_ops_filter_match(struct damon_filter *filter, struct folio *folio) +{ + bool matched = false; + struct mem_cgroup *memcg; + + switch (filter->type) { + case DAMON_FILTER_TYPE_ANON: + if (!folio) { + matched = false; + break; + } + matched = folio_test_anon(folio); + break; + case DAMON_FILTER_TYPE_MEMCG: + if (!folio) { + matched = false; + break; + } + rcu_read_lock(); + memcg = folio_memcg_check(folio); + if (!memcg) + matched = false; + else + matched = filter->memcg_id == mem_cgroup_id(memcg); + rcu_read_unlock(); + break; + default: + break; + } + return matched == filter->matching; +} diff --git a/mm/damon/ops-common.h b/mm/damon/ops-common.h index 172f0f17c4a84d..ee27058251005b 100644 --- a/mm/damon/ops-common.h +++ b/mm/damon/ops-common.h @@ -31,3 +31,5 @@ bool damos_folio_filter_match(struct damos_filter *filter, struct folio *folio); unsigned long damon_migrate_pages(struct list_head *folio_list, int target_nid); bool damos_ops_has_filter(struct damos *s); + +bool damon_ops_filter_match(struct damon_filter *filter, struct folio *folio); diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index d2173a448d0b0c..d7c81829445ba1 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -143,29 +143,8 @@ static bool damon_pa_filter_match(struct damon_filter *filter, struct folio *folio) { bool matched = false; - struct mem_cgroup *memcg; switch (filter->type) { - case DAMON_FILTER_TYPE_ANON: - if (!folio) { - matched = false; - break; - } - matched = folio_test_anon(folio); - break; - case DAMON_FILTER_TYPE_MEMCG: - if (!folio) { - matched = false; - break; - } - rcu_read_lock(); - memcg = folio_memcg_check(folio); - if (!memcg) - matched = false; - else - matched = filter->memcg_id == mem_cgroup_id(memcg); - rcu_read_unlock(); - break; case DAMON_FILTER_TYPE_PGIDLE_UNSET: if (!folio) matched = false; @@ -173,7 +152,7 @@ static bool damon_pa_filter_match(struct damon_filter *filter, matched = damon_folio_young(folio); break; default: - break; + return damon_ops_filter_match(filter, folio); } return matched == filter->matching; } From b8a67f78e629e75d3ee2f418228d1685616c8397 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 9 Sep 2026 07:04:05 -0700 Subject: [PATCH 0531/1012] mm/damon/vaddr: support apply_probe DAMON virtual address space operation set (vaddr) is not supporting apply_probe. Add a minimum support. Do not support hugetlb pages and PGIDLE_UNSET filter type for simplicity. Link: https://lore.kernel.org/20260909140408.104699-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Gutierrez Asier --- mm/damon/vaddr.c | 123 +++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 123 insertions(+) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 20f4784f178173..7c548ec0cf6b5b 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -522,6 +522,128 @@ static void damon_va_prep_probes(struct damon_ctx *ctx, bool set_samples) } } +static bool damon_va_filter_pass(struct folio *folio, struct damon_probe *p) +{ + struct damon_filter *f; + bool pass = true; + + damon_for_each_filter(f, p) { + if (damon_ops_filter_match(f, folio)) { + pass = f->allow; + break; + } + pass = !f->allow; + } + return pass; +} + +struct damon_va_probe_walk_private { + struct damon_ctx *ctx; + struct damon_region *r; +}; + +static void damon_va_probe_folio(struct damon_ctx *ctx, + struct damon_region *r, struct folio *folio) +{ + struct damon_probe *probe; + int i = 0; + + damon_for_each_probe(probe, ctx) { + if (damon_va_filter_pass(folio, probe)) + r->probe_hits[i]++; + i++; + } +} + +static int damon_va_probe_pmd_entry(pmd_t *pmd, unsigned long addr, + unsigned long next, struct mm_walk *walk) +{ + pte_t *pte; + pte_t ptent; + spinlock_t *ptl; + struct folio *folio; + struct damon_va_probe_walk_private *priv = walk->private; + +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + ptl = pmd_trans_huge_lock(pmd, walk->vma); + if (ptl) { + pmd_t pmde = pmdp_get(pmd); + + if (!pmd_present(pmde)) + goto huge_out; + folio = vm_normal_folio_pmd(walk->vma, addr, pmde); + if (!folio) + goto huge_out; + damon_va_probe_folio(priv->ctx, priv->r, folio); + +huge_out: + spin_unlock(ptl); + return 0; + } +#endif /* CONFIG_TRANSPARENT_HUGEPAGE */ + + pte = pte_offset_map_lock(walk->mm, pmd, addr, &ptl); + if (!pte) + return 0; + ptent = ptep_get(pte); + if (!pte_present(ptent)) + goto out; + folio = vm_normal_folio(walk->vma, addr, ptent); + if (!folio) + goto out; + damon_va_probe_folio(priv->ctx, priv->r, folio); + +out: + pte_unmap_unlock(pte, ptl); + return 0; +} + +static void __damon_va_apply_probes(struct damon_ctx *ctx, + struct mm_struct *mm, struct damon_region *r) +{ + struct damon_va_probe_walk_private arg = { + .ctx = ctx, + .r = r, + }; + struct mm_walk_ops damon_probe_walk_ops = { + .pmd_entry = damon_va_probe_pmd_entry, + .hugetlb_entry = NULL, + }; + unsigned long addr = r->sampling_addr; + + if (!mm) + return; + + damon_va_walk_page_range(mm, addr, addr + 1, &damon_probe_walk_ops, + &arg); +} + +static unsigned int damon_va_apply_probes(struct damon_ctx *ctx, + bool set_samples, bool return_max_wsum) +{ + struct damon_target *t; + struct mm_struct *mm; + struct damon_region *r; + unsigned int max_wsum = 0; + + damon_for_each_target(t, ctx) { + mm = damon_get_mm(t); + damon_for_each_region(r, t) { + if (set_samples) + r->sampling_addr = damon_rand(ctx, r->ar.start, + r->ar.end); + __damon_va_apply_probes(ctx, mm, r); + if (return_max_wsum) + max_wsum = max(damon_probe_hits_wsum(r, false, + ctx), max_wsum); + } + if (mm) + mmput(mm); + } + + return max_wsum; +} + static bool damos_va_filter_young_match(struct damos_filter *filter, struct folio *folio, struct vm_area_struct *vma, unsigned long addr, pte_t *ptep, pmd_t *pmdp) @@ -951,6 +1073,7 @@ static int __init damon_va_initcall(void) .prepare_access_checks = damon_va_prepare_access_checks, .check_accesses = damon_va_check_accesses, .prep_probes = damon_va_prep_probes, + .apply_probes = damon_va_apply_probes, .target_valid = damon_va_target_valid, .cleanup_target = damon_va_cleanup_target, .apply_scheme = damon_va_apply_scheme, From 972bcf07e81991bbc6bd7df4e3787939571cb8f0 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 9 Sep 2026 07:04:06 -0700 Subject: [PATCH 0532/1012] mm/damon/vaddr: extend apply_probes() for hugetlb DAMON virtual address space operation set(vaddr) does not support hugetlb pages in apply_probes. Extend it for hugetlb pages. Link: https://lore.kernel.org/20260909140408.104699-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Gutierrez Asier --- mm/damon/vaddr.c | 30 +++++++++++++++++++++++++++++- 1 file changed, 29 insertions(+), 1 deletion(-) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 7c548ec0cf6b5b..45239f05e113b2 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -598,6 +598,34 @@ static int damon_va_probe_pmd_entry(pmd_t *pmd, unsigned long addr, return 0; } +#ifdef CONFIG_HUGETLB_PAGE +static int damon_va_probe_hugetlb_entry(pte_t *pte, unsigned long hmask, + unsigned long addr, unsigned long end, struct mm_walk *walk) +{ + struct damon_va_probe_walk_private *priv = walk->private; + struct hstate *h = hstate_vma(walk->vma); + struct folio *folio; + spinlock_t *ptl; + pte_t entry; + + ptl = huge_pte_lock(h, walk->mm, pte); + entry = huge_ptep_get(walk->mm, addr, pte); + if (!pte_present(entry)) + goto out; + + folio = pfn_folio(pte_pfn(entry)); + folio_get(folio); + damon_va_probe_folio(priv->ctx, priv->r, folio); + folio_put(folio); + +out: + spin_unlock(ptl); + return 0; +} +#else +#define damon_va_probe_hugetlb_entry NULL +#endif /* CONFIG_HUGETLB_PAGE */ + static void __damon_va_apply_probes(struct damon_ctx *ctx, struct mm_struct *mm, struct damon_region *r) { @@ -607,7 +635,7 @@ static void __damon_va_apply_probes(struct damon_ctx *ctx, }; struct mm_walk_ops damon_probe_walk_ops = { .pmd_entry = damon_va_probe_pmd_entry, - .hugetlb_entry = NULL, + .hugetlb_entry = damon_va_probe_hugetlb_entry, }; unsigned long addr = r->sampling_addr; From 0c780b4338190a804bc62db1adb2890ac1340fd2 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Wed, 9 Sep 2026 07:04:07 -0700 Subject: [PATCH 0533/1012] mm/damon/vaddr: support pgidle_unset probe filter type DAMON virtual address space operation set (vaddr) does not support pgidle_unset probe filter type. Add the support. Link: https://lore.kernel.org/20260909140408.104699-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Gutierrez Asier --- mm/damon/vaddr.c | 55 ++++++++++++++++++++++++++++++++++++++++++------ 1 file changed, 48 insertions(+), 7 deletions(-) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 45239f05e113b2..9a38dc89a156ef 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -522,13 +522,49 @@ static void damon_va_prep_probes(struct damon_ctx *ctx, bool set_samples) } } -static bool damon_va_filter_pass(struct folio *folio, struct damon_probe *p) +static bool damon_va_young_addr(struct folio *folio, pte_t *pte, pmd_t *pmd, + struct mm_struct *mm, unsigned long addr) +{ + bool young = false; + + if (pte) + young = pte_young(*pte); + else if (pmd) + young = pmd_young(*pmd); + young = young || !folio_test_idle(folio) || + mmu_notifier_test_young(mm, addr); + return young; +} + +static bool damon_va_filter_match(struct damon_filter *filter, + struct folio *folio, pte_t *pte, pmd_t *pmd, + struct mm_struct *mm, unsigned long addr) +{ + bool matched = false; + + switch (filter->type) { + case DAMON_FILTER_TYPE_PGIDLE_UNSET: + if (!folio) + matched = false; + else + matched = damon_va_young_addr(folio, pte, pmd, mm, + addr); + break; + default: + return damon_ops_filter_match(filter, folio); + } + return matched == filter->matching; +} + +static bool damon_va_filter_pass(struct folio *folio, struct damon_probe *p, + pte_t *pte, pmd_t *pmd, struct mm_struct *mm, + unsigned long addr) { struct damon_filter *f; bool pass = true; damon_for_each_filter(f, p) { - if (damon_ops_filter_match(f, folio)) { + if (damon_va_filter_match(f, folio, pte, pmd, mm, addr)) { pass = f->allow; break; } @@ -543,13 +579,15 @@ struct damon_va_probe_walk_private { }; static void damon_va_probe_folio(struct damon_ctx *ctx, - struct damon_region *r, struct folio *folio) + struct damon_region *r, struct folio *folio, + pte_t *pte, pmd_t *pmd, struct mm_struct *mm) { struct damon_probe *probe; int i = 0; damon_for_each_probe(probe, ctx) { - if (damon_va_filter_pass(folio, probe)) + if (damon_va_filter_pass(folio, probe, pte, pmd, mm, + r->sampling_addr)) r->probe_hits[i]++; i++; } @@ -574,7 +612,8 @@ static int damon_va_probe_pmd_entry(pmd_t *pmd, unsigned long addr, folio = vm_normal_folio_pmd(walk->vma, addr, pmde); if (!folio) goto huge_out; - damon_va_probe_folio(priv->ctx, priv->r, folio); + damon_va_probe_folio(priv->ctx, priv->r, folio, NULL, &pmde, + walk->vma->vm_mm); huge_out: spin_unlock(ptl); @@ -591,7 +630,8 @@ static int damon_va_probe_pmd_entry(pmd_t *pmd, unsigned long addr, folio = vm_normal_folio(walk->vma, addr, ptent); if (!folio) goto out; - damon_va_probe_folio(priv->ctx, priv->r, folio); + damon_va_probe_folio(priv->ctx, priv->r, folio, &ptent, NULL, + walk->vma->vm_mm); out: pte_unmap_unlock(pte, ptl); @@ -615,7 +655,8 @@ static int damon_va_probe_hugetlb_entry(pte_t *pte, unsigned long hmask, folio = pfn_folio(pte_pfn(entry)); folio_get(folio); - damon_va_probe_folio(priv->ctx, priv->r, folio); + damon_va_probe_folio(priv->ctx, priv->r, folio, &entry, NULL, + walk->vma->vm_mm); folio_put(folio); out: From e6369ca57196635a18724b09a29073b0df2c19c9 Mon Sep 17 00:00:00 2001 From: Usama Arif Date: Mon, 7 Sep 2026 09:19:38 -0700 Subject: [PATCH 0534/1012] mm: zswap: don't fail a large-folio swapin whose range is not in zswap thp_swapin_suitable_orders() and shmem_swap_alloc_folio() sample zswap_never_enabled() to decide whether a swapin may use a large folio. zswap_load() samples the same one-way static key again once the read reaches it. Nothing serialises the two reads, and in between the task allocates and pins a high-order folio, which can sleep. If zswap is enabled for the first time in that window, a large folio that was correctly permitted reaches zswap_load(), which rejects every large folio with -EINVAL. swap_read_folio() treats anything other than -ENOENT as "zswap handled it" and skips the backing-device read, so the folio comes back unlocked and not uptodate: SIGBUS for an anonymous fault, -EIO for shmem. The data is intact on the swap device - it was written there before zswap was ever enabled - and the not-uptodate folio stays in the swap cache, so every retry of the fault fails the same way. With panic_on_warn the WARN takes the machine down rather than the task. Scan the range instead of rejecting the folio. The caller has pinned every slot before issuing the read, so zswap cannot start a store or a writeback into the range and the scan is stable. If nothing in the range is in zswap it is all on the backing device: return -ENOENT and let swap_read_folio() read it. A range that does have a slot in zswap is still refused, because zswap stores large folios as order-0 entries and cannot reconstruct one. That stays reachable - a slot shared with another task can be stored inside the same window - and refusing is correct, since the alternative is returning the stale device copy. Report it as -EIO rather than -EINVAL: the request is valid, zswap just cannot serve it. The only caller distinguishes -ENOENT from everything else, so that part is a documentation fix. Link: https://lore.kernel.org/20260907161938.1932355-1-usama.arif@linux.dev Fixes: 242d12c98174 ("mm: support large folios swap-in for sync io devices") Co-developed-by: Alexandre Ghiti Signed-off-by: Alexandre Ghiti Signed-off-by: Usama Arif Signed-off-by: Andrew Morton Acked-by: Yosry Ahmed Acked-by: Nhat Pham Cc: Chengming Zhou Cc: Johannes Weiner --- mm/zswap.c | 58 ++++++++++++++++++++++++++++++++++++++++-------------- 1 file changed, 43 insertions(+), 15 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index 3a6f8901764641..b5411ab164809d 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1555,6 +1555,32 @@ bool zswap_store(struct folio *folio) return ret; } +/** + * zswap_is_present() - is any slot in [entry, entry + nr) in zswap? + * @entry: base swap entry of the range + * @nr: number of contiguous slots to check + * + * Context: The caller must keep the range pinned, otherwise the answer can + * change under it. + * Return: true if at least one slot in the range is in zswap. + */ +static bool zswap_is_present(swp_entry_t entry, unsigned int nr) +{ + pgoff_t offset = swp_offset(entry); + struct xarray *tree = swap_zswap_tree(entry); + unsigned long index = offset; + + /* + * A pinned range is at most SWAPFILE_CLUSTER slots and is aligned to + * its own size, so one tree covers all of it and a single lookup is + * enough. Scanning only part of the range would report a false + * "absent" and let the caller read a stale copy from the device. + */ + BUILD_BUG_ON(SWAPFILE_CLUSTER > ZSWAP_ADDRESS_SPACE_PAGES); + + return xa_find(tree, &index, offset + nr - 1, XA_PRESENT); +} + /** * zswap_load() - load a folio from zswap * @folio: folio to load @@ -1562,15 +1588,12 @@ bool zswap_store(struct folio *folio) * Return: 0 on success, with the folio unlocked and marked up-to-date, or one * of the following error codes: * - * -EIO: if the swapped out content was in zswap, but could not be loaded - * into the page due to a decompression failure. The folio is unlocked, but - * NOT marked up-to-date, so that an IO error is emitted (e.g. do_swap_page() - * will SIGBUS). - * - * -EINVAL: if the swapped out content was in zswap, but the page belongs - * to a large folio, which is not supported by zswap. The folio is unlocked, - * but NOT marked up-to-date, so that an IO error is emitted (e.g. - * do_swap_page() will SIGBUS). + * -EIO: if the swapped out content was in zswap but could not be handed + * back, either because decompression failed or because a slot in a + * large-folio range is still in zswap and zswap cannot reconstruct a large + * folio from per-page entries. The folio is unlocked, but NOT marked + * up-to-date, so that an IO error is emitted (e.g. do_swap_page() will + * SIGBUS). * * -ENOENT: if the swapped out content was not in zswap. The folio remains * locked on return. @@ -1589,13 +1612,18 @@ int zswap_load(struct folio *folio) return -ENOENT; /* - * Large folios should not be swapped in while zswap is being used, as - * they are not properly handled. Zswap does not properly load large - * folios, and a large folio may only be partially in zswap. + * A large folio can legitimately reach zswap_load() with its whole + * range on the backing device, so scan the range rather than rejecting + * it outright. The caller has pinned every slot, so zswap cannot start + * a store or a writeback into the range while we look. */ - if (WARN_ON_ONCE(folio_test_large(folio))) { - folio_unlock(folio); - return -EINVAL; + if (folio_test_large(folio)) { + if (WARN_ON_ONCE(zswap_is_present(swp, + folio_nr_pages(folio)))) { + folio_unlock(folio); + return -EIO; + } + return -ENOENT; } entry = xa_load(tree, offset); From 3cc85b77c7c62fde8d193d48ed1ea7f70ee5ed16 Mon Sep 17 00:00:00 2001 From: Jinmeng Zhou Date: Mon, 7 Sep 2026 21:20:55 +0800 Subject: [PATCH 0535/1012] mm/hugetlb: fix subpool minimum reservation rollback When a reservation request is partially covered by a subpool minimum and the remaining global reservation fails, the error path first calls hugepage_subpool_put_pages() for the subpool-backed portion. It removes the failed global portion from used_hpages only afterwards. hugepage_subpool_put_pages() uses used_hpages to decide whether rsv_hpages should be restored. Since used_hpages still includes the global portion, it can remain at or above min_hpages and prevent that restoration. It then reports the subpool reservation as releasable, causing hugetlb_acct_memory() to incorrectly decrement h->resv_huge_pages. This was reproduced with four 2 MB huge pages and a hugetlbfs mount with size=10M,min_size=8M. After a successful three-page reservation, a two-page reservation which needed one subpool page and one global page failed with -ENOMEM. HugePages_Rsvd incorrectly dropped from four to three even though the subpool minimum was still four pages. Roll back the failed global portion from used_hpages first, so that hugepage_subpool_put_pages() evaluates the minimum reservation against the current usage and returns the correct global adjustment. Link: https://lore.kernel.org/20260907132055.26696-1-zhoujinmeng@bytedance.com Fixes: a833a693a490 ("mm: hugetlb: fix incorrect fallback for subpool") Signed-off-by: Jinmeng Zhou Signed-off-by: Andrew Morton Acked-by: Muchun Song Tested-by: Ackerley Tng Cc: David Hildenbrand Cc: Ma Wupeng Cc: Oscar Salvador Cc: --- mm/hugetlb.c | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index afffaa3d2d3741..5e05711f352d86 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -6865,15 +6865,6 @@ long hugetlb_reserve_pages(struct inode *inode, out_put_pages: spool_resv = chg - gbl_reserve; - if (spool_resv) { - /* put sub pool's reservation back, chg - gbl_reserve */ - gbl_resv = hugepage_subpool_put_pages(spool, spool_resv); - /* - * subpool's reserved pages can not be put back due to race, - * return to hstate. - */ - hugetlb_acct_memory(h, -gbl_resv); - } /* Restore used_hpages for pages that failed global reservation */ if (gbl_reserve && spool) { unsigned long flags; @@ -6883,6 +6874,15 @@ long hugetlb_reserve_pages(struct inode *inode, spool->used_hpages -= gbl_reserve; unlock_or_release_subpool(spool, flags); } + if (spool_resv) { + /* put sub pool's reservation back, chg - gbl_reserve */ + gbl_resv = hugepage_subpool_put_pages(spool, spool_resv); + /* + * subpool's reserved pages can not be put back due to race, + * return to hstate. + */ + hugetlb_acct_memory(h, -gbl_resv); + } out_uncharge_cgroup: hugetlb_cgroup_uncharge_cgroup_rsvd(hstate_index(h), chg * pages_per_huge_page(h), h_cg); From 3159bce98da9e03b2a36e9cf3417adb1da4147ba Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Tue, 8 Sep 2026 09:28:01 +0800 Subject: [PATCH 0536/1012] mm/zswap: publish the initial pool with list_add_rcu() zswap_setup() publishes the pool on the zswap_pools list with a plain list_add(), but the list is walked by concurrent RCU readers holding nothing but rcu_read_lock() through zswap_total_pages(), e.g. /proc/meminfo and the shrinker count path. CPU 0 (writer) CPU 1 (reader) -------------- -------------- zswap_pool_create(): pool->zs_pool = zs_create_pool(); (1) list_add() -> __list_add(): WRITE_ONCE(zswap_pools.next, &pool->list); (2) zswap_total_pages(): pool = READ_ONCE( (a) zswap_pools.next); zs_get_total_pages( (b) pool->zs_pool); If (2) becomes visible to CPU 1 before (1), CPU 1 finds the pool at (a) but dereferences a wild pointer at (b). Publish the node with list_add_rcu(). Link: https://lore.kernel.org/20260908012801.1864430-1-xialonglong2025@163.com Fixes: 91cdcd8d624b ("mm: zswap: optimize zswap pool size tracking") Signed-off-by: Longlong Xia Signed-off-by: Andrew Morton Reported-by: Sashiko Closes: https://sashiko.dev/#/patchset/20260906133601.3563324-1-xialonglong2025%40163.com Acked-by: Yosry Ahmed Acked-by: Nhat Pham Assisted-by: Zcode:GLM-5.3 Cc: Chengming Zhou Cc: Johannes Weiner Cc: --- mm/zswap.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/zswap.c b/mm/zswap.c index b5411ab164809d..2a95aedc08fcca 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1815,7 +1815,7 @@ static int zswap_setup(void) pool = __zswap_pool_create_fallback(); if (pool) { pr_info("loaded using pool %s\n", pool->tfm_name); - list_add(&pool->list, &zswap_pools); + list_add_rcu(&pool->list, &zswap_pools); zswap_has_pool = true; static_branch_enable(&zswap_ever_enabled); } else { From 6a44b3bcf3220bf0689d7c05f4b5391e67f291bd Mon Sep 17 00:00:00 2001 From: Bingfang Guo Date: Thu, 10 Sep 2026 11:46:58 +0800 Subject: [PATCH 0537/1012] mm/memcg: clear folio memcg after changing per memcg stats I notice extremely high swapcached count in the per memcg level memory.stat when running tests with cgroupv1 setup by swapping pages in and out. It seems that the counter never gets decreased so the value is rather useless and confusing to users reading it. So I think fixing it so that the value can reflect the actual swapcache usage correctly could be helpful. __memcg1_swapout() transfers the memsw charge of a folio to its swap entry and clears folio->memcg_data as part of that. In the vmscan swapout path it runs before __swap_cache_del_folio(), which then decrements the swapcache stats through lruvec_stat_mod_folio(). Since folio->memcg_data has already been cleared, folio_memcg() returns NULL and the NR_SWAPCACHE decrement only updates the node-level counter instead of the memcg's lruvec, leaking the per-memcg swapcache count. Users using swaps will read totally meaningless swapcache value from per memcg memory.stat, like 200GB of swapcache on a 64GB setup, which is quite confusing and may trigger monitoring alerts, if any. Move the __memcg1_swapout() call into __swap_cache_del_folio(), after the NR_FILE_PAGES and NR_SWAPCACHE updates but before __swap_cache_do_del_folio() removes the folio from the swap cache. This keeps the stats attributed to the folio's memcg while still recording the swap cgroup with a valid folio->swap. Add a swapout parameter so the plain swap_cache_del_folio() path is left unchanged. Link: https://lore.kernel.org/20260910-memcg-swapcache-stats-fix-v5-1-033f510ba748@tencent.com Fixes: 2732acda82c9 ("mm, swap: use swap cache as the swap in synchronize layer") Signed-off-by: Bingfang Guo Signed-off-by: Andrew Morton Acked-by: Kairui Song Cc: Axel Rasmussen Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kemeng Shi Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Nhat Pham Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie Cc: --- mm/swap.h | 6 ++++-- mm/swap_state.c | 13 ++++++++++--- mm/vmscan.c | 3 +-- 3 files changed, 15 insertions(+), 7 deletions(-) diff --git a/mm/swap.h b/mm/swap.h index 0b5d507739bcb6..b3b54c28929a19 100644 --- a/mm/swap.h +++ b/mm/swap.h @@ -319,7 +319,8 @@ struct folio *swap_cache_alloc_folio(swp_entry_t target_entry, gfp_t gfp_mask, void __swap_cache_add_folio(struct swap_cluster_info *ci, struct folio *folio, swp_entry_t entry); void __swap_cache_del_folio(struct swap_cluster_info *ci, - struct folio *folio, swp_entry_t entry, void *shadow); + struct folio *folio, swp_entry_t entry, void *shadow, + bool swapout); void __swap_cache_replace_folio(struct swap_cluster_info *ci, struct folio *old, struct folio *new); @@ -452,7 +453,8 @@ static inline void swap_cache_del_folio(struct folio *folio) } static inline void __swap_cache_del_folio(struct swap_cluster_info *ci, - struct folio *folio, swp_entry_t entry, void *shadow) + struct folio *folio, swp_entry_t entry, void *shadow, + bool swapout) { } diff --git a/mm/swap_state.c b/mm/swap_state.c index 305877e1f4d7bf..625c185a1ca4d5 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -306,21 +306,28 @@ static void __swap_cache_do_del_folio(struct swap_cluster_info *ci, * @folio: The folio. * @entry: The first swap entry that the folio corresponds to. * @shadow: shadow value to be filled in the swap cache. + * @swapout: whether this folio is being reclaimed after swapout. * * Removes a folio from the swap cache and fills a shadow in place. * This won't put the folio's refcount. The caller has to do that. * * Context: Caller must ensure the folio is locked and in the swap cache * using the index of @entry, and lock the cluster that holds the entries. + * If @swapout is set, the folio should be in reclaim path and IRQs + * should be disabled. */ void __swap_cache_del_folio(struct swap_cluster_info *ci, struct folio *folio, - swp_entry_t entry, void *shadow) + swp_entry_t entry, void *shadow, bool swapout) { unsigned long nr_pages = folio_nr_pages(folio); - __swap_cache_do_del_folio(ci, folio, entry, shadow); node_stat_mod_folio(folio, NR_FILE_PAGES, -nr_pages); lruvec_stat_mod_folio(folio, NR_SWAPCACHE, -nr_pages); + + if (swapout) + __memcg1_swapout(folio, ci); + + __swap_cache_do_del_folio(ci, folio, entry, shadow); } /** @@ -339,7 +346,7 @@ void swap_cache_del_folio(struct folio *folio) swp_entry_t entry = folio->swap; ci = swap_cluster_lock(__swap_entry_to_info(entry), swp_offset(entry)); - __swap_cache_del_folio(ci, folio, entry, NULL); + __swap_cache_del_folio(ci, folio, entry, NULL, false); swap_cluster_unlock(ci); folio_ref_sub(folio, folio_nr_pages(folio)); diff --git a/mm/vmscan.c b/mm/vmscan.c index a49da82ed4797a..8e9c73dcd19bb3 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -807,8 +807,7 @@ static int __remove_mapping(struct address_space *mapping, struct folio *folio, if (reclaimed && !mapping_exiting(mapping)) shadow = workingset_eviction(folio, target_memcg); - __memcg1_swapout(folio, ci); - __swap_cache_del_folio(ci, folio, swap, shadow); + __swap_cache_del_folio(ci, folio, swap, shadow, true); swap_cluster_unlock_irq(ci); } else { void (*free_folio)(struct folio *); From 8f21ccc288d7fd896ca0eaf3b3e95d371e300c1c Mon Sep 17 00:00:00 2001 From: Tianyi Chen Date: Thu, 10 Sep 2026 20:56:44 +0800 Subject: [PATCH 0538/1012] selftests/mm: reject invalid test selections before running tests Patch series "selftests/mm: Validate selections and scope memfd_secret setup", v3. run_vmtests.sh can reach test setup after invalid options or category selections. Its memfd_secret preparation can also change ptrace_scope when that category was not selected. Patch 1 rejects invalid selections before setup. Patch 2 gates memfd_secret preparation on category selection and executable presence. This patch (of 2): getopts reports unknown options and missing arguments, but run_vmtests.sh ignores its error result and continues with test setup. An empty -t argument also falls back to the default selection, while unknown category names can silently select no tests and still reach setup code. Exit on getopts errors and validate category names against the existing list in usage() before any test setup. Reject empty and whitespace-only selections, and normalize category separators so validation and execution agree. Initialize the default selection before parsing options so only -t changes the selection. Link: https://lore.kernel.org/20260910125645.285866-1-diannaaav@gmail.com Link: https://lore.kernel.org/20260910125645.285866-2-diannaaav@gmail.com Fixes: 85463321e726 ("selftests/vm: enable running select groups of tests") Signed-off-by: Tianyi Chen Signed-off-by: Andrew Morton Assisted-by: LLM Cc: David Hildenbrand (Arm) Cc: Joel Savitz Cc: Shuah Khan --- tools/testing/selftests/mm/run_vmtests.sh | 26 +++++++++++++++++++---- 1 file changed, 22 insertions(+), 4 deletions(-) diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index d09f9f6a384ee7..9e62ab4c677520 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -96,26 +96,44 @@ separated by spaces: example: ./run_vmtests.sh -t "hmm mmap ksm" EOF - exit 0 } RUN_ALL=false RUN_DESTRUCTIVE=false TAP_PREFIX="# " +VM_SELFTEST_ITEMS="default" + while getopts "aht:nd" OPT; do case ${OPT} in "a") RUN_ALL=true ;; - "h") usage ;; + "h") usage; exit 0 ;; "t") VM_SELFTEST_ITEMS=${OPTARG} ;; "n") TAP_PREFIX= ;; "d") RUN_DESTRUCTIVE=true ;; + "?") exit 1 ;; esac done shift $((OPTIND -1)) -# default behavior: run all tests -VM_SELFTEST_ITEMS=${VM_SELFTEST_ITEMS:-default} +# Normalize whitespace so validation and test_selected() use the same names. +read -r -a selected_categories <<< "${VM_SELFTEST_ITEMS//$'\n'/ }" +VM_SELFTEST_ITEMS="${selected_categories[*]}" +if [ -z "$VM_SELFTEST_ITEMS" ]; then + echo "No test categories specified" >&2 + exit 1 +fi + +if [ "$VM_SELFTEST_ITEMS" != "default" ]; then + # Keep the documented category list as the source of valid names. + valid_categories=$(usage | sed -n 's/^- //p') + for category in "${selected_categories[@]}"; do + if ! grep -Fxq -- "$category" <<< "$valid_categories"; then + echo "Unknown test category: $category" >&2 + exit 1 + fi + done +fi test_selected() { if [ "$VM_SELFTEST_ITEMS" == "default" ]; then From addbcde521f450f7d834a0cea971d64fb87168dc Mon Sep 17 00:00:00 2001 From: Tianyi Chen Date: Thu, 10 Sep 2026 20:56:45 +0800 Subject: [PATCH 0539/1012] selftests/mm: only prepare ptrace_scope when memfd_secret is selected The memfd_secret setup clears ptrace_scope whenever its test binary is executable, even when a different category was selected. run_test() filters the test invocation, but it does not protect the preceding setup. Check the category selection before entering the memfd_secret block so running unrelated categories does not change ptrace_scope. Keep the existing executable check and the behavior when memfd_secret is selected. Link: https://lore.kernel.org/20260910125645.285866-3-diannaaav@gmail.com Signed-off-by: Tianyi Chen Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Joel Savitz Cc: Shuah Khan --- tools/testing/selftests/mm/run_vmtests.sh | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 9e62ab4c677520..9bbef9410ccc35 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -370,7 +370,7 @@ CATEGORY="process_madv" run_test ./process_madv CATEGORY="vma_merge" run_test ./merge -if [ -x ./memfd_secret ] +if test_selected "memfd_secret" && [ -x ./memfd_secret ] then if [ -f /proc/sys/kernel/yama/ptrace_scope ]; then (echo 0 > /proc/sys/kernel/yama/ptrace_scope 2>&1) | tap_prefix From 1dd91e03f726e23f5fc81e1a0b48ca66fee9af83 Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Thu, 10 Sep 2026 20:35:42 +0800 Subject: [PATCH 0540/1012] mm: zswap: convert zswap_invalidate() to take a range Patch series "mm: zswap: optimize zswap invalidate and store", v3. This series range-ifies zswap_invalidate() to skip xarray lookups when zswap is unused, and reuses it in zswap_store() to eliminate redundant per-slot lookups and open-coded logic. This patch (of 3): zswap_invalidate() takes a swp_entry_t only to unpack it right back into type and offset, and both callers already have those values in hand. Pass type, offset, and nr_entries directly so swap_range_free() can invalidate an entire range in one call instead of looping in the caller. Link: https://lore.kernel.org/20260910123544.818146-1-wangkefeng.wang@huawei.com Link: https://lore.kernel.org/20260910123544.818146-2-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Suggested-by: Johannes Weiner Acked-by: Yosry Ahmed Reviewed-by: Johannes Weiner Cc: Chengming Zhou Cc: Kairui Song Cc: Nhat Pham Cc: Yosry Ahmed --- include/linux/zswap.h | 8 ++++++-- mm/swapfile.c | 4 +--- mm/zswap.c | 29 +++++++++++++++++++---------- 3 files changed, 26 insertions(+), 15 deletions(-) diff --git a/include/linux/zswap.h b/include/linux/zswap.h index 30c193a1207e16..df6cafbe95dc0c 100644 --- a/include/linux/zswap.h +++ b/include/linux/zswap.h @@ -27,7 +27,7 @@ struct zswap_lruvec_state { unsigned long zswap_total_pages(void); bool zswap_store(struct folio *folio); int zswap_load(struct folio *folio); -void zswap_invalidate(swp_entry_t swp); +void zswap_invalidate(int type, pgoff_t offset, unsigned long nr_entries); int zswap_swapon(int type, unsigned long nr_pages); void zswap_swapoff(int type); void zswap_memcg_offline_cleanup(struct mem_cgroup *memcg); @@ -49,7 +49,11 @@ static inline int zswap_load(struct folio *folio) return -ENOENT; } -static inline void zswap_invalidate(swp_entry_t swp) {} +static inline void zswap_invalidate(int type, pgoff_t offset, + unsigned long nr_entries) +{ +} + static inline int zswap_swapon(int type, unsigned long nr_pages) { return 0; diff --git a/mm/swapfile.c b/mm/swapfile.c index a8118f095f4d9a..2c263563b70ebc 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1316,10 +1316,8 @@ static void swap_range_free(struct swap_info_struct *si, unsigned long offset, { unsigned long end = offset + nr_entries - 1; void (*swap_slot_free_notify)(struct block_device *, unsigned long); - unsigned int i; - for (i = 0; i < nr_entries; i++) - zswap_invalidate(swp_entry(si->type, offset + i)); + zswap_invalidate(si->type, offset, nr_entries); if (si->flags & SWP_BLKDEV) swap_slot_free_notify = diff --git a/mm/zswap.c b/mm/zswap.c index 2a95aedc08fcca..9153bdfdd5df37 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -228,10 +228,15 @@ static bool zswap_has_pool; /* One swap address space for each 64M swap space */ #define ZSWAP_ADDRESS_SPACE_SHIFT 14 #define ZSWAP_ADDRESS_SPACE_PAGES (1 << ZSWAP_ADDRESS_SPACE_SHIFT) + +static inline struct xarray *zswap_tree(int type, pgoff_t offset) +{ + return &zswap_trees[type][offset >> ZSWAP_ADDRESS_SPACE_SHIFT]; +} + static inline struct xarray *swap_zswap_tree(swp_entry_t swp) { - return &zswap_trees[swp_type(swp)][swp_offset(swp) - >> ZSWAP_ADDRESS_SPACE_SHIFT]; + return zswap_tree(swp_type(swp), swp_offset(swp)); } #define zswap_pool_debug(msg, p) \ @@ -1656,18 +1661,22 @@ int zswap_load(struct folio *folio) return 0; } -void zswap_invalidate(swp_entry_t swp) +void zswap_invalidate(int type, pgoff_t offset, unsigned long nr_entries) { - pgoff_t offset = swp_offset(swp); - struct xarray *tree = swap_zswap_tree(swp); struct zswap_entry *entry; + struct xarray *tree; + unsigned long i; - if (xa_empty(tree)) - return; + for (i = 0; i < nr_entries; i++) { + tree = zswap_tree(type, offset + i); - entry = xa_erase(tree, offset); - if (entry) - zswap_entry_free(entry); + if (xa_empty(tree)) + continue; + + entry = xa_erase(tree, offset + i); + if (entry) + zswap_entry_free(entry); + } } int zswap_swapon(int type, unsigned long nr_pages) From a2fea19043142c247471eaab66c5c21b1b438b3d Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Thu, 10 Sep 2026 20:35:43 +0800 Subject: [PATCH 0541/1012] mm: zswap: skip xarray walk in zswap_invalidate() when zswap is unused zswap_invalidate() still walks the per-area xarray even when zswap has never been enabled. Add a zswap_never_enabled() to skip it. Link: https://lore.kernel.org/20260910123544.818146-3-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Acked-by: Yosry Ahmed Reviewed-by: Johannes Weiner Cc: Chengming Zhou Cc: Kairui Song Cc: Nhat Pham Cc: Yosry Ahmed --- mm/zswap.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/mm/zswap.c b/mm/zswap.c index 9153bdfdd5df37..bf86651d746499 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1667,6 +1667,9 @@ void zswap_invalidate(int type, pgoff_t offset, unsigned long nr_entries) struct xarray *tree; unsigned long i; + if (zswap_never_enabled()) + return; + for (i = 0; i < nr_entries; i++) { tree = zswap_tree(type, offset + i); From d89f030fcd0f0627cadaf8407d1270baeee08141 Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Thu, 10 Sep 2026 20:35:44 +0800 Subject: [PATCH 0542/1012] mm: zswap: reuse zswap_invalidate() in zswap_store() Reuse zswap_invalidate() in zswap_store() check_old path. Its zswap_never_enabled() and xa_empty() guards avoid redundant xarray lookups and deduplicate the per-slot free logic. Link: https://lore.kernel.org/20260910123544.818146-4-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Acked-by: Yosry Ahmed Reviewed-by: Johannes Weiner Cc: Chengming Zhou Cc: Kairui Song Cc: Nhat Pham Cc: Yosry Ahmed --- mm/zswap.c | 15 ++------------- 1 file changed, 2 insertions(+), 13 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index bf86651d746499..d0b6c229b6169d 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1543,19 +1543,8 @@ bool zswap_store(struct folio *folio) * offsets corresponding to each page of the folio. Otherwise, * writeback could overwrite the new data in the swapfile. */ - if (!ret) { - unsigned type = swp_type(swp); - pgoff_t offset = swp_offset(swp); - struct zswap_entry *entry; - struct xarray *tree; - - for (index = 0; index < nr_pages; ++index) { - tree = swap_zswap_tree(swp_entry(type, offset + index)); - entry = xa_erase(tree, offset + index); - if (entry) - zswap_entry_free(entry); - } - } + if (!ret) + zswap_invalidate(swp_type(swp), swp_offset(swp), nr_pages); return ret; } From 321423e95aa19d3c50f8cf1752978fed2db5318a Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:15 +0100 Subject: [PATCH 0543/1012] mm/khugepaged: drop redundant mm_struct pin in madvise_collapse() Patch series "mm/collapse: separate a collapse from its callers", v4. There is no line between the collapse engine and the callers that ask for a collapse. khugepaged.c holds both, and they reach into each other. - Sixteen tests through the collapse path read cc->is_khugepaged to work out what they are allowed to do, when every one of those decisions was made by the caller before it asked. - collapse_single_pmd() does both halves of a collapse behind one call and drops mmap_lock somewhere in the middle. Which of its paths dropped it is not something a caller can see, so it hands back a bool and the caller keeps track. - MADV_COLLAPSE's implementation -- the walk over the user's range, the per-PMD loop, the errno translation -- sits in khugepaged.c, which is the daemon's file. So: draw the line. State what a caller allows in a policy, split the call in two with the lock as the boundary, and move the syscall to madvise.c. What the engine offers is then three calls, with the lock state written down against each, and a policy the caller fills for itself: collapse_control_init(cc) once, before the first table collapse_policy_*(&cc->policy) what this caller allows collapse_scan_pmd(vma, addr, ...) per table, under mmap_lock collapse_run_pmd(mm, addr, ...) when a scan found work, no mmap_lock The engine stays in khugepaged.c for now; what changes is that it has an interface, and that neither half has to ask about the other. madvise.c gains the operation it should have had all along. This patch (of 13): madvise_collapse() holds an mmgrab() reference across its work. It is redundant. Every caller already holds mm_users: - madvise(2) works on current->mm, which lives as long as the task is in the syscall; - process_madvise(2) reaches a remote mm through mm_access(), which takes an mm_users reference and holds it until the syscall returns; - io_uring passes current->mm; - DAMON takes one with get_task_mm() and drops it after the call. Drop the mmgrab()/mmdrop() pair. Link: https://lore.kernel.org/20260928100630.21870-1-kirill@shutemov.name Link: https://lore.kernel.org/20260928100630.21870-2-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 85ea095906fb70..7096dbf09b031b 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -3252,7 +3252,6 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, cc->is_khugepaged = false; cc->progress = 0; - mmgrab(mm); lru_add_drain_all(); for (addr = hstart; addr < hend; addr += HPAGE_PMD_SIZE) { @@ -3308,7 +3307,6 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, } out_nolock: mmap_assert_locked(mm); - mmdrop(mm); kfree(cc); return thps == ((hend - hstart) >> HPAGE_PMD_SHIFT) ? 0 From b5ecbf6a8d046ae2ad1911867b9ba7cf7746c146 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:16 +0100 Subject: [PATCH 0544/1012] mm/khugepaged: count collapses where khugepaged makes them collapse_single_pmd() bumps khugepaged_pages_collapsed for its caller, and tests cc->is_khugepaged to know whether it should: the counter belongs to the daemon, and MADV_COLLAPSE must not touch it. The daemon sees every result of every collapse it asks for, so it can keep its own counter without the shared path testing who called. Link: https://lore.kernel.org/20260928100630.21870-3-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 11 ++++------- 1 file changed, 4 insertions(+), 7 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 7096dbf09b031b..ec1023032b8090 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -2838,10 +2838,8 @@ static enum scan_result collapse_single_pmd(unsigned long addr, mmap_assert_locked(mm); - if (vma_is_anonymous(vma)) { - result = collapse_scan_pmd(mm, vma, addr, lock_dropped, cc); - goto end; - } + if (vma_is_anonymous(vma)) + return collapse_scan_pmd(mm, vma, addr, lock_dropped, cc); file = get_file(vma->vm_file); pgoff = linear_page_index(vma, addr); @@ -2877,9 +2875,6 @@ static enum scan_result collapse_single_pmd(unsigned long addr, result = SCAN_SUCCEED; mmap_read_unlock(mm); } -end: - if (cc->is_khugepaged && result == SCAN_SUCCEED) - ++khugepaged_pages_collapsed; return result; } @@ -2956,6 +2951,8 @@ static void collapse_scan_mm_slot(unsigned int progress_max, *result = collapse_single_pmd(khugepaged_scan.address, vma, &lock_dropped, cc); + if (*result == SCAN_SUCCEED) + khugepaged_pages_collapsed++; /* move to next address */ khugepaged_scan.address += HPAGE_PMD_SIZE; if (lock_dropped) From 3abac1c24614d8e561fe5fd3662c7637804b22f4 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:17 +0100 Subject: [PATCH 0545/1012] mm/khugepaged: rename mthp_present_ptes bitmap to eligible_ptes The name says less than the bit means. A set bit means not only that the PTE is present, but also that it passed the other checks: uffd, lazyfree, anonymity, sharing. The PTE can be considered a collapse source. mthp_collapse() then reads the bitmap starting at the PMD order, so the bitmap is not specific to mTHP either. Name it for what a set bit means, and update the comments that named it. No functional change. Link: https://lore.kernel.org/20260928100630.21870-4-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 32 ++++++++++++++++---------------- 1 file changed, 16 insertions(+), 16 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index ec1023032b8090..07ae6c2e4fcf35 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -115,8 +115,8 @@ struct collapse_control { /* nodemask for allocation fallback */ nodemask_t alloc_nmask; - /* Each bit represents a single occupied (!none/zero) page. */ - DECLARE_BITMAP(mthp_present_ptes, MAX_PTRS_PER_PTE); + /* Each bit marks a PTE the scan accepted as a collapse source */ + DECLARE_BITMAP(eligible_ptes, MAX_PTRS_PER_PTE); }; /** @@ -627,7 +627,7 @@ static void collapse_control_init_scan(struct collapse_control *cc) { memset(cc->node_load, 0, sizeof(cc->node_load)); nodes_clear(cc->alloc_nmask); - bitmap_zero(cc->mthp_present_ptes, MAX_PTRS_PER_PTE); + bitmap_zero(cc->eligible_ptes, MAX_PTRS_PER_PTE); } static void release_pte_folio(struct folio *folio) @@ -1512,15 +1512,15 @@ static unsigned int max_order_from_offset(unsigned int offset) * mthp_collapse() consumes the bitmap that is generated during * collapse_scan_pmd() to determine what regions and mTHP orders fit best. * - * Each bit in cc->mthp_present_ptes represents a single occupied (!none/zero) - * page. We start at the PMD order and check if it is eligible for collapse; + * Each bit in cc->eligible_ptes marks a PTE the scan accepted as a collapse + * source. We start at the PMD order and check if it is eligible for collapse; * if not, we check the left and right halves of the PTE page table we are * examining at a lower order. * - * For each of these, we determine how many PTE entries are occupied in the - * range of PTE entries we propose to collapse, then we compare this to a - * threshold number of PTE entries which would need to be occupied for a - * collapse to be permitted at that order (accounting for max_ptes_none). + * For each of these, we count the eligible PTEs in the range we propose to + * collapse, then we compare this to the number of eligible PTEs the range + * would need for a collapse to be permitted at that order (accounting for + * max_ptes_none). * * If a collapse is permitted, we attempt to collapse the PTE range into a * mTHP. @@ -1529,7 +1529,7 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, unsigned long address, int referenced, int unmapped, struct collapse_control *cc, unsigned long enabled_orders) { - unsigned int nr_occupied_ptes, nr_ptes, max_ptes_none; + unsigned int nr_eligible_ptes, nr_ptes, max_ptes_none; enum scan_result last_result = SCAN_FAIL; int collapsed = 0; bool alloc_failed = false; @@ -1544,18 +1544,18 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, goto next_order; max_ptes_none = collapse_max_ptes_none(cc, NULL, order); - nr_occupied_ptes = bitmap_weight_from(cc->mthp_present_ptes, offset, + nr_eligible_ptes = bitmap_weight_from(cc->eligible_ptes, offset, offset + nr_ptes); /* * Swap PTEs accepted during the scan are counted in @unmapped, - * not in the present-PTE bitmap. Account them for the PMD-order + * not in cc->eligible_ptes. Account them for the PMD-order * candidate. */ if (is_pmd_order(order)) - nr_occupied_ptes += unmapped; + nr_eligible_ptes += unmapped; - if (nr_occupied_ptes >= nr_ptes - max_ptes_none) { + if (nr_eligible_ptes >= nr_ptes - max_ptes_none) { enum scan_result ret; collapse_address = address + offset * PAGE_SIZE; @@ -1761,8 +1761,8 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, } } - /* Set bit for occupied pages */ - __set_bit(i, cc->mthp_present_ptes); + /* The scan accepted this PTE as a collapse source */ + __set_bit(i, cc->eligible_ptes); /* * Record which node the original page is from and save this * information to cc->node_load[]. From f6677f923f2544574d750ac7d2896e6ec632573c Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:18 +0100 Subject: [PATCH 0546/1012] mm/collapse: add collapse.h for the collapse interface khugepaged.c holds both the users of collapse and the machinery that performs it. The daemon's scan loop, the sysfs tunables, MADV_COLLAPSE's entry point and the collapse itself all sit in one file and reach into each other freely. Nothing marks where a user ends and the engine begins. Start drawing that line. Add mm/collapse.h for what the two sides have to agree on: - enum scan_result - what the engine hands back; - struct collapse_control - the state a request carries. And two constants move with them: - KHUGEPAGED_MAX_PTES_LIMIT -> COLLAPSE_MAX_PTES_LIMIT; - KHUGEPAGED_MIN_MTHP_ORDER -> COLLAPSE_MIN_MTHP_ORDER. Neither is a fact about the daemon, so both lose the KHUGEPAGED_ prefix. No functional change. Link: https://lore.kernel.org/20260928100630.21870-5-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- MAINTAINERS | 1 + mm/collapse.h | 64 +++++++++++++++++++++++++++++++++++++++ mm/khugepaged.c | 79 ++++++++----------------------------------------- 3 files changed, 78 insertions(+), 66 deletions(-) create mode 100644 mm/collapse.h diff --git a/MAINTAINERS b/MAINTAINERS index ca76ab10fe68ed..fe1d70ed5100b5 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17448,6 +17448,7 @@ F: Documentation/admin-guide/mm/transhuge.rst F: include/linux/huge_mm.h F: include/linux/khugepaged.h F: include/trace/events/huge_memory.h +F: mm/collapse.h F: mm/huge_memory.c F: mm/khugepaged.c F: mm/mm_slot.h diff --git a/mm/collapse.h b/mm/collapse.h new file mode 100644 index 00000000000000..b115034d90187a --- /dev/null +++ b/mm/collapse.h @@ -0,0 +1,64 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __MM_COLLAPSE_H +#define __MM_COLLAPSE_H + +#include +#include +#include +#include + +#define COLLAPSE_MAX_PTES_LIMIT (HPAGE_PMD_NR - 1) +#define COLLAPSE_MIN_MTHP_ORDER 2 + +enum scan_result { + SCAN_FAIL, + SCAN_SUCCEED, + SCAN_NO_PTE_TABLE, + SCAN_PMD_MAPPED, + SCAN_EXCEED_NONE_PTE, + SCAN_EXCEED_SWAP_PTE, + SCAN_EXCEED_SHARED_PTE, + SCAN_PTE_NON_PRESENT, + SCAN_PTE_UFFD, + SCAN_PTE_MAPPED_HUGEPAGE, + SCAN_LACK_REFERENCED_PAGE, + SCAN_PAGE_NULL, + SCAN_SCAN_ABORT, + SCAN_PAGE_COUNT, + SCAN_PAGE_LRU, + SCAN_PAGE_LOCK, + SCAN_PAGE_ANON, + SCAN_PAGE_LAZYFREE, + SCAN_PAGE_COMPOUND, + SCAN_ANY_PROCESS, + SCAN_VMA_NULL, + SCAN_VMA_CHECK, + SCAN_ADDRESS_RANGE, + SCAN_DEL_PAGE_LRU, + SCAN_ALLOC_HUGE_PAGE_FAIL, + SCAN_CGROUP_CHARGE_FAIL, + SCAN_TRUNCATED, + SCAN_PAGE_HAS_PRIVATE, + SCAN_STORE_FAILED, + SCAN_COPY_MC, + SCAN_PAGE_FILLED, + SCAN_PAGE_DIRTY_OR_WRITEBACK, +}; + +struct collapse_control { + bool is_khugepaged; + + /* Num pages scanned per node */ + u32 node_load[MAX_NUMNODES]; + + /* Num pages scanned (see khugepaged_pages_to_scan) */ + unsigned int progress; + + /* nodemask for allocation fallback */ + nodemask_t alloc_nmask; + + /* Each bit marks a PTE the scan accepted as a collapse source */ + DECLARE_BITMAP(eligible_ptes, MAX_PTRS_PER_PTE); +}; + +#endif /* __MM_COLLAPSE_H */ diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 07ae6c2e4fcf35..6c7ac9ebcd4b45 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -26,44 +26,10 @@ #include #include +#include "collapse.h" #include "internal.h" -#include "page_alloc.h" #include "mm_slot.h" - -enum scan_result { - SCAN_FAIL, - SCAN_SUCCEED, - SCAN_NO_PTE_TABLE, - SCAN_PMD_MAPPED, - SCAN_EXCEED_NONE_PTE, - SCAN_EXCEED_SWAP_PTE, - SCAN_EXCEED_SHARED_PTE, - SCAN_PTE_NON_PRESENT, - SCAN_PTE_UFFD, - SCAN_PTE_MAPPED_HUGEPAGE, - SCAN_LACK_REFERENCED_PAGE, - SCAN_PAGE_NULL, - SCAN_SCAN_ABORT, - SCAN_PAGE_COUNT, - SCAN_PAGE_LRU, - SCAN_PAGE_LOCK, - SCAN_PAGE_ANON, - SCAN_PAGE_LAZYFREE, - SCAN_PAGE_COMPOUND, - SCAN_ANY_PROCESS, - SCAN_VMA_NULL, - SCAN_VMA_CHECK, - SCAN_ADDRESS_RANGE, - SCAN_DEL_PAGE_LRU, - SCAN_ALLOC_HUGE_PAGE_FAIL, - SCAN_CGROUP_CHARGE_FAIL, - SCAN_TRUNCATED, - SCAN_PAGE_HAS_PRIVATE, - SCAN_STORE_FAILED, - SCAN_COPY_MC, - SCAN_PAGE_FILLED, - SCAN_PAGE_DIRTY_OR_WRITEBACK, -}; +#include "page_alloc.h" #define CREATE_TRACE_POINTS #include @@ -91,7 +57,6 @@ static DECLARE_WAIT_QUEUE_HEAD(khugepaged_wait); * * Note that these are only respected if collapse was initiated by khugepaged. */ -#define KHUGEPAGED_MAX_PTES_LIMIT (HPAGE_PMD_NR - 1) unsigned int khugepaged_max_ptes_none __read_mostly; static unsigned int khugepaged_max_ptes_swap __read_mostly; static unsigned int khugepaged_max_ptes_shared __read_mostly; @@ -101,24 +66,6 @@ static DEFINE_READ_MOSTLY_HASHTABLE(mm_slots_hash, MM_SLOTS_HASH_BITS); static struct kmem_cache *mm_slot_cache __ro_after_init; -#define KHUGEPAGED_MIN_MTHP_ORDER 2 - -struct collapse_control { - bool is_khugepaged; - - /* Num pages scanned per node */ - u32 node_load[MAX_NUMNODES]; - - /* Num pages scanned (see khugepaged_pages_to_scan) */ - unsigned int progress; - - /* nodemask for allocation fallback */ - nodemask_t alloc_nmask; - - /* Each bit marks a PTE the scan accepted as a collapse source */ - DECLARE_BITMAP(eligible_ptes, MAX_PTRS_PER_PTE); -}; - /** * struct khugepaged_scan - cursor for scanning * @mm_head: the head of the mm list to scan @@ -267,7 +214,7 @@ static ssize_t max_ptes_none_store(struct kobject *kobj, unsigned long max_ptes_none; err = kstrtoul(buf, 10, &max_ptes_none); - if (err || max_ptes_none > KHUGEPAGED_MAX_PTES_LIMIT) + if (err || max_ptes_none > COLLAPSE_MAX_PTES_LIMIT) return -EINVAL; khugepaged_max_ptes_none = max_ptes_none; @@ -292,7 +239,7 @@ static ssize_t max_ptes_swap_store(struct kobject *kobj, unsigned long max_ptes_swap; err = kstrtoul(buf, 10, &max_ptes_swap); - if (err || max_ptes_swap > KHUGEPAGED_MAX_PTES_LIMIT) + if (err || max_ptes_swap > COLLAPSE_MAX_PTES_LIMIT) return -EINVAL; khugepaged_max_ptes_swap = max_ptes_swap; @@ -318,7 +265,7 @@ static ssize_t max_ptes_shared_store(struct kobject *kobj, unsigned long max_ptes_shared; err = kstrtoul(buf, 10, &max_ptes_shared); - if (err || max_ptes_shared > KHUGEPAGED_MAX_PTES_LIMIT) + if (err || max_ptes_shared > COLLAPSE_MAX_PTES_LIMIT) return -EINVAL; khugepaged_max_ptes_shared = max_ptes_shared; @@ -378,19 +325,19 @@ static unsigned int collapse_max_ptes_none(struct collapse_control *cc, if (is_pmd_order(order)) return max_ptes_none; /* - * for mTHP collapse with the sysctl value set to KHUGEPAGED_MAX_PTES_LIMIT, + * for mTHP collapse with the sysctl value set to COLLAPSE_MAX_PTES_LIMIT, * scale the maximum number of PTEs to the order of the collapse. */ - if (max_ptes_none == KHUGEPAGED_MAX_PTES_LIMIT) + if (max_ptes_none == COLLAPSE_MAX_PTES_LIMIT) return (1 << order) - 1; /* - * For mTHP collapse of values other than 0 or KHUGEPAGED_MAX_PTES_LIMIT, + * For mTHP collapse of values other than 0 or COLLAPSE_MAX_PTES_LIMIT, * emit a warning and return 0. */ if (max_ptes_none) pr_warn_once("mTHP collapse does not support max_ptes_none" " values other than 0 or %u, defaulting to 0.\n", - KHUGEPAGED_MAX_PTES_LIMIT); + COLLAPSE_MAX_PTES_LIMIT); return 0; } @@ -476,7 +423,7 @@ int __init khugepaged_init(void) return -ENOMEM; khugepaged_pages_to_scan = HPAGE_PMD_NR * 8; - khugepaged_max_ptes_none = KHUGEPAGED_MAX_PTES_LIMIT; + khugepaged_max_ptes_none = COLLAPSE_MAX_PTES_LIMIT; khugepaged_max_ptes_swap = HPAGE_PMD_NR / 8; khugepaged_max_ptes_shared = HPAGE_PMD_NR / 2; @@ -1601,8 +1548,8 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, * any smaller order enabled. When at the smallest order * we must always move to the next offset. */ - if (order > KHUGEPAGED_MIN_MTHP_ORDER && - (enabled_orders & GENMASK(order - 1, 0))) { + if (order > COLLAPSE_MIN_MTHP_ORDER && + (enabled_orders & GENMASK(order - 1, 0))) { order--; continue; } @@ -1666,7 +1613,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, * is then checked again in mthp_collapse() for each attempted order. */ if (enabled_orders != BIT(HPAGE_PMD_ORDER)) - max_ptes_none = KHUGEPAGED_MAX_PTES_LIMIT; + max_ptes_none = COLLAPSE_MAX_PTES_LIMIT; pte = pte_offset_map_lock(mm, pmd, start_addr, &ptl); if (!pte) { From 9d5e15c706aa76216f6f8f98c6eb2c7413d780ba Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:19 +0100 Subject: [PATCH 0547/1012] mm/collapse: state what a collapse may do in the policy Tests scattered through the collapse path decide what a collapse is allowed to do by asking whether khugepaged started it. Between them they settle: - which VMAs are eligible, and how hard to try for a folio; - how many empty, swapped-out or shared PTEs a window may contain, and whether a sub-PMD window is held to a stricter rule than a PMD; - whether a range has to look used, and whether a MADV_FREE'd page is left alone; - whether the PMD is mapped as part of the request, and whether dirty pages are worth writing back and retrying. None of those is a fact about khugepaged. Each is something the caller decided before asking, and the collapse code should not have to look up who called to find out. Add struct collapse_policy for the caller to fill: khugepaged from its own settings, MADV_COLLAPSE from the fact that a user asked explicitly. Every test becomes a read of a field, and cc->is_khugepaged goes, having no reader left. The PTE limits come as two sets, one for a PMD-sized window and one for anything smaller, so that the helpers pick a set for the order and read it. khugepaged takes no swapped-out or shared PTE into a sub-PMD window, and empty PTEs only when the knob says all or nothing; the warning for a knob value in between moves to where khugepaged fills its policy. The fields only one side reads say which: anon_ or file_. David Hildenbrand asked for both. khugepaged fills the policy once per scan pass, MADV_COLLAPSE once per call. That is the one change in behaviour. The max_ptes_* limits and the defrag setting behind the allocation mask are sampled once per pass rather than on every table. A table scanned early in a pass and one scanned late are then treated alike. collapse_file() also drops a NULL check on the collapse_control. It has one call site, reached only from collapse_single_pmd(), which dereferences cc unconditionally, so the check was already dead. Link: https://lore.kernel.org/20260928100630.21870-6-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Tested-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/collapse.h | 41 +++++++++++++- mm/khugepaged.c | 147 +++++++++++++++++++++++++----------------------- 2 files changed, 118 insertions(+), 70 deletions(-) diff --git a/mm/collapse.h b/mm/collapse.h index b115034d90187a..dcd11707195518 100644 --- a/mm/collapse.h +++ b/mm/collapse.h @@ -45,8 +45,47 @@ enum scan_result { SCAN_PAGE_DIRTY_OR_WRITEBACK, }; +/* How many PTEs of a window may be missing, swapped out or shared */ +struct collapse_limits { + /* Counted over a PMD-sized window; HPAGE_PMD_NR means "no limit" */ + unsigned int max_ptes_none; + unsigned int max_ptes_swap; + unsigned int max_ptes_shared; +}; + +/* What a collapse is allowed to do, decided by the caller that asks for it */ +struct collapse_policy { + /* Limits for a PMD-sized window */ + struct collapse_limits pmd; + + /* + * Limits for a smaller window. Its max_ptes_none is either 0 or + * COLLAPSE_MAX_PTES_LIMIT, the latter meaning all but one PTE of the + * window whatever its order; any other value counts as 0. + */ + struct collapse_limits sub_pmd; + + /* Leave clean lazyfree folios to reclaim rather than collapse them */ + bool anon_skip_lazyfree; + + /* Refuse an anonymous range with no sign of use */ + bool anon_require_referenced; + + /* Map the PMD over a file collapse instead of leaving it to a fault */ + bool file_install_pmd; + + /* Write dirty pages back and retry once instead of refusing them */ + bool file_writeback_dirty; + + /* How hard to try for a destination folio */ + gfp_t gfp; + + /* Which VMAs are eligible, as thp_vma_allowable_orders() spells it */ + enum tva_type tva_type; +}; + struct collapse_control { - bool is_khugepaged; + struct collapse_policy policy; /* Num pages scanned per node */ u32 node_load[MAX_NUMNODES]; diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 6c7ac9ebcd4b45..35b282ed39ff1d 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -314,30 +314,17 @@ static bool pte_none_or_zero(pte_t pte) static unsigned int collapse_max_ptes_none(struct collapse_control *cc, struct vm_area_struct *vma, unsigned int order) { - const unsigned int max_ptes_none = khugepaged_max_ptes_none; + unsigned int max_ptes_none; if (vma && userfaultfd_armed(vma)) return 0; - /* for MADV_COLLAPSE, allow any empty/shared zeropage PTEs */ - if (!cc->is_khugepaged) - return HPAGE_PMD_NR; - /* for PMD collapse, respect the user defined maximum */ if (is_pmd_order(order)) - return max_ptes_none; - /* - * for mTHP collapse with the sysctl value set to COLLAPSE_MAX_PTES_LIMIT, - * scale the maximum number of PTEs to the order of the collapse. - */ + return cc->policy.pmd.max_ptes_none; + + /* Below PMD order: all but one PTE of the window, or none */ + max_ptes_none = cc->policy.sub_pmd.max_ptes_none; if (max_ptes_none == COLLAPSE_MAX_PTES_LIMIT) return (1 << order) - 1; - /* - * For mTHP collapse of values other than 0 or COLLAPSE_MAX_PTES_LIMIT, - * emit a warning and return 0. - */ - if (max_ptes_none) - pr_warn_once("mTHP collapse does not support max_ptes_none" - " values other than 0 or %u, defaulting to 0.\n", - COLLAPSE_MAX_PTES_LIMIT); return 0; } @@ -353,20 +340,9 @@ static unsigned int collapse_max_ptes_none(struct collapse_control *cc, static unsigned int collapse_max_ptes_shared(struct collapse_control *cc, unsigned int order) { - /* - * For MADV_COLLAPSE, do not restrict the number of PTEs that map shared - * anonymous pages. - */ - if (!cc->is_khugepaged) - return HPAGE_PMD_NR; - /* - * for mTHP collapse do not allow collapsing anonymous memory pages that - * are shared between processes. - */ - if (!is_pmd_order(order)) - return 0; - /* for PMD collapse, respect the user defined maximum */ - return khugepaged_max_ptes_shared; + if (is_pmd_order(order)) + return cc->policy.pmd.max_ptes_shared; + return cc->policy.sub_pmd.max_ptes_shared; } /** @@ -381,17 +357,9 @@ static unsigned int collapse_max_ptes_shared(struct collapse_control *cc, static unsigned int collapse_max_ptes_swap(struct collapse_control *cc, unsigned int order) { - /* - * For MADV_COLLAPSE, do not restrict the number PTEs entries or - * pagecache entries that are non-present. - */ - if (!cc->is_khugepaged) - return HPAGE_PMD_NR; - /* for mTHP collapse do not allow any non-present PTEs or pagecache entries */ - if (!is_pmd_order(order)) - return 0; - /* for PMD collapse, respect the user defined maximum */ - return khugepaged_max_ptes_swap; + if (is_pmd_order(order)) + return cc->policy.pmd.max_ptes_swap; + return cc->policy.sub_pmd.max_ptes_swap; } int hugepage_madvise(struct vm_area_struct *vma, @@ -678,7 +646,7 @@ static enum scan_result __collapse_huge_page_isolate(struct vm_area_struct *vma, * If the vma has the VM_DROPPABLE flag, the collapse will * preserve the lazyfree property without needing to skip. */ - if (cc->is_khugepaged && !(vma->vm_flags & VM_DROPPABLE) && + if (cc->policy.anon_skip_lazyfree && !(vma->vm_flags & VM_DROPPABLE) && folio_test_lazyfree(folio) && !pte_dirty(pteval)) { result = SCAN_PAGE_LAZYFREE; goto out; @@ -767,12 +735,12 @@ static enum scan_result __collapse_huge_page_isolate(struct vm_area_struct *vma, if (folio_test_large(folio)) list_add_tail(&folio->lru, compound_pagelist); next: - if (cc->is_khugepaged && + if (cc->policy.anon_require_referenced && folio_pte_referenced(folio, vma, addr, pteval)) referenced++; } - if (unlikely(cc->is_khugepaged && !referenced)) { + if (unlikely(cc->policy.anon_require_referenced && !referenced)) { result = SCAN_LACK_REFERENCED_PAGE; } else { result = SCAN_SUCCEED; @@ -938,9 +906,7 @@ static void khugepaged_alloc_sleep(void) remove_wait_queue(&khugepaged_wait, &wait); } -static struct collapse_control khugepaged_collapse_control = { - .is_khugepaged = true, -}; +static struct collapse_control khugepaged_collapse_control; static bool collapse_scan_abort(int nid, struct collapse_control *cc) { @@ -976,6 +942,52 @@ static inline gfp_t alloc_hugepage_khugepaged_gfpmask(void) return khugepaged_defrag() ? GFP_TRANSHUGE : GFP_TRANSHUGE_LIGHT; } +/* khugepaged collapses on its own initiative, so it obeys its own settings */ +static void collapse_policy_khugepaged(struct collapse_policy *p) +{ + p->pmd.max_ptes_none = READ_ONCE(khugepaged_max_ptes_none); + p->pmd.max_ptes_swap = READ_ONCE(khugepaged_max_ptes_swap); + p->pmd.max_ptes_shared = READ_ONCE(khugepaged_max_ptes_shared); + + /* + * A sub-PMD window takes no swapped-out and no shared PTE: reading + * pages back or breaking CoW is not worth it for an mTHP. Empty PTEs + * it takes all or nothing, since anything in between would let one + * collapse feed the next. + */ + p->sub_pmd.max_ptes_none = p->pmd.max_ptes_none; + if (p->sub_pmd.max_ptes_none && + p->sub_pmd.max_ptes_none != COLLAPSE_MAX_PTES_LIMIT) + pr_warn_once("mTHP collapse does not support max_ptes_none values other than 0 or %u, defaulting to 0.\n", + COLLAPSE_MAX_PTES_LIMIT); + p->sub_pmd.max_ptes_swap = 0; + p->sub_pmd.max_ptes_shared = 0; + + p->anon_skip_lazyfree = true; + p->anon_require_referenced = true; + p->file_install_pmd = false; + p->file_writeback_dirty = false; + p->gfp = alloc_hugepage_khugepaged_gfpmask(); + p->tva_type = TVA_KHUGEPAGED; +} + +/* MADV_COLLAPSE was asked for explicitly, so it is not held to those */ +static void collapse_policy_madvise(struct collapse_policy *p) +{ + p->pmd.max_ptes_none = HPAGE_PMD_NR; + p->pmd.max_ptes_swap = HPAGE_PMD_NR; + p->pmd.max_ptes_shared = HPAGE_PMD_NR; + /* Never read: MADV_COLLAPSE collapses to PMD order only */ + p->sub_pmd = p->pmd; + + p->anon_skip_lazyfree = false; + p->anon_require_referenced = false; + p->file_install_pmd = true; + p->file_writeback_dirty = true; + p->gfp = GFP_TRANSHUGE; + p->tva_type = TVA_FORCED_COLLAPSE; +} + #ifdef CONFIG_NUMA static int collapse_find_target_node(struct collapse_control *cc) { @@ -1013,8 +1025,7 @@ static enum scan_result hugepage_vma_revalidate(struct mm_struct *mm, unsigned l struct collapse_control *cc, unsigned int order) { struct vm_area_struct *vma; - enum tva_type type = cc->is_khugepaged ? TVA_KHUGEPAGED : - TVA_FORCED_COLLAPSE; + enum tva_type type = cc->policy.tva_type; if (unlikely(collapse_test_exit_or_disable(mm))) return SCAN_ANY_PROCESS; @@ -1197,8 +1208,7 @@ static enum scan_result __collapse_huge_page_swapin(struct mm_struct *mm, static enum scan_result alloc_charge_folio(struct folio **foliop, struct mm_struct *mm, struct collapse_control *cc, unsigned int order) { - gfp_t gfp = (cc->is_khugepaged ? alloc_hugepage_khugepaged_gfpmask() : - GFP_TRANSHUGE); + gfp_t gfp = cc->policy.gfp; int node = collapse_find_target_node(cc); struct folio *folio; @@ -1581,7 +1591,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, const unsigned int max_ptes_shared = collapse_max_ptes_shared(cc, HPAGE_PMD_ORDER); const unsigned int max_ptes_swap = collapse_max_ptes_swap(cc, HPAGE_PMD_ORDER); unsigned int max_ptes_none = collapse_max_ptes_none(cc, vma, HPAGE_PMD_ORDER); - enum tva_type tva_flags = cc->is_khugepaged ? TVA_KHUGEPAGED : TVA_FORCED_COLLAPSE; + enum tva_type tva_flags = cc->policy.tva_type; pmd_t *pmd; pte_t *pte, *_pte, pteval; int i; @@ -1681,7 +1691,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, * If the vma has the VM_DROPPABLE flag, the collapse will * preserve the lazyfree property without needing to skip. */ - if (cc->is_khugepaged && !(vma->vm_flags & VM_DROPPABLE) && + if (cc->policy.anon_skip_lazyfree && !(vma->vm_flags & VM_DROPPABLE) && folio_test_lazyfree(folio) && !pte_dirty(pteval)) { result = SCAN_PAGE_LAZYFREE; failed_pfn = folio_pfn(folio); @@ -1747,13 +1757,13 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, goto out_unmap; } - if (cc->is_khugepaged && + if (cc->policy.anon_require_referenced && folio_pte_referenced(folio, vma, addr, pteval)) referenced++; } - if (cc->is_khugepaged && - (!referenced || - (unmapped && referenced < HPAGE_PMD_NR / 2))) { + if (cc->policy.anon_require_referenced && + (!referenced || + (unmapped && referenced < HPAGE_PMD_NR / 2))) { result = SCAN_LACK_REFERENCED_PAGE; } else { result = SCAN_SUCCEED; @@ -2605,11 +2615,11 @@ static enum scan_result collapse_file(struct mm_struct *mm, unsigned long addr, xas_unlock_irq(&xas); /* - * Remove pte page tables, so we can re-fault the page as huge. - * If MADV_COLLAPSE, adjust result to call try_collapse_pte_mapped_thp(). + * Remove pte page tables, so we can re-fault the page as huge. A + * caller that wants the PMD mapped now is told to go and do that. */ retract_page_tables(mapping, start); - if (cc && !cc->is_khugepaged) + if (cc->policy.file_install_pmd) result = SCAN_PTE_MAPPED_HUGEPAGE; folio_unlock(new_folio); @@ -2796,11 +2806,8 @@ static enum scan_result collapse_single_pmd(unsigned long addr, retry: result = collapse_scan_file(mm, addr, file, pgoff, cc); - /* - * For MADV_COLLAPSE, when encountering dirty pages, try to writeback, - * then retry the collapse one time. - */ - if (!cc->is_khugepaged && result == SCAN_PAGE_DIRTY_OR_WRITEBACK && + /* Dirty pages are worth a writeback and one more try, if asked for */ + if (cc->policy.file_writeback_dirty && result == SCAN_PAGE_DIRTY_OR_WRITEBACK && !triggered_wb && mapping_can_writeback(file->f_mapping)) { const loff_t lstart = (loff_t)pgoff << PAGE_SHIFT; const loff_t lend = lstart + HPAGE_PMD_SIZE - 1; @@ -2817,7 +2824,7 @@ static enum scan_result collapse_single_pmd(unsigned long addr, result = SCAN_ANY_PROCESS; else result = try_collapse_pte_mapped_thp(mm, addr, - !cc->is_khugepaged); + cc->policy.file_install_pmd); if (result == SCAN_PMD_MAPPED) result = SCAN_SUCCEED; mmap_read_unlock(mm); @@ -2966,6 +2973,8 @@ static void khugepaged_do_scan(struct collapse_control *cc) lru_add_drain_all(); + collapse_policy_khugepaged(&cc->policy); + cc->progress = 0; while (true) { cond_resched(); @@ -3193,7 +3202,7 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, cc = kmalloc_obj(*cc); if (!cc) return -ENOMEM; - cc->is_khugepaged = false; + collapse_policy_madvise(&cc->policy); cc->progress = 0; lru_add_drain_all(); From b095a046551dfce651e6a49a6fa526def62a284e Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:20 +0100 Subject: [PATCH 0548/1012] mm/collapse: drop the collapse_possible() wrapper collapse_possible() only forwards to collapse_possible_orders() and turns its mask into a bool. Its three callers can test the mask themselves. No functional change. Link: https://lore.kernel.org/20260928100630.21870-7-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 15 +++++---------- 1 file changed, 5 insertions(+), 10 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 35b282ed39ff1d..63e999ff1fc51f 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -494,17 +494,11 @@ static unsigned long collapse_possible_orders(struct vm_area_struct *vma, return thp_vma_allowable_orders(vma, vm_flags, tva_flags, orders); } -static bool collapse_possible(struct vm_area_struct *vma, - vm_flags_t vm_flags, enum tva_type tva_flags) -{ - return collapse_possible_orders(vma, vm_flags, tva_flags); -} - void khugepaged_enter_vma(struct vm_area_struct *vma, vm_flags_t vm_flags) { - if (!mm_flags_test(MMF_VM_HUGEPAGE, vma->vm_mm) && hugepage_enabled() - && collapse_possible(vma, vm_flags, TVA_KHUGEPAGED)) + if (!mm_flags_test(MMF_VM_HUGEPAGE, vma->vm_mm) && hugepage_enabled() && + collapse_possible_orders(vma, vm_flags, TVA_KHUGEPAGED)) __khugepaged_enter(vma->vm_mm); } @@ -2878,7 +2872,8 @@ static void collapse_scan_mm_slot(unsigned int progress_max, cc->progress++; break; } - if (!collapse_possible(vma, vma->vm_flags, TVA_KHUGEPAGED)) { + if (!collapse_possible_orders(vma, vma->vm_flags, + TVA_KHUGEPAGED)) { cc->progress++; continue; } @@ -3190,7 +3185,7 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, BUG_ON(vma->vm_start > start); BUG_ON(vma->vm_end < end); - if (!collapse_possible(vma, vma->vm_flags, TVA_FORCED_COLLAPSE)) + if (!collapse_possible_orders(vma, vma->vm_flags, TVA_FORCED_COLLAPSE)) return -EINVAL; hstart = ALIGN(start, HPAGE_PMD_SIZE); From 2f84fd3cb35eaea3ed6e72fa65183be3e9955381 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:21 +0100 Subject: [PATCH 0549/1012] mm/collapse: name the per-table scan reset for what it resets collapse_control_init_scan() resets what one scan accumulates: the node load, the allocation nodemask and the eligible-PTE bitmap. It runs before every PTE table a scan is given, not once per control, so its name points at the wrong thing. Call it collapse_scan_reset(). Preparation for giving a control a real init and release, run once each for a whole series of scans. Two names a word apart would then stand for two different jobs. No functional change. Link: https://lore.kernel.org/20260928100630.21870-8-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 63e999ff1fc51f..a43c4d1b2ef6dd 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -532,7 +532,7 @@ void __khugepaged_exit(struct mm_struct *mm) } } -static void collapse_control_init_scan(struct collapse_control *cc) +static void collapse_scan_reset(struct collapse_control *cc) { memset(cc->node_load, 0, sizeof(cc->node_load)); nodes_clear(cc->alloc_nmask); @@ -1607,7 +1607,7 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, goto out; } - collapse_control_init_scan(cc); + collapse_scan_reset(cc); enabled_orders = collapse_possible_orders(vma, vma->vm_flags, tva_flags); @@ -2678,7 +2678,7 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, present = 0; swap = 0; - collapse_control_init_scan(cc); + collapse_scan_reset(cc); rcu_read_lock(); xas_for_each(&xas, folio, start + HPAGE_PMD_NR - 1) { if (xas_retry(&xas, folio)) From a3e820e392e514938f7dfbd866ff0df0708d3fea Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:22 +0100 Subject: [PATCH 0550/1012] mm/collapse: call collapse_file() from collapse_single_pmd() collapse_scan_file() reads the page cache to decide whether a table is worth collapsing and, when it is, calls collapse_file() itself. The caller cannot get between the decision and the collapse. Move the collapse_file() call up into collapse_single_pmd(), so the scan stops at the decision. Two things change with it. The writeback retry re-runs collapse_file() alone instead of rescanning first; collapse_file() repeats the scan's checks under the page cache lock anyway. And mm_khugepaged_scan_file fires before the collapse, so for an accepted table its status reads SCAN_SUCCEED; what the collapse made of the table is for mm_khugepaged_collapse_file to report. Preparation for splitting a collapse into a scan under mmap_lock and a run without it. The file scan has to stop where the anonymous one will. Link: https://lore.kernel.org/20260928100630.21870-9-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Assisted-by: LLM Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 21 +++++++++++++-------- 1 file changed, 13 insertions(+), 8 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index a43c4d1b2ef6dd..e6a89609650b76 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -2760,13 +2760,9 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, else cc->progress += HPAGE_PMD_NR; - if (result == SCAN_SUCCEED) { - if (present < HPAGE_PMD_NR - max_ptes_none) { - result = SCAN_EXCEED_NONE_PTE; - count_vm_event(THP_SCAN_EXCEED_NONE_PTE); - } else { - result = collapse_file(mm, addr, file, start, cc); - } + if (result == SCAN_SUCCEED && present < HPAGE_PMD_NR - max_ptes_none) { + result = SCAN_EXCEED_NONE_PTE; + count_vm_event(THP_SCAN_EXCEED_NONE_PTE); } trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present, swap, result); @@ -2797,8 +2793,16 @@ static enum scan_result collapse_single_pmd(unsigned long addr, mmap_read_unlock(mm); *lock_dropped = true; -retry: + + /* + * SCAN_PTE_MAPPED_HUGEPAGE is work too: the page cache already holds + * the PMD folio, and only the PTE table is left to retract. + */ result = collapse_scan_file(mm, addr, file, pgoff, cc); + if (result != SCAN_SUCCEED) + goto put; +retry: + result = collapse_file(mm, addr, file, pgoff, cc); /* Dirty pages are worth a writeback and one more try, if asked for */ if (cc->policy.file_writeback_dirty && result == SCAN_PAGE_DIRTY_OR_WRITEBACK && @@ -2810,6 +2814,7 @@ static enum scan_result collapse_single_pmd(unsigned long addr, triggered_wb = true; goto retry; } +put: fput(file); if (result == SCAN_PTE_MAPPED_HUGEPAGE) { From a9a282938e5d46a753a911fecc3231749b73855d Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:23 +0100 Subject: [PATCH 0551/1012] mm/collapse: separate scanning a PTE table from collapsing it A collapse is two jobs. One reads a PTE table under mmap_lock and decides whether the range is worth collapsing. The other allocates, isolates, copies and flushes, and wants the lock given up first. collapse_single_pmd() did both, so the boundary between them was somewhere in the middle of a function. Give each half its own function: - collapse_scan_pmd() scans one table. The anonymous scan that used to carry that name keeps its body as collapse_scan_anon_pmd(), and collapse_scan_pmd() is now the entry that picks the anonymous or the file side. - collapse_run_pmd() does the collapse the scan asked for, and is handed what the scan returned. SCAN_SUCCEED means there is something to collapse. SCAN_PTE_MAPPED_HUGEPAGE means the page cache already holds the PMD folio and only the PTE table is left to retract. Both are work for the run; anything else is why there is nothing to do. collapse_single_pmd() is now the two of them with the mmap_lock drop in between, so its callers see what they saw before. collapse_control_init() sets a control up before its first scan. What the scan found and the run needs travels in collapse_control. For an anonymous table that is the orders and the referenced and swapped-out counts, which mthp_collapse() and collapse_huge_page() now read from there instead of taking as arguments. For a file it is the file itself and the offset in it: a file collapse works on the page cache and never sees a VMA, so the scan takes the reference while it still has one and the run gives it back. The file scan moves under mmap_lock with the anonymous one, where before the lock was given up first. The lock is now held over the page cache walk, an RCU walk over one table's worth of slots with no PTL, and taken fewer times. collapse_scan_mm_slot() ends its walk whenever the lock was dropped, so a refused file table used to cost khugepaged an unlock, a trip back through khugepaged_do_scan(), a relock and a VMA lookup. Now only a table that goes on to be collapsed does. Link: https://lore.kernel.org/20260928100630.21870-10-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Zi Yan Assisted-by: LLM Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/collapse.h | 14 +++++ mm/khugepaged.c | 141 ++++++++++++++++++++++++++++++++---------------- 2 files changed, 108 insertions(+), 47 deletions(-) diff --git a/mm/collapse.h b/mm/collapse.h index dcd11707195518..ca7b367c89cb39 100644 --- a/mm/collapse.h +++ b/mm/collapse.h @@ -98,6 +98,20 @@ struct collapse_control { /* Each bit marks a PTE the scan accepted as a collapse source */ DECLARE_BITMAP(eligible_ptes, MAX_PTRS_PER_PTE); + + /* + * What a scan found and the run after it needs. Live only between the + * two, and read by nobody else. + * + * The file side takes a reference while it still has the VMA, since a + * file collapse works on the page cache and never sees one; the run is + * what gives it back. + */ + unsigned long scan_orders; + int scan_referenced; + int scan_unmapped; + struct file *scan_file; + pgoff_t scan_pgoff; }; #endif /* __MM_COLLAPSE_H */ diff --git a/mm/khugepaged.c b/mm/khugepaged.c index e6a89609650b76..87bbba6ce59ab9 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1252,8 +1252,8 @@ static pgtable_t alloc_deposit_pte_table(struct mm_struct *mm) * while allocating a THP, as that could trigger direct reclaim/compaction. * Note that the VMA must be rechecked after grabbing the mmap_lock again. */ -static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long start_addr, - int referenced, int unmapped, struct collapse_control *cc, +static enum scan_result collapse_huge_page(struct mm_struct *mm, + unsigned long start_addr, struct collapse_control *cc, unsigned int order) { const unsigned long pmd_addr = start_addr & HPAGE_PMD_MASK; @@ -1300,14 +1300,14 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, unsigned long s goto out_nolock; } - if (unmapped) { + if (cc->scan_unmapped) { /* * __collapse_huge_page_swapin() will return with mmap_lock * released when it fails. So we jump out_nolock directly in * that case. Continuing to collapse causes inconsistency. */ result = __collapse_huge_page_swapin(mm, vma, start_addr, pmd, - referenced, order); + cc->scan_referenced, order); if (result != SCAN_SUCCEED) goto out_nolock; } @@ -1476,9 +1476,8 @@ static unsigned int max_order_from_offset(unsigned int offset) * If a collapse is permitted, we attempt to collapse the PTE range into a * mTHP. */ -static enum scan_result mthp_collapse(struct mm_struct *mm, - unsigned long address, int referenced, int unmapped, - struct collapse_control *cc, unsigned long enabled_orders) +static enum scan_result mthp_collapse(struct mm_struct *mm, unsigned long address, + struct collapse_control *cc) { unsigned int nr_eligible_ptes, nr_ptes, max_ptes_none; enum scan_result last_result = SCAN_FAIL; @@ -1491,7 +1490,7 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, while (offset < HPAGE_PMD_NR) { nr_ptes = 1UL << order; - if (!test_bit(order, &enabled_orders)) + if (!test_bit(order, &cc->scan_orders)) goto next_order; max_ptes_none = collapse_max_ptes_none(cc, NULL, order); @@ -1499,19 +1498,18 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, offset + nr_ptes); /* - * Swap PTEs accepted during the scan are counted in @unmapped, - * not in cc->eligible_ptes. Account them for the PMD-order - * candidate. + * Swap PTEs accepted during the scan are counted in + * cc->scan_unmapped, not in cc->eligible_ptes. Account them for + * the PMD-order candidate. */ if (is_pmd_order(order)) - nr_eligible_ptes += unmapped; + nr_eligible_ptes += cc->scan_unmapped; if (nr_eligible_ptes >= nr_ptes - max_ptes_none) { enum scan_result ret; collapse_address = address + offset * PAGE_SIZE; - ret = collapse_huge_page(mm, collapse_address, referenced, - unmapped, cc, order); + ret = collapse_huge_page(mm, collapse_address, cc, order); switch (ret) { /* Cases where we continue to next collapse candidate */ @@ -1553,7 +1551,7 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, * we must always move to the next offset. */ if (order > COLLAPSE_MIN_MTHP_ORDER && - (enabled_orders & GENMASK(order - 1, 0))) { + (cc->scan_orders & GENMASK(order - 1, 0))) { order--; continue; } @@ -1578,14 +1576,14 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, return last_result; } -static enum scan_result collapse_scan_pmd(struct mm_struct *mm, - struct vm_area_struct *vma, unsigned long start_addr, - bool *lock_dropped, struct collapse_control *cc) +static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma, + unsigned long start_addr, struct collapse_control *cc) { const unsigned int max_ptes_shared = collapse_max_ptes_shared(cc, HPAGE_PMD_ORDER); const unsigned int max_ptes_swap = collapse_max_ptes_swap(cc, HPAGE_PMD_ORDER); unsigned int max_ptes_none = collapse_max_ptes_none(cc, vma, HPAGE_PMD_ORDER); enum tva_type tva_flags = cc->policy.tva_type; + struct mm_struct *mm = vma->vm_mm; pmd_t *pmd; pte_t *pte, *_pte, pteval; int i; @@ -1765,12 +1763,9 @@ static enum scan_result collapse_scan_pmd(struct mm_struct *mm, out_unmap: pte_unmap_unlock(pte, ptl); if (result == SCAN_SUCCEED) { - /* collapse_huge_page() expects the lock to be dropped before calling */ - mmap_read_unlock(mm); - result = mthp_collapse(mm, start_addr, referenced, - unmapped, cc, enabled_orders); - /* mmap_lock was released above, set lock_dropped */ - *lock_dropped = true; + cc->scan_orders = enabled_orders; + cc->scan_referenced = referenced; + cc->scan_unmapped = unmapped; } out: trace_mm_khugepaged_scan_pmd(mm, failed_pfn, referenced, @@ -2765,41 +2760,67 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, count_vm_event(THP_SCAN_EXCEED_NONE_PTE); } - trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present, swap, result); + trace_mm_khugepaged_scan_file(mm, failed_pfn, file, present, swap, + result); return result; } -/* - * Try to collapse a single PMD starting at a PMD aligned addr, and return - * the results. - */ -static enum scan_result collapse_single_pmd(unsigned long addr, - struct vm_area_struct *vma, bool *lock_dropped, - struct collapse_control *cc) +static void collapse_control_init(struct collapse_control *cc) +{ + cc->progress = 0; + cc->scan_file = NULL; +} + +static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, + unsigned long addr, struct collapse_control *cc) { - struct mm_struct *mm = vma->vm_mm; - bool triggered_wb = false; enum scan_result result; - struct file *file; pgoff_t pgoff; - mmap_assert_locked(mm); + mmap_assert_locked(vma->vm_mm); + /* Whatever the last scan found has to have been run by now */ + if (WARN_ON_ONCE(cc->scan_file)) { + fput(cc->scan_file); + cc->scan_file = NULL; + } if (vma_is_anonymous(vma)) - return collapse_scan_pmd(mm, vma, addr, lock_dropped, cc); + return collapse_scan_anon_pmd(vma, addr, cc); - file = get_file(vma->vm_file); pgoff = linear_page_index(vma, addr); - - mmap_read_unlock(mm); - *lock_dropped = true; - + result = collapse_scan_file(vma->vm_mm, addr, vma->vm_file, pgoff, cc); /* * SCAN_PTE_MAPPED_HUGEPAGE is work too: the page cache already holds - * the PMD folio, and only the PTE table is left to retract. + * the PMD folio, and retracting the PTE table is the run's job. */ - result = collapse_scan_file(mm, addr, file, pgoff, cc); - if (result != SCAN_SUCCEED) + if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) + return result; + + /* + * A file collapse works on the page cache and never sees a VMA, so take + * what it needs from this one while it is still here. + */ + cc->scan_file = get_file(vma->vm_file); + cc->scan_pgoff = pgoff; + return result; +} + +static enum scan_result collapse_run_pmd(struct mm_struct *mm, + unsigned long addr, enum scan_result result, + struct collapse_control *cc) +{ + struct file *file = cc->scan_file; + bool triggered_wb = false; + pgoff_t pgoff; + + if (!file) + return mthp_collapse(mm, addr, cc); + + cc->scan_file = NULL; + pgoff = cc->scan_pgoff; + + /* The scan found the PMD folio in place: nothing to collapse */ + if (result == SCAN_PTE_MAPPED_HUGEPAGE) goto put; retry: result = collapse_file(mm, addr, file, pgoff, cc); @@ -2817,6 +2838,10 @@ static enum scan_result collapse_single_pmd(unsigned long addr, put: fput(file); + /* + * A PMD folio is in the page cache, whether the collapse just put it + * there or found it: retract the PTE table, and map the PMD if asked. + */ if (result == SCAN_PTE_MAPPED_HUGEPAGE) { mmap_read_lock(mm); if (collapse_test_exit_or_disable(mm)) @@ -2831,6 +2856,28 @@ static enum scan_result collapse_single_pmd(unsigned long addr, return result; } +/* + * Try to collapse a single PMD starting at a PMD aligned addr, and return + * the results. + */ +static enum scan_result collapse_single_pmd(unsigned long addr, + struct vm_area_struct *vma, bool *lock_dropped, + struct collapse_control *cc) +{ + struct mm_struct *mm = vma->vm_mm; + enum scan_result result; + + result = collapse_scan_pmd(vma, addr, cc); + if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) + return result; + + /* The collapse takes its own locks, so give this up */ + mmap_read_unlock(mm); + *lock_dropped = true; + + return collapse_run_pmd(mm, addr, result, cc); +} + static void collapse_scan_mm_slot(unsigned int progress_max, enum scan_result *result, struct collapse_control *cc) __releases(&khugepaged_mm_lock) @@ -2973,9 +3020,9 @@ static void khugepaged_do_scan(struct collapse_control *cc) lru_add_drain_all(); + collapse_control_init(cc); collapse_policy_khugepaged(&cc->policy); - cc->progress = 0; while (true) { cond_resched(); @@ -3202,8 +3249,8 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, cc = kmalloc_obj(*cc); if (!cc) return -ENOMEM; + collapse_control_init(cc); collapse_policy_madvise(&cc->policy); - cc->progress = 0; lru_add_drain_all(); From 846175c8027c9e2d03143ce5dbb5ea2eca9440eb Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:24 +0100 Subject: [PATCH 0552/1012] mm/collapse: open-code collapse_single_pmd() in its two callers A scan and a collapse want different things from mmap_lock. The scan reads one PTE table under the lock the caller holds, refuses most of the time, and the caller moves on to the next table without letting go. The collapse allocates, may sleep in writeback and takes the lock for write itself, so the lock it is handed is of no use to it. collapse_single_pmd() kept that boundary inside itself. It dropped the lock on some paths and not others, and reported which by way of a bool its callers had to carry along and then act on. Both callers already act on a drop, khugepaged by ending its walk and madvise_collapse() by looking its VMA up again. The code that has to know is not the code that does it. Open-code it in the two callers. Each scans under the lock it already holds and, when the scan found work, gives the lock up before running the collapse. The scan then has one rule, called locked and returning locked, and the run another, called unlocked. Nothing is left to report, so khugepaged's lock_dropped and madvise_collapse()'s mmap_unlocked both go. The engine never touches a lock it did not take, and how a caller locks its scan is the caller's business alone. khugepaged's walk carries on to the next table while the scan keeps refusing, and ends once a collapse has taken the lock from under it. madvise_collapse() re-finds its VMA after a collapse, which it did before, and now uses a NULL vma to say that it has to. It still reports the drop to its own caller, from the line that does it. Preparation for moving madvise_collapse() out of khugepaged.c: what it needs from the engine is then two calls with one lock rule each. The lock is given up and taken again at the same points as before. No functional change. Link: https://lore.kernel.org/20260928100630.21870-11-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 103 +++++++++++++++++++++++------------------------- 1 file changed, 50 insertions(+), 53 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 87bbba6ce59ab9..1b35778335bead 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -2856,28 +2856,6 @@ static enum scan_result collapse_run_pmd(struct mm_struct *mm, return result; } -/* - * Try to collapse a single PMD starting at a PMD aligned addr, and return - * the results. - */ -static enum scan_result collapse_single_pmd(unsigned long addr, - struct vm_area_struct *vma, bool *lock_dropped, - struct collapse_control *cc) -{ - struct mm_struct *mm = vma->vm_mm; - enum scan_result result; - - result = collapse_scan_pmd(vma, addr, cc); - if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) - return result; - - /* The collapse takes its own locks, so give this up */ - mmap_read_unlock(mm); - *lock_dropped = true; - - return collapse_run_pmd(mm, addr, result, cc); -} - static void collapse_scan_mm_slot(unsigned int progress_max, enum scan_result *result, struct collapse_control *cc) __releases(&khugepaged_mm_lock) @@ -2940,7 +2918,7 @@ static void collapse_scan_mm_slot(unsigned int progress_max, VM_BUG_ON(khugepaged_scan.address & ~HPAGE_PMD_MASK); while (khugepaged_scan.address < hend) { - bool lock_dropped = false; + unsigned long addr; cond_resched(); if (unlikely(collapse_test_exit_or_disable(mm))) @@ -2950,23 +2928,30 @@ static void collapse_scan_mm_slot(unsigned int progress_max, khugepaged_scan.address + HPAGE_PMD_SIZE > hend); - *result = collapse_single_pmd(khugepaged_scan.address, - vma, &lock_dropped, cc); - if (*result == SCAN_SUCCEED) - khugepaged_pages_collapsed++; + addr = khugepaged_scan.address; /* move to next address */ khugepaged_scan.address += HPAGE_PMD_SIZE; - if (lock_dropped) - /* - * We released mmap_lock so break loop. Note - * that we drop mmap_lock before all hugepage - * allocations, so if allocation fails, we are - * guaranteed to break here and report the - * correct result back to caller. - */ - goto breakouterloop_mmap_lock; - if (cc->progress >= progress_max) - goto breakouterloop; + + *result = collapse_scan_pmd(vma, addr, cc); + /* Nothing to do here, and the lock is still ours */ + if (*result != SCAN_SUCCEED && + *result != SCAN_PTE_MAPPED_HUGEPAGE) { + if (cc->progress >= progress_max) + goto breakouterloop; + continue; + } + + /* + * A collapse takes its own locks and is slow enough + * that a writer should not wait behind it, so give the + * lock up. That ends this walk: vma and the mm are + * whatever the collapse leaves them. + */ + mmap_read_unlock(mm); + *result = collapse_run_pmd(mm, addr, *result, cc); + if (*result == SCAN_SUCCEED) + khugepaged_pages_collapsed++; + goto breakouterloop_mmap_lock; } } breakouterloop: @@ -3232,7 +3217,6 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, unsigned long hstart, hend, addr; enum scan_result last_fail = SCAN_FAIL; int thps = 0; - bool mmap_unlocked = false; BUG_ON(vma->vm_start > start); BUG_ON(vma->vm_end < end); @@ -3255,25 +3239,40 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, lru_add_drain_all(); for (addr = hstart; addr < hend; addr += HPAGE_PMD_SIZE) { - enum scan_result result = SCAN_FAIL; + struct vm_area_struct *found; + enum scan_result result; - if (mmap_unlocked) { + /* + * A collapse gives the lock up, so the VMA has to be found + * again after one: it can shrink while nothing is held. A scan + * that finds nothing to collapse leaves the lock alone, so a + * range that is already collapsed walks on without relocking. + */ + if (!vma) { cond_resched(); mmap_read_lock(mm); - mmap_unlocked = false; - *lock_dropped = true; - result = hugepage_vma_revalidate(mm, addr, false, &vma, + result = hugepage_vma_revalidate(mm, addr, false, &found, cc, HPAGE_PMD_ORDER); if (result != SCAN_SUCCEED) { last_fail = result; - goto out_nolock; + goto out_locked; } - + vma = found; hend = min(hend, vma->vm_end & HPAGE_PMD_MASK); } - result = collapse_single_pmd(addr, vma, &mmap_unlocked, cc); + result = collapse_scan_pmd(vma, addr, cc); + /* Nothing to do here, and the lock is still ours */ + if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) + goto tally; + /* The collapse takes its own locks, so give this up */ + mmap_read_unlock(mm); + *lock_dropped = true; + vma = NULL; + + result = collapse_run_pmd(mm, addr, result, cc); +tally: switch (result) { case SCAN_SUCCEED: case SCAN_PMD_MAPPED: @@ -3295,17 +3294,15 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, default: last_fail = result; /* Other error, exit */ - goto out_maybelock; + goto out; } } -out_maybelock: +out: /* Caller expects us to hold mmap_lock on return */ - if (mmap_unlocked) { - *lock_dropped = true; + if (!vma) mmap_read_lock(mm); - } -out_nolock: +out_locked: mmap_assert_locked(mm); kfree(cc); From 950100db87047461e1d7c946fe9627c88b7ac84d Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:25 +0100 Subject: [PATCH 0553/1012] mm/collapse: work out the orders a VMA allows once per VMA The scan asked collapse_possible_orders() for every PTE table, for an answer that is a property of the VMA. Both callers walk a VMA a table at a time, so let them work it out once and pass the mask in. It is only good while the lock that produced it is held, so madvise_collapse() takes it again after every collapse. The mask is then sampled once per VMA rather than once per table. A thp enabled knob written during a walk takes effect one VMA later, and cannot widen a collapse: hugepage_vma_revalidate() tests the order again under the lock the collapse retakes. Link: https://lore.kernel.org/20260928100630.21870-12-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Baolin Wang Assisted-by: LLM Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/khugepaged.c | 32 ++++++++++++++++++-------------- 1 file changed, 18 insertions(+), 14 deletions(-) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 1b35778335bead..c961b8d121f572 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1577,12 +1577,12 @@ static enum scan_result mthp_collapse(struct mm_struct *mm, unsigned long addres } static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma, - unsigned long start_addr, struct collapse_control *cc) + unsigned long start_addr, struct collapse_control *cc, + unsigned long enabled_orders) { const unsigned int max_ptes_shared = collapse_max_ptes_shared(cc, HPAGE_PMD_ORDER); const unsigned int max_ptes_swap = collapse_max_ptes_swap(cc, HPAGE_PMD_ORDER); unsigned int max_ptes_none = collapse_max_ptes_none(cc, vma, HPAGE_PMD_ORDER); - enum tva_type tva_flags = cc->policy.tva_type; struct mm_struct *mm = vma->vm_mm; pmd_t *pmd; pte_t *pte, *_pte, pteval; @@ -1593,7 +1593,6 @@ static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma, struct folio *folio = NULL; unsigned long failed_pfn = -1; unsigned long addr; - unsigned long enabled_orders; spinlock_t *ptl; int node = NUMA_NO_NODE, unmapped = 0; @@ -1607,8 +1606,6 @@ static enum scan_result collapse_scan_anon_pmd(struct vm_area_struct *vma, collapse_scan_reset(cc); - enabled_orders = collapse_possible_orders(vma, vma->vm_flags, tva_flags); - /* * If PMD is the only enabled order, enforce max_ptes_none, otherwise * scan all pages to populate the bitmap for mTHP collapse. The bitmap @@ -2772,7 +2769,8 @@ static void collapse_control_init(struct collapse_control *cc) } static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, - unsigned long addr, struct collapse_control *cc) + unsigned long addr, struct collapse_control *cc, + unsigned long orders) { enum scan_result result; pgoff_t pgoff; @@ -2785,7 +2783,7 @@ static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, } if (vma_is_anonymous(vma)) - return collapse_scan_anon_pmd(vma, addr, cc); + return collapse_scan_anon_pmd(vma, addr, cc, orders); pgoff = linear_page_index(vma, addr); result = collapse_scan_file(vma->vm_mm, addr, vma->vm_file, pgoff, cc); @@ -2895,15 +2893,17 @@ static void collapse_scan_mm_slot(unsigned int progress_max, vma_iter_init(&vmi, mm, khugepaged_scan.address); for_each_vma(vmi, vma) { - unsigned long hstart, hend; + unsigned long hstart, hend, orders; cond_resched(); if (unlikely(collapse_test_exit_or_disable(mm))) { cc->progress++; break; } - if (!collapse_possible_orders(vma, vma->vm_flags, - TVA_KHUGEPAGED)) { + /* One mask for the whole VMA */ + orders = collapse_possible_orders(vma, vma->vm_flags, + TVA_KHUGEPAGED); + if (!orders) { cc->progress++; continue; } @@ -2932,7 +2932,7 @@ static void collapse_scan_mm_slot(unsigned int progress_max, /* move to next address */ khugepaged_scan.address += HPAGE_PMD_SIZE; - *result = collapse_scan_pmd(vma, addr, cc); + *result = collapse_scan_pmd(vma, addr, cc, orders); /* Nothing to do here, and the lock is still ours */ if (*result != SCAN_SUCCEED && *result != SCAN_PTE_MAPPED_HUGEPAGE) { @@ -3214,14 +3214,16 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, { struct collapse_control *cc; struct mm_struct *mm = vma->vm_mm; - unsigned long hstart, hend, addr; + unsigned long hstart, hend, addr, orders; enum scan_result last_fail = SCAN_FAIL; int thps = 0; BUG_ON(vma->vm_start > start); BUG_ON(vma->vm_end < end); - if (!collapse_possible_orders(vma, vma->vm_flags, TVA_FORCED_COLLAPSE)) + orders = collapse_possible_orders(vma, vma->vm_flags, + TVA_FORCED_COLLAPSE); + if (!orders) return -EINVAL; hstart = ALIGN(start, HPAGE_PMD_SIZE); @@ -3259,9 +3261,11 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, } vma = found; hend = min(hend, vma->vm_end & HPAGE_PMD_MASK); + orders = collapse_possible_orders(vma, vma->vm_flags, + TVA_FORCED_COLLAPSE); } - result = collapse_scan_pmd(vma, addr, cc); + result = collapse_scan_pmd(vma, addr, cc, orders); /* Nothing to do here, and the lock is still ours */ if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) goto tally; From e1b77486663d597418d3464484b830bdcab29216 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:26 +0100 Subject: [PATCH 0554/1012] mm/collapse: declare the collapse interface in collapse.h A collapse takes three calls: - collapse_control_init() - set up the control a caller carries; - collapse_scan_pmd() - scan one PTE table, under mmap_lock; - collapse_run_pmd() - collapse what the scan found, no mmap_lock. All three are static in khugepaged.c, as are collapse_possible_orders(), which says what a VMA allows, and the revalidate a caller needs once a collapse has given the mmap_lock up. No other file can ask for a collapse without them. Declare them in collapse.h, with a comment stating the order they are called in and who holds the lock over each step. Each function gets a kerneldoc comment where it is defined: what it takes, what it does, and the lock state on entry and exit. hugepage_vma_revalidate() becomes collapse_vma_revalidate(): it is part of what a collapse offers now, not a helper of the daemon. Preparation for implementing MADV_COLLAPSE in madvise.c. No functional change. Link: https://lore.kernel.org/20260928100630.21870-13-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Zi Yan Assisted-by: LLM Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- mm/collapse.h | 45 ++++++++++++++++++++++++++ mm/khugepaged.c | 86 ++++++++++++++++++++++++++++++++++++++++--------- 2 files changed, 115 insertions(+), 16 deletions(-) diff --git a/mm/collapse.h b/mm/collapse.h index ca7b367c89cb39..b19351fbe19b94 100644 --- a/mm/collapse.h +++ b/mm/collapse.h @@ -114,4 +114,49 @@ struct collapse_control { pgoff_t scan_pgoff; }; +/* Which orders a VMA may collapse to, zero when it may not collapse at all */ +unsigned long collapse_possible_orders(struct vm_area_struct *vma, + vm_flags_t vm_flags, enum tva_type tva_flags); + +/* + * A caller states what it allows in cc->policy and then hands over one PTE + * table's worth of a VMA at a time: + * + * collapse_control_init(cc) once, before the first table + * collapse_scan_pmd(vma, addr, ...) per table + * collapse_run_pmd(mm, addr, result, cc) when a scan found work + * + * The caller holds mmap_lock for reading over the scan and passes an address + * within @vma, aligned to the PTE table to scan. + * + * The scan returns with that lock still held. Almost every table it is + * offered has nothing to collapse, so a caller walks a whole VMA under the one + * lock it took to get there. SCAN_SUCCEED means there is + * something to collapse. SCAN_PTE_MAPPED_HUGEPAGE means the page cache + * already holds the PMD folio and only the PTE table is left to retract. + * Both are work for the run, which is handed what the scan returned; anything + * else is why there is nothing to do. + * + * The run is called without the lock and returns without it, taking what it + * needs in between: what it does -- allocate, isolate, copy, flush -- is slow + * enough that a writer would wait behind it. The caller gives the lock up + * first, and with it @vma and anything derived under it, so a caller carrying + * on has to look up again with collapse_vma_revalidate(). The run revalidates + * for itself rather than trusting what the scan saw. + * + * A scan that found something has to be run: the file side takes a reference on + * the file while it still has the VMA to take it from, and the run is what + * gives it back. + */ +void collapse_control_init(struct collapse_control *cc); +enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, + unsigned long addr, struct collapse_control *cc, + unsigned long orders); +enum scan_result collapse_run_pmd(struct mm_struct *mm, unsigned long addr, + enum scan_result result, struct collapse_control *cc); +enum scan_result collapse_vma_revalidate(struct mm_struct *mm, + unsigned long address, bool expect_anon, + struct vm_area_struct **vmap, struct collapse_control *cc, + unsigned int order); + #endif /* __MM_COLLAPSE_H */ diff --git a/mm/khugepaged.c b/mm/khugepaged.c index c961b8d121f572..86b591ec2c492b 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -476,11 +476,18 @@ void __khugepaged_enter(struct mm_struct *mm) wake_up_interruptible(&khugepaged_wait); } -/* - * Check what orders are possible based on the vma and collapse type. - * This is used to determine if mTHP collapse is a viable option. +/** + * collapse_possible_orders - which orders a VMA may collapse to + * @vma: the VMA + * @vm_flags: its flags, passed separately where they are about to change + * @tva_flags: who is asking, as thp_vma_allowable_orders() spells it + * + * khugepaged may collapse anonymous memory to any enabled order; everything + * else collapses to PMD order only. + * + * Return: the orders as a bitmask, zero when the VMA may not collapse at all. */ -static unsigned long collapse_possible_orders(struct vm_area_struct *vma, +unsigned long collapse_possible_orders(struct vm_area_struct *vma, vm_flags_t vm_flags, enum tva_type tva_flags) { unsigned long orders; @@ -1008,13 +1015,22 @@ static int collapse_find_target_node(struct collapse_control *cc) } #endif -/* - * If mmap_lock temporarily dropped, revalidate vma - * after taking the mmap_lock again. - * Returns enum scan_result value. +/** + * collapse_vma_revalidate - look a VMA up again after mmap_lock was dropped + * @mm: the mm + * @address: an address within the PTE table being collapsed + * @expect_anon: the collapse started on an anonymous VMA + * @vmap: the VMA found, if any + * @cc: the control, for the policy that says who is asking + * @order: the order the collapse is going for + * + * Called with mmap_lock held, for reading or writing, once it has been given up + * and taken back. The VMA has to span the whole PMD whatever @order is; with + * @expect_anon it also has to be anonymous and have an anon_vma. + * + * Return: SCAN_SUCCEED, or why a collapse of @order at @address is off. */ - -static enum scan_result hugepage_vma_revalidate(struct mm_struct *mm, unsigned long address, +enum scan_result collapse_vma_revalidate(struct mm_struct *mm, unsigned long address, bool expect_anon, struct vm_area_struct **vmap, struct collapse_control *cc, unsigned int order) { @@ -1287,7 +1303,7 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, } mmap_read_lock(mm); - result = hugepage_vma_revalidate(mm, pmd_addr, /*expect_anon=*/ true, + result = collapse_vma_revalidate(mm, pmd_addr, /*expect_anon=*/ true, &vma, cc, order); if (result != SCAN_SUCCEED) { mmap_read_unlock(mm); @@ -1322,7 +1338,7 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, * mmap_lock. */ mmap_write_lock(mm); - result = hugepage_vma_revalidate(mm, pmd_addr, /*expect_anon=*/ true, + result = collapse_vma_revalidate(mm, pmd_addr, /*expect_anon=*/ true, &vma, cc, order); if (result != SCAN_SUCCEED) goto out_up_write; @@ -2762,13 +2778,36 @@ static enum scan_result collapse_scan_file(struct mm_struct *mm, return result; } -static void collapse_control_init(struct collapse_control *cc) +/** + * collapse_control_init - set up a control before its first scan + * @cc: the control the caller carries across its scans + * + * cc->policy is the caller's to fill. + */ +void collapse_control_init(struct collapse_control *cc) { cc->progress = 0; cc->scan_file = NULL; } -static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, +/** + * collapse_scan_pmd - scan one PTE table for a collapse candidate + * @vma: the VMA the table belongs to + * @addr: start of the table, PMD aligned + * @cc: the caller's control + * @orders: the orders the caller allows for @vma + * + * Called with mmap_lock held for reading and returns with it still held. + * Almost every table it is offered has nothing to collapse, so a caller walks + * a whole VMA under the one lock it took to get there. + * + * Return: SCAN_SUCCEED when there is something to collapse; + * SCAN_PTE_MAPPED_HUGEPAGE when the page cache already holds the PMD folio and + * only the PTE table is left to retract. Both are work for collapse_run_pmd(), + * which is handed what the scan returned. Anything else is why there is + * nothing to do. + */ +enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, unsigned long addr, struct collapse_control *cc, unsigned long orders) { @@ -2803,7 +2842,22 @@ static enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, return result; } -static enum scan_result collapse_run_pmd(struct mm_struct *mm, +/** + * collapse_run_pmd - collapse the table a scan found work in + * @mm: the mm + * @addr: start of the table, as given to the scan + * @result: what the scan returned + * @cc: the control the scan ran with + * + * Called without mmap_lock and returns without it, taking what it needs in + * between: what it does -- allocate, isolate, copy, flush -- is slow enough + * that a writer would wait behind it. The caller gives the lock up first, + * and with it the VMA and anything derived under it. The run revalidates for + * itself rather than trusting what the scan saw. + * + * Return: what the collapse made of the table. + */ +enum scan_result collapse_run_pmd(struct mm_struct *mm, unsigned long addr, enum scan_result result, struct collapse_control *cc) { @@ -3253,7 +3307,7 @@ int madvise_collapse(struct vm_area_struct *vma, unsigned long start, if (!vma) { cond_resched(); mmap_read_lock(mm); - result = hugepage_vma_revalidate(mm, addr, false, &found, + result = collapse_vma_revalidate(mm, addr, false, &found, cc, HPAGE_PMD_ORDER); if (result != SCAN_SUCCEED) { last_fail = result; From b86ee986de6066d15df32e6d3f3cab5d1a18f186 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Mon, 28 Sep 2026 11:06:27 +0100 Subject: [PATCH 0555/1012] mm/collapse: implement MADV_COLLAPSE in madvise.c MADV_COLLAPSE is a madvise operation, but its implementation sat in khugepaged.c. The daemon's file therefore also held a syscall's worth of code that has nothing to do with the daemon: the walk over the user's range, the per-PMD loop, and the errno translation. Move it to madvise.c, among the operations it belongs with, along with the errno map and the policy it states for itself. It takes a struct madvise_behavior like every one of those operations, which is where the range, the VMA and the lock-dropped flag it used to be handed separately already live. It stays a caller of the interface khugepaged uses, so nothing about the collapse changes. The !CONFIG_TRANSPARENT_HUGEPAGE stub moves in with it. Link: https://lore.kernel.org/20260928100630.21870-14-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Zi Yan Assisted-by: LLM Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Jann Horn --- include/linux/huge_mm.h | 9 --- mm/khugepaged.c | 159 +------------------------------------ mm/madvise.c | 170 +++++++++++++++++++++++++++++++++++++++- 3 files changed, 170 insertions(+), 168 deletions(-) diff --git a/include/linux/huge_mm.h b/include/linux/huge_mm.h index c745f7ad22987f..8ca0fa3be2acb1 100644 --- a/include/linux/huge_mm.h +++ b/include/linux/huge_mm.h @@ -510,8 +510,6 @@ change_huge_pud(struct mmu_gather *tlb, struct vm_area_struct *vma, int hugepage_madvise(struct vm_area_struct *vma, vm_flags_t *vm_flags, int advice); -int madvise_collapse(struct vm_area_struct *vma, unsigned long start, - unsigned long end, bool *lock_dropped); void vma_adjust_trans_huge(struct vm_area_struct *vma, unsigned long start, unsigned long end, struct vm_area_struct *next); spinlock_t *__pmd_trans_huge_lock(pmd_t *pmd, struct vm_area_struct *vma); @@ -715,13 +713,6 @@ static inline int hugepage_madvise(struct vm_area_struct *vma, return -EINVAL; } -static inline int madvise_collapse(struct vm_area_struct *vma, - unsigned long start, - unsigned long end, bool *lock_dropped) -{ - return -EINVAL; -} - static inline void vma_adjust_trans_huge(struct vm_area_struct *vma, unsigned long start, unsigned long end, diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 86b591ec2c492b..8a5c7f38096ef7 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -972,23 +972,6 @@ static void collapse_policy_khugepaged(struct collapse_policy *p) p->tva_type = TVA_KHUGEPAGED; } -/* MADV_COLLAPSE was asked for explicitly, so it is not held to those */ -static void collapse_policy_madvise(struct collapse_policy *p) -{ - p->pmd.max_ptes_none = HPAGE_PMD_NR; - p->pmd.max_ptes_swap = HPAGE_PMD_NR; - p->pmd.max_ptes_shared = HPAGE_PMD_NR; - /* Never read: MADV_COLLAPSE collapses to PMD order only */ - p->sub_pmd = p->pmd; - - p->anon_skip_lazyfree = false; - p->anon_require_referenced = false; - p->file_install_pmd = true; - p->file_writeback_dirty = true; - p->gfp = GFP_TRANSHUGE; - p->tva_type = TVA_FORCED_COLLAPSE; -} - #ifdef CONFIG_NUMA static int collapse_find_target_node(struct collapse_control *cc) { @@ -2857,9 +2840,8 @@ enum scan_result collapse_scan_pmd(struct vm_area_struct *vma, * * Return: what the collapse made of the table. */ -enum scan_result collapse_run_pmd(struct mm_struct *mm, - unsigned long addr, enum scan_result result, - struct collapse_control *cc) +enum scan_result collapse_run_pmd(struct mm_struct *mm, unsigned long addr, + enum scan_result result, struct collapse_control *cc) { struct file *file = cc->scan_file; bool triggered_wb = false; @@ -3230,140 +3212,3 @@ bool current_is_khugepaged(void) { return kthread_func(current) == khugepaged; } - -static int madvise_collapse_errno(enum scan_result r) -{ - /* - * MADV_COLLAPSE breaks from existing madvise(2) conventions to provide - * actionable feedback to caller, so they may take an appropriate - * fallback measure depending on the nature of the failure. - */ - switch (r) { - case SCAN_ALLOC_HUGE_PAGE_FAIL: - return -ENOMEM; - case SCAN_CGROUP_CHARGE_FAIL: - case SCAN_EXCEED_NONE_PTE: - return -EBUSY; - /* Resource temporary unavailable - trying again might succeed */ - case SCAN_PAGE_COUNT: - case SCAN_PAGE_LOCK: - case SCAN_PAGE_LRU: - case SCAN_DEL_PAGE_LRU: - case SCAN_PAGE_FILLED: - case SCAN_PAGE_HAS_PRIVATE: - case SCAN_PAGE_DIRTY_OR_WRITEBACK: - return -EAGAIN; - /* - * Other: Trying again likely not to succeed / error intrinsic to - * specified memory range. khugepaged likely won't be able to collapse - * either. - */ - default: - return -EINVAL; - } -} - -int madvise_collapse(struct vm_area_struct *vma, unsigned long start, - unsigned long end, bool *lock_dropped) -{ - struct collapse_control *cc; - struct mm_struct *mm = vma->vm_mm; - unsigned long hstart, hend, addr, orders; - enum scan_result last_fail = SCAN_FAIL; - int thps = 0; - - BUG_ON(vma->vm_start > start); - BUG_ON(vma->vm_end < end); - - orders = collapse_possible_orders(vma, vma->vm_flags, - TVA_FORCED_COLLAPSE); - if (!orders) - return -EINVAL; - - hstart = ALIGN(start, HPAGE_PMD_SIZE); - hend = ALIGN_DOWN(end, HPAGE_PMD_SIZE); - - if (hstart >= hend) - return 0; - - cc = kmalloc_obj(*cc); - if (!cc) - return -ENOMEM; - collapse_control_init(cc); - collapse_policy_madvise(&cc->policy); - - lru_add_drain_all(); - - for (addr = hstart; addr < hend; addr += HPAGE_PMD_SIZE) { - struct vm_area_struct *found; - enum scan_result result; - - /* - * A collapse gives the lock up, so the VMA has to be found - * again after one: it can shrink while nothing is held. A scan - * that finds nothing to collapse leaves the lock alone, so a - * range that is already collapsed walks on without relocking. - */ - if (!vma) { - cond_resched(); - mmap_read_lock(mm); - result = collapse_vma_revalidate(mm, addr, false, &found, - cc, HPAGE_PMD_ORDER); - if (result != SCAN_SUCCEED) { - last_fail = result; - goto out_locked; - } - vma = found; - hend = min(hend, vma->vm_end & HPAGE_PMD_MASK); - orders = collapse_possible_orders(vma, vma->vm_flags, - TVA_FORCED_COLLAPSE); - } - - result = collapse_scan_pmd(vma, addr, cc, orders); - /* Nothing to do here, and the lock is still ours */ - if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) - goto tally; - - /* The collapse takes its own locks, so give this up */ - mmap_read_unlock(mm); - *lock_dropped = true; - vma = NULL; - - result = collapse_run_pmd(mm, addr, result, cc); -tally: - switch (result) { - case SCAN_SUCCEED: - case SCAN_PMD_MAPPED: - ++thps; - break; - /* Whitelisted set of results where continuing OK */ - case SCAN_NO_PTE_TABLE: - case SCAN_PTE_NON_PRESENT: - case SCAN_PTE_UFFD: - case SCAN_LACK_REFERENCED_PAGE: - case SCAN_PAGE_NULL: - case SCAN_PAGE_COUNT: - case SCAN_PAGE_LOCK: - case SCAN_PAGE_COMPOUND: - case SCAN_PAGE_LRU: - case SCAN_DEL_PAGE_LRU: - last_fail = result; - break; - default: - last_fail = result; - /* Other error, exit */ - goto out; - } - } - -out: - /* Caller expects us to hold mmap_lock on return */ - if (!vma) - mmap_read_lock(mm); -out_locked: - mmap_assert_locked(mm); - kfree(cc); - - return thps == ((hend - hstart) >> HPAGE_PMD_SHIFT) ? 0 - : madvise_collapse_errno(last_fail); -} diff --git a/mm/madvise.c b/mm/madvise.c index 963337f93a7a11..acb5215e2f079f 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -38,6 +38,7 @@ #include "internal.h" #include "swap.h" +#include "collapse.h" #define __MADV_SET_ANON_VMA_NAME (-1) @@ -906,6 +907,172 @@ bool madvise_dontneed_free_valid_vma(struct madvise_behavior *madv_behavior) return true; } +#ifdef CONFIG_TRANSPARENT_HUGEPAGE + +/* MADV_COLLAPSE was asked for explicitly, so it is not held to those */ +static void collapse_policy_madvise(struct collapse_policy *p) +{ + p->pmd.max_ptes_none = HPAGE_PMD_NR; + p->pmd.max_ptes_swap = HPAGE_PMD_NR; + p->pmd.max_ptes_shared = HPAGE_PMD_NR; + /* Never read: MADV_COLLAPSE collapses to PMD order only */ + p->sub_pmd = p->pmd; + + p->anon_skip_lazyfree = false; + p->anon_require_referenced = false; + p->file_install_pmd = true; + p->file_writeback_dirty = true; + p->gfp = GFP_TRANSHUGE; + p->tva_type = TVA_FORCED_COLLAPSE; +} + +static int madvise_collapse_errno(enum scan_result r) +{ + /* + * MADV_COLLAPSE breaks from existing madvise(2) conventions to provide + * actionable feedback to caller, so they may take an appropriate + * fallback measure depending on the nature of the failure. + */ + switch (r) { + case SCAN_ALLOC_HUGE_PAGE_FAIL: + return -ENOMEM; + case SCAN_CGROUP_CHARGE_FAIL: + case SCAN_EXCEED_NONE_PTE: + return -EBUSY; + /* Resource temporary unavailable - trying again might succeed */ + case SCAN_PAGE_COUNT: + case SCAN_PAGE_LOCK: + case SCAN_PAGE_LRU: + case SCAN_DEL_PAGE_LRU: + case SCAN_PAGE_FILLED: + case SCAN_PAGE_HAS_PRIVATE: + case SCAN_PAGE_DIRTY_OR_WRITEBACK: + return -EAGAIN; + /* + * Other: Trying again likely not to succeed / error intrinsic to + * specified memory range. khugepaged likely won't be able to collapse + * either. + */ + default: + return -EINVAL; + } +} + +static int madvise_collapse(struct madvise_behavior *madv_behavior) +{ + struct madvise_behavior_range *range = &madv_behavior->range; + struct vm_area_struct *vma = madv_behavior->vma; + struct mm_struct *mm = madv_behavior->mm; + struct collapse_control *cc; + unsigned long hstart, hend, addr, orders; + enum scan_result last_fail = SCAN_FAIL; + int thps = 0; + + BUG_ON(vma->vm_start > range->start); + BUG_ON(vma->vm_end < range->end); + + orders = collapse_possible_orders(vma, vma->vm_flags, + TVA_FORCED_COLLAPSE); + if (!orders) + return -EINVAL; + + hstart = ALIGN(range->start, HPAGE_PMD_SIZE); + hend = ALIGN_DOWN(range->end, HPAGE_PMD_SIZE); + + if (hstart >= hend) + return 0; + + cc = kmalloc_obj(*cc); + if (!cc) + return -ENOMEM; + collapse_control_init(cc); + collapse_policy_madvise(&cc->policy); + + lru_add_drain_all(); + + for (addr = hstart; addr < hend; addr += HPAGE_PMD_SIZE) { + struct vm_area_struct *found; + enum scan_result result; + + /* + * A collapse gives the lock up, so the VMA has to be found + * again after one: it can shrink while nothing is held. A scan + * that finds nothing to collapse leaves the lock alone, so a + * range that is already collapsed walks on without relocking. + */ + if (!vma) { + cond_resched(); + mmap_read_lock(mm); + result = collapse_vma_revalidate(mm, addr, false, &found, + cc, HPAGE_PMD_ORDER); + if (result != SCAN_SUCCEED) { + last_fail = result; + goto out_locked; + } + vma = found; + hend = min(hend, vma->vm_end & HPAGE_PMD_MASK); + orders = collapse_possible_orders(vma, vma->vm_flags, + TVA_FORCED_COLLAPSE); + } + + result = collapse_scan_pmd(vma, addr, cc, orders); + /* Nothing to do here, and the lock is still ours */ + if (result != SCAN_SUCCEED && result != SCAN_PTE_MAPPED_HUGEPAGE) + goto tally; + + /* The collapse takes its own locks, so give this up */ + mmap_read_unlock(mm); + mark_mmap_lock_dropped(madv_behavior); + vma = NULL; + + result = collapse_run_pmd(mm, addr, result, cc); +tally: + switch (result) { + case SCAN_SUCCEED: + case SCAN_PMD_MAPPED: + ++thps; + break; + /* Whitelisted set of results where continuing OK */ + case SCAN_NO_PTE_TABLE: + case SCAN_PTE_NON_PRESENT: + case SCAN_PTE_UFFD: + case SCAN_LACK_REFERENCED_PAGE: + case SCAN_PAGE_NULL: + case SCAN_PAGE_COUNT: + case SCAN_PAGE_LOCK: + case SCAN_PAGE_COMPOUND: + case SCAN_PAGE_LRU: + case SCAN_DEL_PAGE_LRU: + last_fail = result; + break; + default: + last_fail = result; + /* Other error, exit */ + goto out; + } + } + +out: + /* Caller expects us to hold mmap_lock on return */ + if (!vma) + mmap_read_lock(mm); +out_locked: + mmap_assert_locked(mm); + kfree(cc); + + return thps == ((hend - hstart) >> HPAGE_PMD_SHIFT) ? 0 + : madvise_collapse_errno(last_fail); +} + +#else /* CONFIG_TRANSPARENT_HUGEPAGE */ + +static int madvise_collapse(struct madvise_behavior *madv_behavior) +{ + return -EINVAL; +} + +#endif /* CONFIG_TRANSPARENT_HUGEPAGE */ + static long madvise_dontneed_free(struct madvise_behavior *madv_behavior) { struct mm_struct *mm = madv_behavior->mm; @@ -1373,8 +1540,7 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior) case MADV_DONTNEED_LOCKED: return madvise_dontneed_free(madv_behavior); case MADV_COLLAPSE: - return madvise_collapse(vma, range->start, range->end, - &madv_behavior->lock_dropped); + return madvise_collapse(madv_behavior); case MADV_GUARD_INSTALL: return madvise_guard_install(madv_behavior); case MADV_GUARD_REMOVE: From 67f463b2139a486b0e3976d328c275c21444b2c5 Mon Sep 17 00:00:00 2001 From: Huaisheng Ye Date: Wed, 9 Sep 2026 15:46:42 +0800 Subject: [PATCH 0556/1012] mm/hugetlb: account for allowed nodes when gathering surplus pages Hugetlb reservations are accounted globally, but hugetlb_acct_memory() also verifies that the current cpuset and MPOL_BIND policy contain enough free huge pages to add a new reservation. gather_surplus_pages() calculates its allocation shortfall from the global free and reserved counters. If the global pool has enough free pages, but those pages reside outside the nodes allowed by the task, it allocates no surplus pages. The subsequent allowed_mems_nr() check then rejects the reservation and mmap() fails with ENOMEM, even when nr_overcommit_hugepages permits allocating surplus pages on the allowed nodes. Calculate both the global shortfall and the shortfall within the allowed nodes, and allocate the larger of the two. Include surplus pages allocated outside hugetlb_lock in both calculations when rechecking after reacquiring the lock. These pages are constrained by alloc_nodemask, so they satisfy both shortages. Easy way to reproduce this issue with 2+ NUMA nodes system: # echo 0 > /sys/kernel/mm/hugepages/hugepages-2048kB/nr_hugepage # echo 3 > /sys/devices/system/node/node0/hugepages/hugepages-2048kB/nr_hugepages # echo 1 > /sys/kernel/mm/hugepages/hugepages-2048kB/nr_overcommit_hugepages # cd tools/testing/selftests/mm # numactl --membind=1 ./hugetlb-mmap 2 21 TAP version 13 # [INFO] detected hugetlb page size: 2048 KiB # [INFO] detected hugetlb page size: 1048576 KiB # 2048 kB hugepages 1..2 # Mapping 2 Mbytes Bail out! mmap: Cannot allocate memory (12) # Planned tests != run tests (2 != 0) # Totals: pass:0 fail:0 xfail:0 xpass:0 skip:0 error:0 This fixes hugetlb mappings when, for example, a task runs with MPOL_BIND on Node 1 while the existing free huge pages are on Node 0. Similar issue also could be found in ltp if the free pages of global pool reside outside the nodes allowed by the application. # cd ltp/testcases/kernel/mem/hugetlb/hugemmap/ # numactl --cpunodebind=0 --membind=1 ./hugemmap10 Link: https://lore.kernel.org/20260909074642.7308-1-yehuaisheng@open-hieco.net Fixes: e4e574b767ba ("hugetlb: Try to grow hugetlb pool for MAP_SHARED mappings") Signed-off-by: Huaisheng Ye Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: David Hildenbrand Cc: Oscar Salvador --- mm/hugetlb.c | 21 +++++++++++++++++---- 1 file changed, 17 insertions(+), 4 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 5e05711f352d86..fd00141b089a98 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -125,6 +125,7 @@ struct mutex *hugetlb_fault_mutex_table __ro_after_init; /* Forward declaration */ static int hugetlb_acct_memory(struct hstate *h, long delta); +static unsigned int allowed_mems_nr(struct hstate *h); static void hugetlb_vma_lock_free(struct vm_area_struct *vma); static void hugetlb_vma_lock_alloc(struct vm_area_struct *vma); static void __hugetlb_vma_unlock_write_free(struct vm_area_struct *vma); @@ -2252,6 +2253,19 @@ static nodemask_t *policy_mbind_nodemask(gfp_t gfp) return NULL; } +/* + * Reservations are globally accounted, but they must also be backed by free + * pages on nodes allowed by the current cpuset and MPOL_BIND policy. + */ +static long surplus_pages_needed(struct hstate *h, long delta, long allocated) +{ + long global_free = (long)h->free_huge_pages + allocated; + long allowed_free = (long)allowed_mems_nr(h) + allocated; + + return max((long)h->resv_huge_pages + delta - global_free, + delta - allowed_free); +} + /* * Increase the hugetlb pool such that it can accommodate a reservation * of size 'delta'. @@ -2274,7 +2288,7 @@ static int gather_surplus_pages(struct hstate *h, long delta) alloc_nodemask = cpuset_current_mems_allowed; lockdep_assert_held(&hugetlb_lock); - needed = (h->resv_huge_pages + delta) - h->free_huge_pages; + needed = surplus_pages_needed(h, delta, 0); if (needed <= 0) { h->resv_huge_pages += delta; return 0; @@ -2305,11 +2319,10 @@ static int gather_surplus_pages(struct hstate *h, long delta) /* * After retaking hugetlb_lock, we need to recalculate 'needed' - * because either resv_huge_pages or free_huge_pages may have changed. + * because either resv_huge_pages or the free page counts may have changed. */ spin_lock_irq(&hugetlb_lock); - needed = (h->resv_huge_pages + delta) - - (h->free_huge_pages + allocated); + needed = surplus_pages_needed(h, delta, allocated); if (needed > 0) { if (alloc_ok) goto retry; From 1a30d4996693ed5d2d421000cdcf892760208da6 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Thu, 10 Sep 2026 14:44:15 +0800 Subject: [PATCH 0557/1012] selftests/mm: fix ptrace PEEKDATA check in memfd_secret test try_ptrace() treats PTRACE_PEEKDATA return value as a boolean check. A successful read returns non-zero data (memory filled with 0x55), causing the test to incorrectly report PASS when secret memory protection is broken. Check the return value against -1 instead. The test should only pass when PTRACE_PEEKDATA fails, which means secret memory protection works. Link: https://lore.kernel.org/20260910064415.71623-1-hongfu.li@linux.dev Fixes: 76fe17ef588a ("secretmem: test: add basic selftest for memfd_secret(2)") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes (ARM) Acked-by: Mike Rapoport (Microsoft) Cc: David Hildenbrand Cc: James Bottomley Cc: Liam R. Howlett Cc: Michal Hocko Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/memfd_secret.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/memfd_secret.c b/tools/testing/selftests/mm/memfd_secret.c index c55d84c5e613a4..aef774be87f14a 100644 --- a/tools/testing/selftests/mm/memfd_secret.c +++ b/tools/testing/selftests/mm/memfd_secret.c @@ -145,7 +145,8 @@ static void try_ptrace(int fd, int pipefd[2]) exit(KSFT_FAIL); } - if (ptrace(PTRACE_PEEKDATA, ppid, mem, 0)) + /* PEEKDATA on secret memory must fail, else protection is broken. */ + if (ptrace(PTRACE_PEEKDATA, ppid, mem, 0) == -1) exit(KSFT_PASS); exit(KSFT_FAIL); From a250847dfb061014eba0d6ffe7bfd66e258d1df1 Mon Sep 17 00:00:00 2001 From: Qiqi Liu Date: Tue, 8 Sep 2026 18:23:56 +0800 Subject: [PATCH 0558/1012] mm: page_alloc: add missing hooks to bulk allocation path The bulk allocation path in alloc_pages_bulk_noprof() currently misses trace_mm_page_alloc() and kmsan_alloc_page() calls, leaving bulk-allocated pages invisible to ftrace/BPF/perf and leaving KMSAN shadow memory stale. Add both calls in the bulk loop to match the standard allocation path, placing them before set_page_refcounted() for consistency. The gfp mask passed to kmsan_alloc_page() is stripped of __GFP_RECLAIM because the bulk loop runs under the PCP spinlock, and KMSAN's stack depot allocation must not sleep. Both are no-ops when their respective features are disabled, so there is no overhead in production kernels. Link: https://lore.kernel.org/20260908102356.344075-1-liuqiqi@kylinos.cn Signed-off-by: Qiqi Liu Signed-off-by: Andrew Morton Suggested-by: Gregory Price Reviewed-by: Vlastimil Babka (SUSE) Reviewed-by: Gregory Price (Meta) Reviewed-by: Zi Yan Cc: Brendan Jackman Cc: Johannes Weiner Cc: Michal Hocko Cc: Suren Baghdasaryan --- mm/page_alloc.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index be1b6672368296..5dc259788bcc12 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -5300,6 +5300,8 @@ unsigned long alloc_pages_bulk_noprof(gfp_t gfp, int preferred_nid, nr_account++; prep_new_page(page, 0, gfp, ALLOC_DEFAULT); + trace_mm_page_alloc(page, 0, gfp, ac.migratetype); + kmsan_alloc_page(page, 0, gfp & ~__GFP_RECLAIM); set_page_refcounted(page); page_array[nr_populated++] = page; } From 78123d832ea312ee72ab5aa528ac764363fe3d97 Mon Sep 17 00:00:00 2001 From: Sergey Senozhatsky Date: Mon, 7 Sep 2026 19:57:28 +0900 Subject: [PATCH 0559/1012] zram: convert to SG-list zsmalloc object read API Patch series "zsmallc: remove old object read API". zram remains the only user of old zsmalloc object read API. This series removes the old API and converts zram to use the new SG-list based API. This patch (of 2): zram remains the last user of old zsmalloc object read API, that performed linearisation on the zsmalloc side. There is a new SG-list API, that has a bunch of benefits. Switch zram to SG-list zsmalloc object read API. Link: https://lore.kernel.org/20260907105739.1793316-1-senozhatsky@chromium.org Link: https://lore.kernel.org/20260907105739.1793316-2-senozhatsky@chromium.org Signed-off-by: Sergey Senozhatsky Signed-off-by: Andrew Morton Cc: Minchan Kim Cc: Nhat Pham --- drivers/block/zram/zcomp.c | 23 ++++++++++++-- drivers/block/zram/zcomp.h | 4 ++- drivers/block/zram/zram_drv.c | 56 ++++++++++++++++++++--------------- 3 files changed, 55 insertions(+), 28 deletions(-) diff --git a/drivers/block/zram/zcomp.c b/drivers/block/zram/zcomp.c index 974c4691887e6f..028e2f0f587e3f 100644 --- a/drivers/block/zram/zcomp.c +++ b/drivers/block/zram/zcomp.c @@ -7,6 +7,8 @@ #include #include #include +#include +#include #include #include @@ -158,17 +160,32 @@ int zcomp_compress(struct zcomp *comp, struct zcomp_strm *zstrm, } int zcomp_decompress(struct zcomp *comp, struct zcomp_strm *zstrm, - const void *src, unsigned int src_len, void *dst) + struct scatterlist *sg, unsigned int src_len, void *dst) { struct zcomp_req req = { - .src = src, .dst = dst, .src_len = src_len, .dst_len = PAGE_SIZE, }; + void *src = NULL; + int ret; might_sleep(); - return comp->ops->decompress(comp->params, &zstrm->ctx, &req); + + if (sg_is_last(sg)) { + /* the object is contained within one page, read it in-place */ + src = kmap_local_page(sg_page(sg)); + req.src = src + sg->offset; + } else { + /* the object spans two pages, linearize it into local copy */ + sg_copy_to_buffer(sg, 2, zstrm->local_copy, src_len); + req.src = zstrm->local_copy; + } + + ret = comp->ops->decompress(comp->params, &zstrm->ctx, &req); + if (src) + kunmap_local(src); + return ret; } int zcomp_cpu_up_prepare(unsigned int cpu, struct hlist_node *node) diff --git a/drivers/block/zram/zcomp.h b/drivers/block/zram/zcomp.h index 81a0f3f6ff4871..f38fd31f9e4e02 100644 --- a/drivers/block/zram/zcomp.h +++ b/drivers/block/zram/zcomp.h @@ -5,6 +5,8 @@ #include +struct scatterlist; + #define ZCOMP_PARAM_NOT_SET INT_MIN struct deflate_params { @@ -91,6 +93,6 @@ void zcomp_stream_put(struct zcomp_strm *zstrm); int zcomp_compress(struct zcomp *comp, struct zcomp_strm *zstrm, const void *src, unsigned int *dst_len); int zcomp_decompress(struct zcomp *comp, struct zcomp_strm *zstrm, - const void *src, unsigned int src_len, void *dst); + struct scatterlist *sg, unsigned int src_len, void *dst); #endif /* _ZCOMP_H_ */ diff --git a/drivers/block/zram/zram_drv.c b/drivers/block/zram/zram_drv.c index 2359eaa6f53184..024402438bd1d4 100644 --- a/drivers/block/zram/zram_drv.c +++ b/drivers/block/zram/zram_drv.c @@ -32,6 +32,7 @@ #include #include #include +#include #include #include @@ -1347,9 +1348,9 @@ static int decompress_bdev_page(struct zram *zram, struct page *page, unsigned long index) { struct zcomp_strm *zstrm; + struct scatterlist sg[1]; unsigned int size; int ret, prio; - void *src; slot_lock(zram, index); /* Since slot was unlocked we need to make sure it's still ZRAM_WB */ @@ -1368,13 +1369,18 @@ static int decompress_bdev_page(struct zram *zram, struct page *page, size = get_slot_size(zram, index); prio = get_slot_comp_priority(zram, index); + sg_init_table(sg, 1); + sg_set_page(sg, page, size, 0); + zstrm = zcomp_stream_get(zram->comps[prio]); - src = kmap_local_page(page); - ret = zcomp_decompress(zram->comps[prio], zstrm, src, size, + ret = zcomp_decompress(zram->comps[prio], zstrm, sg, size, zstrm->local_copy); - if (!ret) - copy_page(src, zstrm->local_copy); - kunmap_local(src); + if (!ret) { + void *dst = kmap_local_page(page); + + copy_page(dst, zstrm->local_copy); + kunmap_local(dst); + } zcomp_stream_put(zstrm); slot_unlock(zram, index); @@ -2086,15 +2092,19 @@ static int read_same_filled_page(struct zram *zram, struct page *page, static int read_incompressible_page(struct zram *zram, struct page *page, unsigned long index) { + struct scatterlist sg[2]; unsigned long handle; void *src, *dst; handle = get_slot_handle(zram, index); - src = zs_obj_read_begin(zram->mem_pool, handle, PAGE_SIZE, NULL); + zs_obj_read_sg_begin(zram->mem_pool, handle, sg, PAGE_SIZE); + /* an incompressible object never spans two pages */ + src = kmap_local_page(sg_page(sg)); dst = kmap_local_page(page); copy_page(dst, src); kunmap_local(dst); - zs_obj_read_end(zram->mem_pool, handle, PAGE_SIZE, src); + kunmap_local(src); + zs_obj_read_sg_end(zram->mem_pool, handle); return 0; } @@ -2103,9 +2113,10 @@ static int read_compressed_page(struct zram *zram, struct page *page, unsigned long index) { struct zcomp_strm *zstrm; + struct scatterlist sg[2]; unsigned long handle; unsigned int size; - void *src, *dst; + void *dst; int ret, prio; handle = get_slot_handle(zram, index); @@ -2113,12 +2124,11 @@ static int read_compressed_page(struct zram *zram, struct page *page, prio = get_slot_comp_priority(zram, index); zstrm = zcomp_stream_get(zram->comps[prio]); - src = zs_obj_read_begin(zram->mem_pool, handle, size, - zstrm->local_copy); + zs_obj_read_sg_begin(zram->mem_pool, handle, sg, size); dst = kmap_local_page(page); - ret = zcomp_decompress(zram->comps[prio], zstrm, src, size, dst); + ret = zcomp_decompress(zram->comps[prio], zstrm, sg, size, dst); kunmap_local(dst); - zs_obj_read_end(zram->mem_pool, handle, size, src); + zs_obj_read_sg_end(zram->mem_pool, handle); zcomp_stream_put(zstrm); return ret; @@ -2128,25 +2138,23 @@ static int read_compressed_page(struct zram *zram, struct page *page, static int read_from_zspool_raw(struct zram *zram, struct page *page, unsigned long index) { - struct zcomp_strm *zstrm; + struct scatterlist sg[2]; unsigned long handle; unsigned int size; - void *src; + void *dst; handle = get_slot_handle(zram, index); size = get_slot_size(zram, index); /* - * We need to get stream just for ->local_copy buffer, in - * case if object spans two physical pages. No decompression - * takes place here, as we read raw compressed data. + * No decompression takes place here, we copy out raw compressed + * data directly into the destination page. */ - zstrm = zcomp_stream_get(zram->comps[ZRAM_PRIMARY_COMP]); - src = zs_obj_read_begin(zram->mem_pool, handle, size, - zstrm->local_copy); - memcpy_to_page(page, 0, src, size); - zs_obj_read_end(zram->mem_pool, handle, size, src); - zcomp_stream_put(zstrm); + zs_obj_read_sg_begin(zram->mem_pool, handle, sg, size); + dst = kmap_local_page(page); + sg_copy_to_buffer(sg, sg_nents(sg), dst, size); + kunmap_local(dst); + zs_obj_read_sg_end(zram->mem_pool, handle); memzero_page(page, size, PAGE_SIZE - size); From 7579ef6cc9648e3502a29e5e039092b90ea5bd96 Mon Sep 17 00:00:00 2001 From: Sergey Senozhatsky Date: Mon, 7 Sep 2026 19:57:29 +0900 Subject: [PATCH 0560/1012] zsmalloc: remove old object read API There are no users left of the old API, all have switched to SG-list object read API. Remove it. Link: https://lore.kernel.org/20260907105739.1793316-3-senozhatsky@chromium.org Signed-off-by: Sergey Senozhatsky Signed-off-by: Andrew Morton Cc: Minchan Kim Cc: Nhat Pham --- include/linux/zsmalloc.h | 4 --- mm/zsmalloc.c | 77 ---------------------------------------- 2 files changed, 81 deletions(-) diff --git a/include/linux/zsmalloc.h b/include/linux/zsmalloc.h index 478410c880b1fe..5b7298a5026b8d 100644 --- a/include/linux/zsmalloc.h +++ b/include/linux/zsmalloc.h @@ -40,10 +40,6 @@ unsigned int zs_lookup_class_index(struct zs_pool *pool, unsigned int size); void zs_pool_stats(struct zs_pool *pool, struct zs_pool_stats *stats); -void *zs_obj_read_begin(struct zs_pool *pool, unsigned long handle, - size_t mem_len, void *local_copy); -void zs_obj_read_end(struct zs_pool *pool, unsigned long handle, - size_t mem_len, void *handle_mem); void zs_obj_read_sg_begin(struct zs_pool *pool, unsigned long handle, struct scatterlist *sg, size_t mem_len); void zs_obj_read_sg_end(struct zs_pool *pool, unsigned long handle); diff --git a/mm/zsmalloc.c b/mm/zsmalloc.c index 825022a7a328fe..11be37c4317189 100644 --- a/mm/zsmalloc.c +++ b/mm/zsmalloc.c @@ -1134,83 +1134,6 @@ unsigned long zs_get_total_pages(struct zs_pool *pool) } EXPORT_SYMBOL_GPL(zs_get_total_pages); -void *zs_obj_read_begin(struct zs_pool *pool, unsigned long handle, - size_t mem_len, void *local_copy) -{ - struct zspage *zspage; - struct zpdesc *zpdesc; - unsigned long obj, off; - unsigned int obj_idx; - struct size_class *class; - void *addr; - - /* Guarantee we can get zspage from handle safely */ - read_lock(&pool->lock); - obj = handle_to_obj(handle); - obj_to_location(obj, &zpdesc, &obj_idx); - zspage = get_zspage(zpdesc); - - /* Make sure migration doesn't move any pages in this zspage */ - zspage_read_lock(zspage); - read_unlock(&pool->lock); - - class = zspage_class(pool, zspage); - off = offset_in_page(class->size * obj_idx); - - if (!ZsHugePage(zspage)) - off += ZS_HANDLE_SIZE; - - if (off + mem_len <= PAGE_SIZE) { - /* this object is contained entirely within a page */ - addr = kmap_local_zpdesc(zpdesc); - addr += off; - } else { - size_t sizes[2]; - - /* this object spans two pages */ - sizes[0] = PAGE_SIZE - off; - sizes[1] = mem_len - sizes[0]; - addr = local_copy; - - memcpy_from_page(addr, zpdesc_page(zpdesc), - off, sizes[0]); - zpdesc = get_next_zpdesc(zpdesc); - memcpy_from_page(addr + sizes[0], - zpdesc_page(zpdesc), - 0, sizes[1]); - } - - return addr; -} -EXPORT_SYMBOL_GPL(zs_obj_read_begin); - -void zs_obj_read_end(struct zs_pool *pool, unsigned long handle, - size_t mem_len, void *handle_mem) -{ - struct zspage *zspage; - struct zpdesc *zpdesc; - unsigned long obj, off; - unsigned int obj_idx; - struct size_class *class; - - obj = handle_to_obj(handle); - obj_to_location(obj, &zpdesc, &obj_idx); - zspage = get_zspage(zpdesc); - class = zspage_class(pool, zspage); - off = offset_in_page(class->size * obj_idx); - - if (!ZsHugePage(zspage)) - off += ZS_HANDLE_SIZE; - - if (off + mem_len <= PAGE_SIZE) { - handle_mem -= off; - kunmap_local(handle_mem); - } - - zspage_read_unlock(zspage); -} -EXPORT_SYMBOL_GPL(zs_obj_read_end); - void zs_obj_read_sg_begin(struct zs_pool *pool, unsigned long handle, struct scatterlist *sg, size_t mem_len) { From af9db0190d280c447ef1bf067a7b6ed8ce6b6a2f Mon Sep 17 00:00:00 2001 From: Kemeng Shi Date: Mon, 7 Sep 2026 17:13:53 +0800 Subject: [PATCH 0561/1012] mm, swap: fix potential NULL dereference when trying a sleep table allocation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "mm, swap: some random fixes and cleanups", v3. This series contains some random fixes and cleanups. More details can be found in respective patches. This patch (of 4): The root cause of this issue is because multi-tables are updated in non atomic context. To be more specific, the issue could be triggerred as following: swap_alloc_fast swap_cluster_populate() /* Try a sleep allocation */ spin_unlock(&ci->lock); swap_cluster_alloc_table() rcu_assign_pointer(ci->table, table); ci = swap_cluster_lock(si, offset) cluster_is_usable(ci, order) if (!cluster_table_is_alloced(ci)) // ok alloc_swap_scan_cluster() cluster_scan_range() __swap_table_get() /* free table when more table allocation fails */ ci->memcg_table = kzalloc_obj(*ci->memcg_table, gfp); if (!ci->memcg_table) swap_cluster_free_table() rcu_assign_pointer(ci->table, NULL); table = rcu_dereference_check(ci->table, lockdep_is_held(&ci->lock)); atomic_long_read(&table[off]); // NULL dereference Since memory order guarantee between ci->table, as well as between ci->table and ci->zero_bitmap, fix the issue by making tables visible at the end of swap_cluster_populate(). Current memory order guarantee is as following: On write side: rcu_assign_pointer(ci->table, table) will offer release to ensure zero_bitmap and memcg_table visible before ci->table. On read side: folio_alloc_swap swap_alloc_fast/swap_alloc_slow /* ci->table: protected by cluster lock */ swap_cluster_lock cluster_is_usable ... __swap_table_set ... swap_cluster_unlock mem_cgroup_try_charge_swap ... /* memcg_table: protected by cluster lock */ swap_cluster_get_and_lock __swap_cgroup_set swap_cluster_unlock swap_writeout swap_zeromap_folio_set /* zero_bitmap: protected by cluster lock */ swap_cluster_get_and_lock __swap_table_set_zero swap_cluster_unlock Link: https://lore.kernel.org/20260907091356.53026-1-shikemeng@huaweicloud.com Link: https://lore.kernel.org/20260907091356.53026-2-shikemeng@huaweicloud.com Fixes: b197d41462c2 ("mm/memcg, swap: store cgroup id in cluster table directly") Signed-off-by: Kemeng Shi Signed-off-by: Andrew Morton Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: Nhat Pham Cc: Kairui Song Cc: Luiz Capitulino Cc: Youngjun Park --- mm/swapfile.c | 30 ++++++++++++++++++++---------- 1 file changed, 20 insertions(+), 10 deletions(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 2c263563b70ebc..8df8b2c2e5b405 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -416,6 +416,17 @@ static void swap_cluster_free_table_folio_rcu_cb(struct rcu_head *head) folio_put(folio); } +static void swap_cluster_free_count_table(struct swap_table *table) +{ + if (!SWP_TABLE_USE_PAGE) { + kmem_cache_free(swap_table_cachep, table); + return; + } + + call_rcu(&(folio_page(virt_to_folio(table), 0)->rcu_head), + swap_cluster_free_table_folio_rcu_cb); +} + static void swap_cluster_free_table(struct swap_cluster_info *ci) { struct swap_table *table; @@ -435,13 +446,7 @@ static void swap_cluster_free_table(struct swap_cluster_info *ci) return; rcu_assign_pointer(ci->table, NULL); - if (!SWP_TABLE_USE_PAGE) { - kmem_cache_free(swap_table_cachep, table); - return; - } - - call_rcu(&(folio_page(virt_to_folio(table), 0)->rcu_head), - swap_cluster_free_table_folio_rcu_cb); + swap_cluster_free_count_table(table); } static int swap_cluster_alloc_table(struct swap_cluster_info *ci, gfp_t gfp) @@ -464,14 +469,12 @@ static int swap_cluster_alloc_table(struct swap_cluster_info *ci, gfp_t gfp) if (!table) return -ENOMEM; - rcu_assign_pointer(ci->table, table); - #ifdef CONFIG_MEMCG if (!mem_cgroup_disabled()) { VM_WARN_ON_ONCE(ci->memcg_table); ci->memcg_table = kzalloc_obj(*ci->memcg_table, gfp); if (!ci->memcg_table) { - swap_cluster_free_table(ci); + swap_cluster_free_count_table(table); return -ENOMEM; } } @@ -482,9 +485,16 @@ static int swap_cluster_alloc_table(struct swap_cluster_info *ci, gfp_t gfp) ci->zero_bitmap = bitmap_zalloc(SWAPFILE_CLUSTER, gfp); if (!ci->zero_bitmap) { swap_cluster_free_table(ci); + swap_cluster_free_count_table(table); return -ENOMEM; } #endif + + /* + * Make tables visible to cluster_is_usable() after everything is + * ready. + */ + rcu_assign_pointer(ci->table, table); return 0; } From cdd3522a5c6c2b82e15ceab7da96c9e475160594 Mon Sep 17 00:00:00 2001 From: Kemeng Shi Date: Mon, 7 Sep 2026 17:13:54 +0800 Subject: [PATCH 0562/1012] mm, swap: move setup_swap_clusters_info() after SWP_SOLIDSTATE initialization In setup_swap_clusters_info(), SWP_SOLIDSTATE is used to decide global_cluster allocation. Move setup_swap_clusters_info() after SWP_SOLIDSTATE initialization to avoid unneeded global_cluster allocation. Link: https://lore.kernel.org/20260907091356.53026-3-shikemeng@huaweicloud.com Fixes: 451c6326105b ("mm, swap: clean up swapon process and locking") Signed-off-by: Kemeng Shi Signed-off-by: Andrew Morton Reviewed-by: Luiz Capitulino Acked-by: Kairui Song Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: Nhat Pham Cc: Youngjun Park --- mm/swapfile.c | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 8df8b2c2e5b405..40ceab21ed5b8d 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -3763,11 +3763,6 @@ SYSCALL_DEFINE2(swapon, const char __user *, specialfile, int, swap_flags) maxpages = si->max; - /* Set up the swap cluster info */ - error = setup_swap_clusters_info(si, swap_header, maxpages); - if (error) - goto bad_swap_unlock_inode; - if (si->bdev && bdev_stable_writes(si->bdev)) si->flags |= SWP_STABLE_WRITES; @@ -3781,6 +3776,14 @@ SYSCALL_DEFINE2(swapon, const char __user *, specialfile, int, swap_flags) inced_nr_rotate_swap = true; } + /* + * Set up the swap cluster info after SWP_ flags handling as + * setup_swap_clusters_info() checks SWP_SOLIDSTATE. + */ + error = setup_swap_clusters_info(si, swap_header, maxpages); + if (error) + goto bad_swap_unlock_inode; + if ((swap_flags & SWAP_FLAG_DISCARD) && si->bdev && bdev_max_discard_sectors(si->bdev)) { /* From b59ae10874c1e379075101677c6b90a4166c0b4c Mon Sep 17 00:00:00 2001 From: Kemeng Shi Date: Mon, 7 Sep 2026 17:13:55 +0800 Subject: [PATCH 0563/1012] mm, swap: return early from swap_extend_table_try_free() on first non-zero entry Return immediately when the first non-zero swap count is found as any non-zero swap count prevents freeing extend_table and further iteration is pointless. Link: https://lore.kernel.org/20260907091356.53026-4-shikemeng@huaweicloud.com Signed-off-by: Kemeng Shi Signed-off-by: Andrew Morton Reviewed-by: Youngjun Park Acked-by: Kairui Song Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: Luiz Capitulino Cc: Nhat Pham --- mm/swapfile.c | 9 +++------ 1 file changed, 3 insertions(+), 6 deletions(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 40ceab21ed5b8d..7b35ca90776a8a 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1523,20 +1523,17 @@ int swap_retry_table_alloc(swp_entry_t entry, gfp_t gfp) static void swap_extend_table_try_free(struct swap_cluster_info *ci) { unsigned long i; - bool can_free = true; if (!ci->extend_table) return; for (i = 0; i < SWAPFILE_CLUSTER; i++) { if (ci->extend_table[i]) - can_free = false; + return; } - if (can_free) { - kfree(ci->extend_table); - ci->extend_table = NULL; - } + kfree(ci->extend_table); + ci->extend_table = NULL; } /* Decrease the swap count of one slot, without freeing it */ From b87ab931a1314ed5a87a66d9df51de9da4d739af Mon Sep 17 00:00:00 2001 From: Kemeng Shi Date: Mon, 7 Sep 2026 17:13:56 +0800 Subject: [PATCH 0564/1012] mm, swap: remove unneeded swap_extend_table_try_free() in swap_dup_entries_cluster() Since commit 0475fde0f68de ("mm, swap: avoid leaving unused extend table after alloc race"), extend table is always allocated when any swap count reach MAX - 1 and is always freed when swap count decrease to MAX - 1 or MAX - 2 with cluster lock held. So the extend table will always be freed properly when decrease swap count in __swap_cluster_put_entry(). So swap_extend_table_try_free() outside of __swap_cluster_put_entry() is unneeded and can be removed. Link: https://lore.kernel.org/20260907091356.53026-5-shikemeng@huaweicloud.com Signed-off-by: Kemeng Shi Signed-off-by: Andrew Morton Reviewed-by: Youngjun Park Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: Kairui Song Cc: Luiz Capitulino Cc: Nhat Pham --- mm/swapfile.c | 1 - 1 file changed, 1 deletion(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 7b35ca90776a8a..2cd0d0ba966c38 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1725,7 +1725,6 @@ static int swap_dup_entries_cluster(struct swap_info_struct *si, failed: while (ci_off-- > ci_start) __swap_cluster_put_entry(ci, ci_off); - swap_extend_table_try_free(ci); swap_cluster_unlock(ci); return err; } From 44a9b66103897da15cc4f8e7275a59ca80db2d67 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Mon, 7 Sep 2026 09:36:54 +0300 Subject: [PATCH 0565/1012] docs/core-api: memory-allocation: clarify when to use kzalloc_obj and kzalloc Make it abundantly clear that in the most cases objects should be allocated with kzalloc_obj() and buffers should be allocated with kzalloc(). Link: https://lore.kernel.org/20260907063654.2248617-1-rppt@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Acked-by: Vlastimil Babka (SUSE) Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: David Hildenbrand (Arm) Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Michal Hocko Cc: Randy Dunlap Cc: SeongJae Park Cc: Suren Baghdasaryan --- Documentation/core-api/memory-allocation.rst | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/Documentation/core-api/memory-allocation.rst b/Documentation/core-api/memory-allocation.rst index 823f7fa57429b7..648222b6c85763 100644 --- a/Documentation/core-api/memory-allocation.rst +++ b/Documentation/core-api/memory-allocation.rst @@ -23,12 +23,14 @@ answer, although very likely you should use kzalloc_obj(); -or +if you need memory for an object and :: kzalloc(, GFP_KERNEL); +if you need memory for a buffer. + Of course there are cases when other allocation APIs and different GFP flags must be used. From 4f8d8b18f38b1df87016459dc21d4aa6ebe6f100 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Mon, 7 Sep 2026 10:54:44 +0800 Subject: [PATCH 0566/1012] mm/page_counter: avoid integer overflow in effective_protection() Patch series "mm/mglru: fix ineffective memory protection for non-kswapd reclaim", v4. For MGLRU, memory.min/low is not honored during non-kswapd global reclaim (global direct reclaim and root-level memory.reclaim), because these paths shrink memcgs using stale protection (emin/elow). Patch 2 is the actual fix. Patch 1 is a prerequisite: an integer overflow in effective_protection(), spotted by the sashiko review tool, which patch 2's new caller would also be exposed to. This patch (of 2): effective_protection() scales a parent's protection by a ratio of page counts, e.g. for recursive protection: (parent_effective - siblings_protected) * (usage - protected) / (parent_usage - siblings_protected) The multiply is done at unsigned long width before dividing. On systems with >= 16TB RAM the product can exceed 2^64 and wrap, giving a bogus protection value and silently breaking memory.min/low enforcement. Use mul_u64_u64_div_u64() to multiply in a 128-bit intermediate. Because usage and parent_usage are not read atomically (a child is charged before its parent), usage - protected can briefly exceed the divisor, making the quotient overflow 64 bits and trap (#DE on x86). Cap it so the ratio stays <= 1. Reported by the sashiko review tool [1]. Link: https://lore.kernel.org/20260907025445.1836238-1-ridong.chen@linux.dev Link: https://lore.kernel.org/20260907025445.1836238-2-ridong.chen@linux.dev Link: https://sashiko.dev/#/patchset/20260826133054.88529-1-ridong.chen@linux.dev?part=1 [1] Fixes: bc50bcc6e00b ("mm: memcontrol: clean up and document effective low/min calculations") Fixes: 8a931f801340 ("mm: memcontrol: recursive memory.low protection") Reviewed-by: Barry Song Reviewed-by: Johannes Weiner Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Assisted-by: Claude:claude-opus-4-8 Cc: Axel Rasmussen Cc: Chris Down Cc: David Hildenbrand Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt Cc: Tejun Heo Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: --- mm/page_counter.c | 21 +++++++++++++++------ 1 file changed, 15 insertions(+), 6 deletions(-) diff --git a/mm/page_counter.c b/mm/page_counter.c index 450543f4b318b6..98322803941a70 100644 --- a/mm/page_counter.c +++ b/mm/page_counter.c @@ -8,6 +8,7 @@ #include #include #include +#include #include #include #include @@ -376,7 +377,8 @@ static unsigned long effective_protection(unsigned long usage, * otherwise get a smaller chunk than what they claimed. */ if (siblings_protected > parent_effective) - return protected * parent_effective / siblings_protected; + return mul_u64_u64_div_u64(protected, parent_effective, + siblings_protected); /* * Ok, utilized protection of all children is within what the @@ -417,13 +419,20 @@ static unsigned long effective_protection(unsigned long usage, if (parent_effective > siblings_protected && parent_usage > siblings_protected && usage > protected) { - unsigned long unclaimed; + unsigned long parent_unclaimed, parent_unprotected, unprotected; - unclaimed = parent_effective - siblings_protected; - unclaimed *= usage - protected; - unclaimed /= parent_usage - siblings_protected; + parent_unclaimed = parent_effective - siblings_protected; + parent_unprotected = parent_usage - siblings_protected; - ep += unclaimed; + /* + * The usages aren't read atomically, so a child can transiently + * appear to use more than its parent, making the ratio exceed 1 + * and the quotient overflow 64 bits (#DE on x86). Cap it. + */ + unprotected = min(usage - protected, parent_unprotected); + + ep += mul_u64_u64_div_u64(parent_unclaimed, unprotected, + parent_unprotected); } return ep; From 1d7a33cf5b7bc24b4f05c9f297c2f62d77c73e39 Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Mon, 7 Sep 2026 10:54:45 +0800 Subject: [PATCH 0567/1012] mm/mglru: fix ineffective memory protection for non-kswapd reclaim For MGLRU, memory.min/low is not honored during global proactive reclaim (writing to the root memory.reclaim) and global direct reclaim, because these paths shrink memcgs using stale protection (emin/elow). It can be reproduced as follows: # echo 7 > /sys/kernel/mm/lru_gen/enabled # cd /sys/fs/cgroup # mkdir -p a/b # echo 100M > a/memory.min # echo +memory > a/cgroup.subtree_control # echo 100M > a/b/memory.min # echo $$ > a/b/cgroup.procs # dd if=/dev/zero of=/tmp/testfile bs=1M count=200 # cat a/b/memory.current 222650368 # echo 500M > memory.reclaim -bash: echo: write error: Resource temporarily unavailable # cat a/b/memory.current 6070272 memory.min is 100M, yet reclaim drops a/b down to 6M, breaking the protection. The traditional LRU path is not affected because shrink_node() calls mem_cgroup_calculate_protection() for each memcg it visits during a top-down tree walk. Commit 30d77b7eef01 ("mm/mglru: fix ineffective protection calculation") moved the protection computation into lru_gen_age_node(), which only runs for kswapd. Non-kswapd global reclaim reaches shrink_one() through lru_gen_shrink_node() -> shrink_many() without any protection computation, so emin/elow are whatever a previous kswapd run left behind - or zero if kswapd never ran on this node. Relying on a prior kswapd pass is not correct either: a memcg's emin/elow are derived from its ancestors' memory.min/low settings and from children_min_usage, both of which change over time, so emin/elow go stale even after kswapd has run and must be recomputed at the point of reclaim. Introduce mem_cgroup_calculate_protection_path() which computes emin/elow along the root-to-target path only, by iterating through the cgroup ancestors array top-down. This avoids the full tree traversal that would be needed with mem_cgroup_calculate_protection(), limiting the cost to O(depth) per memcg - typically 3-5 levels. Call it from shrink_one() for the non-kswapd path so that each memcg about to be shrunk has correct protection values. Link: https://lore.kernel.org/20260907025445.1836238-3-ridong.chen@linux.dev Fixes: e4dde56cd208 ("mm: multi-gen LRU: per-node lru_gen_folio lists") Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Reviewed-by: Barry Song Reviewed-by: Johannes Weiner Assisted-by: Claude:claude-opus-4-8 Cc: Axel Rasmussen Cc: Chris Down Cc: David Hildenbrand Cc: Kairui Song Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Muchun Song Cc: Roman Gushchin Cc: Shakeel Butt Cc: Tejun Heo Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao Cc: --- include/linux/memcontrol.h | 10 +++++++++ mm/memcontrol.c | 45 ++++++++++++++++++++++++++++++++++++++ mm/vmscan.c | 8 ++++++- 3 files changed, 62 insertions(+), 1 deletion(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 4f720791a31c95..46bf724cae7af9 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -1921,6 +1921,16 @@ static inline bool memcg_is_dying(struct mem_cgroup *memcg) } #endif /* CONFIG_MEMCG */ +#if defined(CONFIG_MEMCG) && defined(CONFIG_LRU_GEN) +void mem_cgroup_calculate_protection_path(struct mem_cgroup *root, + struct mem_cgroup *memcg); +#else +static inline void mem_cgroup_calculate_protection_path(struct mem_cgroup *root, + struct mem_cgroup *memcg) +{ +} +#endif + #if defined(CONFIG_MEMCG) && defined(CONFIG_ZSWAP) bool obj_cgroup_may_zswap(struct obj_cgroup *objcg); void obj_cgroup_charge_zswap(struct obj_cgroup *objcg, size_t size); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index cf53d4ac7ecc18..791e536efaebe8 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5267,6 +5267,51 @@ void mem_cgroup_calculate_protection(struct mem_cgroup *root, page_counter_calculate_protection(&root->memory, &memcg->memory, recursive_protection); } +#ifdef CONFIG_LRU_GEN +/** + * mem_cgroup_calculate_protection_path - compute protection along a path + * @root: the top ancestor of the sub-tree being checked (NULL for root_mem_cgroup) + * @memcg: the target memory cgroup + * + * Walk the ancestor path from @root down to @memcg and compute the effective + * protection at each level. This is safe for isolated queries because it + * ensures parents are computed before children. + */ +void mem_cgroup_calculate_protection_path(struct mem_cgroup *root, + struct mem_cgroup *memcg) +{ + bool recursive_protection = + cgrp_dfl_root.flags & CGRP_ROOT_MEMORY_RECURSIVE_PROT; + struct cgroup *cg; + int root_level, i; + + if (mem_cgroup_disabled()) + return; + + if (!root) + root = root_mem_cgroup; + + if (memcg == root) + return; + + root_level = root->css.cgroup->level; + cg = memcg->css.cgroup; + + rcu_read_lock(); + for (i = root_level + 1; i <= cg->level; i++) { + struct mem_cgroup *cur; + + cur = mem_cgroup_from_css(cgroup_css(cg->ancestors[i], + &memory_cgrp_subsys)); + if (cur) + page_counter_calculate_protection(&root->memory, + &cur->memory, + recursive_protection); + } + rcu_read_unlock(); +} +#endif /* CONFIG_LRU_GEN */ + static int charge_memcg(struct folio *folio, struct mem_cgroup *memcg, gfp_t gfp) { diff --git a/mm/vmscan.c b/mm/vmscan.c index 8e9c73dcd19bb3..80041e2b8049c7 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -5251,7 +5251,13 @@ static int shrink_one(struct lruvec *lruvec, struct scan_control *sc) struct mem_cgroup *memcg = lruvec_memcg(lruvec); struct pglist_data *pgdat = lruvec_pgdat(lruvec); - /* lru_gen_age_node() called mem_cgroup_calculate_protection() */ + /* + * For kswapd, mem_cgroup_calculate_protection() has already + * been called during the top-down cgroup traversal. + */ + if (!current_is_kswapd()) + mem_cgroup_calculate_protection_path(NULL, memcg); + if (mem_cgroup_below_min(NULL, memcg)) return MEMCG_LRU_YOUNG; From 606161fe9df33473f63cf51b3a27c5e2f6c76649 Mon Sep 17 00:00:00 2001 From: Tianyi Chen Date: Tue, 8 Sep 2026 17:55:15 +0800 Subject: [PATCH 0568/1012] tools/testing/vma: cover hole filling through __mmap_region() The mmap tests extend existing mappings one neighbor at a time, while merge tests construct merge state directly. Neither exercises filling a hole between compatible mappings through the mmap setup and completion path. Fill a gap through __mmap_region() and require both neighbors to merge into one VMA. Repeat with only the new mapping's execute permission set and require three separate VMAs. Check boundaries, permissions, page offsets, map_count and cleanup. Link: https://lore.kernel.org/178886112560.138404.17741948638665342936.vma-v2@tychen.cc Signed-off-by: Tianyi Chen Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Assisted-by: Codex:GPT-6 Cc: Jann Horn Cc: Liam R. Howlett Cc: Pedro Falcato Cc: Vlastimil Babka --- tools/testing/vma/tests/mmap.c | 68 ++++++++++++++++++++++++++++++++++ 1 file changed, 68 insertions(+) diff --git a/tools/testing/vma/tests/mmap.c b/tools/testing/vma/tests/mmap.c index fa73faff226263..53e4abe6a63e6c 100644 --- a/tools/testing/vma/tests/mmap.c +++ b/tools/testing/vma/tests/mmap.c @@ -45,6 +45,72 @@ static bool test_mmap_region_basic(void) return true; } +static bool mmap_region_fill_hole(bool merge) +{ + const vma_flags_t vma_flags = mk_vma_flags(VMA_READ_BIT, VMA_WRITE_BIT, + VMA_MAYREAD_BIT, VMA_MAYWRITE_BIT, VMA_MAYEXEC_BIT); + vma_flags_t middle_flags = vma_flags; + struct mm_struct mm = {}; + struct vm_area_struct *vma; + unsigned long addr; + int count = 0; + VMA_ITERATOR(vmi, &mm, 0); + + current->mm = &mm; + if (!merge) + vma_flags_set(&middle_flags, VMA_EXEC_BIT); + + /* Map at 0x300000, length 0x3000. */ + addr = __mmap_region(NULL, 0x300000, 0x3000, vma_flags, 0x300, NULL); + ASSERT_EQ(addr, 0x300000); + + /* Map at 0x306000, length 0x3000, leaving a hole. */ + addr = __mmap_region(NULL, 0x306000, 0x3000, vma_flags, 0x306, NULL); + ASSERT_EQ(addr, 0x306000); + ASSERT_EQ(mm.map_count, 2); + + /* Map at 0x303000, length 0x3000, filling the hole. */ + addr = __mmap_region(NULL, 0x303000, 0x3000, middle_flags, 0x303, NULL); + ASSERT_EQ(addr, 0x303000); + ASSERT_EQ(mm.map_count, merge ? 1 : 3); + + vma_iter_set(&vmi, 0); + for_each_vma(vmi, vma) { + const unsigned long start = 0x300000 + count * 0x3000; + const unsigned long end = merge ? 0x309000 : start + 0x3000; + /* Only the middle VMA in the non-merge case has VMA_EXEC. */ + const bool is_middle_vma = count == 1; + const bool expect_exec_vma = is_middle_vma && !merge; + + ASSERT_EQ(vma->vm_start, start); + ASSERT_EQ(vma->vm_end, end); + ASSERT_EQ(vma_start_pgoff(vma), start >> PAGE_SHIFT); + ASSERT_EQ(vma_start_anon_pgoff(vma), start >> PAGE_SHIFT); + + ASSERT_TRUE(vma_test_all(vma, VMA_READ_BIT, VMA_WRITE_BIT, + VMA_MAYREAD_BIT, VMA_MAYWRITE_BIT, + VMA_MAYEXEC_BIT)); + ASSERT_EQ(vma_test(vma, VMA_EXEC_BIT), expect_exec_vma); + + count++; + } + + ASSERT_EQ(count, mm.map_count); + + ASSERT_EQ(cleanup_mm(&mm, &vmi), count); + return true; +} + +static bool test_mmap_region_fill_hole_merge(void) +{ + return mmap_region_fill_hole(true); +} + +static bool test_mmap_region_fill_hole_flags_mismatch(void) +{ + return mmap_region_fill_hole(false); +} + static bool test_pure_anon_dev_zero(void) { const vma_flags_t vma_flags = mk_vma_flags(VMA_READ_BIT, VMA_WRITE_BIT, @@ -84,5 +150,7 @@ static bool test_pure_anon_dev_zero(void) static void run_mmap_tests(int *num_tests, int *num_fail) { TEST(mmap_region_basic); + TEST(mmap_region_fill_hole_merge); + TEST(mmap_region_fill_hole_flags_mismatch); TEST(pure_anon_dev_zero); } From 23840442170eaf625f4d73e4ce383b38671672b2 Mon Sep 17 00:00:00 2001 From: Longlong Xia Date: Sun, 6 Sep 2026 21:59:38 +0800 Subject: [PATCH 0569/1012] mm/zswap: enable zswap_ever_enabled in zswap_pool_create() If zswap is enabled by default at boot and pool creation fails, then a pool is later created by updating the compressor, data written to zswap is corrupted on swapin. Enable the static key in zswap_pool_create(), covering boot-time and runtime pool creation with a single site. Verified with fault injection on a stock kernel (compressor builtin, CONFIG_ZSWAP_DEFAULT_ON=n): 1. Boot with zswap.enabled=1; pool creation fails, init completes pool-less (static key off). 2. Echo an available compressor name to zswap.compressor. 3. Enable zswap. 4. madvise(MADV_PAGEOUT) a pattern-verified 512 MiB region, then fault it back in and verify. Step 4 reads back 131072/131072 zeroed pages (zswpin=0, zswpout=131072) without this patch; all pages intact (zswpin=131072) with it. Link: https://lore.kernel.org/20260906135938.3568108-1-xialonglong2025@163.com Fixes: 2d4d2b1cfb85 ("mm: zswap: add zswap_never_enabled()") Signed-off-by: Longlong Xia Signed-off-by: Andrew Morton Suggested-by: Yosry Ahmed Acked-by: Yosry Ahmed Assisted-by: Zcode:GLM-5.3 Cc: Chengming Zhou Cc: Johannes Weiner Cc: Nhat Pham Cc: --- mm/zswap.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/zswap.c b/mm/zswap.c index d0b6c229b6169d..cfaedc85dfedb3 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -329,6 +329,8 @@ static struct zswap_pool *zswap_pool_create(char *compressor) zswap_pool_debug("created", pool); + static_branch_enable(&zswap_ever_enabled); + return pool; ref_fail: @@ -1818,7 +1820,6 @@ static int zswap_setup(void) pr_info("loaded using pool %s\n", pool->tfm_name); list_add_rcu(&pool->list, &zswap_pools); zswap_has_pool = true; - static_branch_enable(&zswap_ever_enabled); } else { pr_err("pool creation failed\n"); zswap_enabled = false; From 7cfa5101094c090ad59f82cc7253dada10f90bb9 Mon Sep 17 00:00:00 2001 From: Jianyue Wu Date: Sun, 6 Sep 2026 15:47:17 +0800 Subject: [PATCH 0570/1012] mm/zswap: release retired pools via queue_rcu_work() instead of synchronize_rcu() Patch series "mm/zswap: shrink zswap_entry via a pool id", v6. Every stored page has a struct zswap_entry, so its size is pure per-page overhead. On 64-bit it is currently 56 bytes, of which 8 bytes are a pointer to the owning zswap_pool. Only a handful of pools are ever live: a new pool is created only when the compressor is (re)set, and pools are reused across compressor switches. A list cannot look a pool up by id. An allocating xarray can, which lets each zswap_entry store a u8 instead of a pointer. This series: 1. Releases retired pools with queue_rcu_work() instead of a worker calling synchronize_rcu(), so the release worker no longer blocks on an RCU grace period. 2. Replaces the zswap_pools list with an allocating xarray (XA_FLAGS_ALLOC1 | XA_FLAGS_LOCK_BH) and a separate RCU-protected current-pool pointer, giving each pool a stable small id. Ids start at 1. The reserved id 0 is never allocated, so looking it up resolves to NULL. The table grows as needed up to 255 live pools (u8 pool_idx), not a fixed slot array. 3. Stores that u8 pool id in each zswap_entry instead of the pool pointer. The u8 fits in padding after the bool referenced field. On 64-bit that shrinks the entry from 56 to 48 bytes, which fits 73 to 85 objects in a 4K slab (~2MiB of metadata saved per 1GiB of data held in zswap). The 255-id cap counts every pool still in the xarray. Switching compressor kills the old pool, but that pool stays in the table until its last entry drops the pool's ref, so a draining pool still occupies an id. The id is reused only after xa_erase. Switching back to a compressor whose pool is still in the table resurrects it instead of allocating a new id. If every id is occupied, creating a pool for another compressor fails and the switch is rejected. Pool table locking: xa_for_each() walks and the current-pool pointer use an explicit rcu_read_lock(), because xa_for_each()'s own RCU does not span the loop body. xa_load() takes RCU around the lookup itself, so zswap_entry_pool() needs no extra rcu_read_lock(). The returned pool stays valid because a live entry pins it via percpu_ref, so the id cannot be reused under it. xa_lock is taken only in xa_alloc_bh() and xa_erase_bh(). Benchmark (x86_64, compressor=lzo, MADV_PAGEOUT store + fault-in load): - e2e store+load median latency: no measurable regression vs baseline at matched stored_delta Each store, free, and decompress looks up the pool with xa_load() instead of following a pointer. With only a handful of live pools the xarray walk is short. This patch (of 3): When a pool's last reference is dropped, __zswap_pool_empty() removes it from the pool list and schedules __zswap_pool_release(), which calls synchronize_rcu() to wait for readers before tearing the pool down. synchronize_rcu() is a synchronous, potentially long wait. Replace it with queue_rcu_work(): __zswap_pool_empty() hands the pool to queue_rcu_work(), which waits for a grace period asynchronously and then runs __zswap_pool_release() from a worker for the sleepable teardown (__zswap_pool_empty() can run in atomic context and must not block). The grace-period guarantee is unchanged; the retirement path just no longer blocks on it. Link: https://lore.kernel.org/20260906-shrink_zswap_entry_v6-v6-0-ac4cf61565fb@gmail.com Link: https://lore.kernel.org/20260906-shrink_zswap_entry_v6-v6-1-ac4cf61565fb@gmail.com Signed-off-by: Jianyue Wu Signed-off-by: Andrew Morton Suggested-by: Yosry Ahmed Acked-by: Yosry Ahmed Cc: Chengming Zhou Cc: Chris Li Cc: Johannes Weiner Cc: Nhat Pham --- mm/zswap.c | 12 +++++------- 1 file changed, 5 insertions(+), 7 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index cfaedc85dfedb3..526327266e7621 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -155,7 +155,7 @@ struct zswap_pool { struct crypto_acomp_ctx __percpu *acomp_ctx; struct percpu_ref ref; struct list_head list; - struct work_struct release_work; + struct rcu_work release_rwork; struct hlist_node node; char tfm_name[CRYPTO_MAX_ALG_NAME]; }; @@ -386,10 +386,8 @@ static void zswap_pool_destroy(struct zswap_pool *pool) static void __zswap_pool_release(struct work_struct *work) { - struct zswap_pool *pool = container_of(work, typeof(*pool), - release_work); - - synchronize_rcu(); + struct zswap_pool *pool = container_of(to_rcu_work(work), + typeof(*pool), release_rwork); /* nobody should have been able to get a ref... */ WARN_ON(!percpu_ref_is_zero(&pool->ref)); @@ -413,8 +411,8 @@ static void __zswap_pool_empty(struct percpu_ref *ref) list_del_rcu(&pool->list); - INIT_WORK(&pool->release_work, __zswap_pool_release); - schedule_work(&pool->release_work); + INIT_RCU_WORK(&pool->release_rwork, __zswap_pool_release); + queue_rcu_work(system_percpu_wq, &pool->release_rwork); spin_unlock_bh(&zswap_pools_lock); } From da625bae8442acd27af1e15c414d7d295ce7bb22 Mon Sep 17 00:00:00 2001 From: Jianyue Wu Date: Sun, 6 Sep 2026 15:47:18 +0800 Subject: [PATCH 0571/1012] mm/zswap: replace the zswap_pools list with an allocating xarray Originally zswap kept its pools on an RCU list whose head also served as the current pool. Convert the pool table to an allocating xarray keyed by a small integer id, and track the current pool with a separate RCU-protected pointer. The xarray gives each pool a stable id for a later zswap_entry shrink. XA_FLAGS_ALLOC1 starts ids at 1, so id 0 remains reserved. The id range is bounded by ZSWAP_MAX_POOL_ID because the later entry field is a u8. Keep compressor switching close to the previous flow: look up an existing pool with xa_for_each(), resurrect it if reused, or create a new one. zswap_pool_create() allocates the pool's id and publishes it into the xarray as its final step, so the create call either fully publishes or fully unwinds on failure. Publishing makes the pool live, so a caller that later fails (e.g. param_set_charp()) must still kill the pool to erase it from the xarray. Compressor switches update zswap_current_pool with rcu_assign_pointer(), serialized by the module parameter lock, so no xa_lock is needed for that update. The pool walk above is lockless under RCU. xa_lock is taken only to allocate (xa_alloc_bh()) and erase (xa_erase_bh()) xarray entries. Link: https://lore.kernel.org/20260906-shrink_zswap_entry_v6-v6-2-ac4cf61565fb@gmail.com Signed-off-by: Jianyue Wu Signed-off-by: Andrew Morton Suggested-by: Nhat Pham Suggested-by: Yosry Ahmed Suggested-by: Johannes Weiner Acked-by: Yosry Ahmed Cc: Chengming Zhou Cc: Chris Li --- mm/zswap.c | 122 +++++++++++++++++++++++++++++------------------------ 1 file changed, 66 insertions(+), 56 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index 526327266e7621..8b6b1dce79480e 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -34,6 +34,7 @@ #include #include #include +#include #include #include @@ -154,12 +155,23 @@ struct zswap_pool { struct zs_pool *zs_pool; struct crypto_acomp_ctx __percpu *acomp_ctx; struct percpu_ref ref; - struct list_head list; struct rcu_work release_rwork; struct hlist_node node; + u8 idx; char tfm_name[CRYPTO_MAX_ALG_NAME]; }; +/* + * Live pools keyed by id (1..ZSWAP_MAX_POOL_ID). XA_FLAGS_ALLOC1 keeps id 0 + * reserved so it is never handed to a live pool. XA_FLAGS_LOCK_BH makes the + * xa_lock softirq-safe: it is taken from __zswap_pool_empty(), which runs from + * a percpu_ref release callback in softirq context. + */ +#define ZSWAP_FIRST_POOL_ID 1 +#define ZSWAP_MAX_POOL_ID U8_MAX +static DEFINE_XARRAY_FLAGS(zswap_pools, XA_FLAGS_ALLOC1 | XA_FLAGS_LOCK_BH); +static struct zswap_pool __rcu *zswap_current_pool; + /* Global LRU lists shared by all zswap pools. */ static struct list_lru zswap_list_lru; @@ -200,10 +212,6 @@ struct zswap_entry { static struct xarray *zswap_trees[MAX_SWAPFILES]; static unsigned int nr_zswap_trees[MAX_SWAPFILES]; -/* RCU-protected iteration */ -static LIST_HEAD(zswap_pools); -/* protects zswap_pools list modification */ -static DEFINE_SPINLOCK(zswap_pools_lock); /* pool counter to provide unique names to zsmalloc */ static atomic_t zswap_pools_count = ATOMIC_INIT(0); @@ -280,6 +288,7 @@ static struct zswap_pool *zswap_pool_create(char *compressor) struct zswap_pool *pool; char name[38]; /* 'zswap' + 32 char (max) num + \0 */ int ret, cpu; + u32 id; if (!zswap_has_pool && !strcmp(compressor, ZSWAP_PARAM_UNSET)) return NULL; @@ -325,7 +334,22 @@ static struct zswap_pool *zswap_pool_create(char *compressor) PERCPU_REF_ALLOW_REINIT, GFP_KERNEL); if (ret) goto ref_fail; - INIT_LIST_HEAD(&pool->list); + + /* + * Publish only after the pool is fully built, so lockless walkers + * never see a half-initialized pool. The _bh variant pairs with the + * softirq-context xa_lock taken in __zswap_pool_empty(). + */ + ret = xa_alloc_bh(&zswap_pools, &id, pool, + XA_LIMIT(ZSWAP_FIRST_POOL_ID, ZSWAP_MAX_POOL_ID), + GFP_KERNEL); + if (ret) { + if (ret == -EBUSY) + pr_err("cannot allocate pool id (max %d live pools)\n", + ZSWAP_MAX_POOL_ID - ZSWAP_FIRST_POOL_ID + 1); + goto xa_fail; + } + pool->idx = id; zswap_pool_debug("created", pool); @@ -333,6 +357,8 @@ static struct zswap_pool *zswap_pool_create(char *compressor) return pool; +xa_fail: + percpu_ref_exit(&pool->ref); ref_fail: cpuhp_state_remove_instance(CPUHP_MM_ZSWP_POOL_PREPARE, &pool->node); @@ -393,28 +419,22 @@ static void __zswap_pool_release(struct work_struct *work) WARN_ON(!percpu_ref_is_zero(&pool->ref)); percpu_ref_exit(&pool->ref); - /* pool is now off zswap_pools list and has no references. */ + /* The pool is no longer in zswap_pools and has no references. */ zswap_pool_destroy(pool); } -static struct zswap_pool *zswap_pool_current(void); - static void __zswap_pool_empty(struct percpu_ref *ref) { struct zswap_pool *pool; pool = container_of(ref, typeof(*pool), ref); - spin_lock_bh(&zswap_pools_lock); - - WARN_ON(pool == zswap_pool_current()); + WARN_ON(pool == rcu_access_pointer(zswap_current_pool)); - list_del_rcu(&pool->list); + xa_erase_bh(&zswap_pools, pool->idx); INIT_RCU_WORK(&pool->release_rwork, __zswap_pool_release); queue_rcu_work(system_percpu_wq, &pool->release_rwork); - - spin_unlock_bh(&zswap_pools_lock); } static int __must_check zswap_pool_tryget(struct zswap_pool *pool) @@ -440,20 +460,13 @@ static struct zswap_pool *__zswap_pool_current(void) { struct zswap_pool *pool; - pool = list_first_or_null_rcu(&zswap_pools, typeof(*pool), list); + pool = rcu_dereference(zswap_current_pool); WARN_ONCE(!pool && zswap_has_pool, "%s: no page storage pool!\n", __func__); return pool; } -static struct zswap_pool *zswap_pool_current(void) -{ - assert_spin_locked(&zswap_pools_lock); - - return __zswap_pool_current(); -} - static struct zswap_pool *zswap_pool_current_get(void) { struct zswap_pool *pool; @@ -469,23 +482,28 @@ static struct zswap_pool *zswap_pool_current_get(void) return pool; } -/* type and compressor must be null-terminated */ +/* compressor must be null-terminated */ static struct zswap_pool *zswap_pool_find_get(char *compressor) { struct zswap_pool *pool; + unsigned long id; - assert_spin_locked(&zswap_pools_lock); - - list_for_each_entry_rcu(pool, &zswap_pools, list) { + /* + * __zswap_pool_empty() can erase from zswap_pools in softirq while we + * walk. rcu_read_lock() keeps the walk consistent and each pool alive + * across tryget(). xa_for_each()'s own RCU does not span the loop body. + */ + rcu_read_lock(); + xa_for_each(&zswap_pools, id, pool) { if (strcmp(pool->tfm_name, compressor)) continue; /* if we can't get it, it's about to be destroyed */ - if (!zswap_pool_tryget(pool)) - continue; - return pool; + if (zswap_pool_tryget(pool)) + break; } + rcu_read_unlock(); - return NULL; + return pool; } static unsigned long zswap_max_pages(void) @@ -502,9 +520,14 @@ unsigned long zswap_total_pages(void) { struct zswap_pool *pool; unsigned long total = 0; + unsigned long id; + /* + * rcu_read_lock() keeps each pool alive across zs_get_total_pages(). + * xa_for_each()'s own RCU does not span the loop body. + */ rcu_read_lock(); - list_for_each_entry_rcu(pool, &zswap_pools, list) + xa_for_each(&zswap_pools, id, pool) total += zs_get_total_pages(pool->zs_pool); rcu_read_unlock(); @@ -561,20 +584,13 @@ static int zswap_compressor_param_set(const char *val, const struct kernel_param return -ENOENT; } - spin_lock_bh(&zswap_pools_lock); - pool = zswap_pool_find_get(s); - if (pool) { + if (!pool) { + pool = zswap_pool_create(s); + } else { zswap_pool_debug("using existing", pool); - WARN_ON(pool == zswap_pool_current()); - list_del_rcu(&pool->list); - } - - spin_unlock_bh(&zswap_pools_lock); + WARN_ON(pool == rcu_access_pointer(zswap_current_pool)); - if (!pool) - pool = zswap_pool_create(s); - else { /* * Restore the initial ref dropped by percpu_ref_kill() * when the pool was decommissioned and switch it again @@ -591,24 +607,18 @@ static int zswap_compressor_param_set(const char *val, const struct kernel_param else ret = -EINVAL; - spin_lock_bh(&zswap_pools_lock); - + /* + * Compressor switches are serialized by the kernel param lock, so this + * is the only writer of zswap_current_pool: no xa_lock needed. + */ if (!ret) { - put_pool = zswap_pool_current(); - list_add_rcu(&pool->list, &zswap_pools); + put_pool = rcu_access_pointer(zswap_current_pool); + rcu_assign_pointer(zswap_current_pool, pool); zswap_has_pool = true; } else if (pool) { - /* - * Add the possibly pre-existing pool to the end of the pools - * list; if it's new (and empty) then it'll be removed and - * destroyed by the put after we drop the lock - */ - list_add_tail_rcu(&pool->list, &zswap_pools); put_pool = pool; } - spin_unlock_bh(&zswap_pools_lock); - /* * Drop the ref from either the old current pool, * or the new pool we failed to add @@ -1816,7 +1826,7 @@ static int zswap_setup(void) pool = __zswap_pool_create_fallback(); if (pool) { pr_info("loaded using pool %s\n", pool->tfm_name); - list_add_rcu(&pool->list, &zswap_pools); + rcu_assign_pointer(zswap_current_pool, pool); zswap_has_pool = true; } else { pr_err("pool creation failed\n"); From d5b277778b4d1c37900ff9df4aa82093a92b5195 Mon Sep 17 00:00:00 2001 From: Jianyue Wu Date: Sun, 6 Sep 2026 15:47:19 +0800 Subject: [PATCH 0572/1012] mm/zswap: reference the pool by id to shrink struct zswap_entry struct zswap_entry is one allocation per stored page, so its size is pure overhead. It currently embeds an 8-byte pool pointer, even though the live pools now sit in an allocating xarray keyed by a small integer id that fits in a u8. Replace the per-entry pool pointer with that u8 id and resolve it through the xarray with xa_load(). xa_load() does its own RCU-protected lookup, so the caller needs no rcu_read_lock() section of its own. The resolved pool stays valid because a live entry pins it via percpu_ref (taken in zswap_store_page()), so its id cannot be reused. A live entry never uses the reserved id 0, so a zeroed id resolves to NULL and trips a WARN rather than aliasing a live pool. The u8 fits in the padding after the bool referenced field, shrinking the entry from 56 to 48 bytes on 64-bit. This raises objs_per_slab from 73 to 85 and saves about 2MiB of metadata per 1GiB of data held in zswap. Link: https://lore.kernel.org/20260906-shrink_zswap_entry_v6-v6-3-ac4cf61565fb@gmail.com Signed-off-by: Jianyue Wu Signed-off-by: Andrew Morton Suggested-by: Chris Li Acked-by: Yosry Ahmed Cc: Chengming Zhou Cc: Johannes Weiner Cc: Nhat Pham --- mm/zswap.c | 37 ++++++++++++++++++++++++++++++------- 1 file changed, 30 insertions(+), 7 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index 8b6b1dce79480e..ff7c6742af50f5 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -194,7 +194,7 @@ static struct shrinker *zswap_shrinker; * writeback logic. The entry is only reclaimed by the writeback * logic if referenced is unset. See comments in the shrinker * section for context. - * pool - the zswap_pool the entry's data is in + * pool_idx - id of the zswap_pool that the entry's data is in. * handle - zsmalloc allocation handle that stores the compressed page data * objcg - the obj_cgroup that the compressed memory is charged to * lru - handle to the pool's lru used to evict pages. @@ -203,12 +203,22 @@ struct zswap_entry { swp_entry_t swpentry; unsigned int length; bool referenced; - struct zswap_pool *pool; + u8 pool_idx; unsigned long handle; struct obj_cgroup *objcg; struct list_head lru; }; +/* + * No RCU section is needed around the returned pointer: a stored entry pins + * its pool via percpu_ref (taken in zswap_store_page()), so the id cannot be + * reused under us. Callers WARN and handle a NULL from a corrupt pool_idx. + */ +static struct zswap_pool *zswap_entry_pool(struct zswap_entry *entry) +{ + return xa_load(&zswap_pools, entry->pool_idx); +} + static struct xarray *zswap_trees[MAX_SWAPFILES]; static unsigned int nr_zswap_trees[MAX_SWAPFILES]; @@ -766,9 +776,13 @@ static void zswap_entry_cache_free(struct zswap_entry *entry) */ static void zswap_entry_free(struct zswap_entry *entry) { + struct zswap_pool *pool = zswap_entry_pool(entry); + zswap_lru_del(entry); - zs_free(entry->pool->zs_pool, entry->handle); - zswap_pool_put(entry->pool); + if (!WARN_ON_ONCE(!pool)) { + zs_free(pool->zs_pool, entry->handle); + zswap_pool_put(pool); + } if (entry->objcg) { obj_cgroup_uncharge_zswap(entry->objcg, entry->length); obj_cgroup_put(entry->objcg); @@ -924,12 +938,15 @@ static bool zswap_compress(struct folio *folio, long index, static bool zswap_decompress(struct zswap_entry *entry, struct folio *folio) { - struct zswap_pool *pool = entry->pool; + struct zswap_pool *pool = zswap_entry_pool(entry); struct scatterlist input[2]; /* zsmalloc returns an SG list 1-2 entries */ struct scatterlist output; struct crypto_acomp_ctx *acomp_ctx; int ret = 0, dlen; + if (WARN_ON_ONCE(!pool)) + return false; + acomp_ctx = raw_cpu_ptr(pool->acomp_ctx); mutex_lock(&acomp_ctx->mutex); zs_obj_read_sg_begin(pool->zs_pool, entry->handle, input, entry->length); @@ -965,7 +982,7 @@ static bool zswap_decompress(struct zswap_entry *entry, struct folio *folio) pr_alert_ratelimited("Decompression error from zswap (%d:%lu %s %u->%d)\n", swp_type(entry->swpentry), swp_offset(entry->swpentry), - entry->pool->tfm_name, + pool->tfm_name, entry->length, dlen); return false; } @@ -1423,6 +1440,13 @@ static bool zswap_store_page(struct folio *folio, long index, if (!zswap_compress(folio, index, entry, pool)) goto compress_failed; + /* + * Set pool_idx before the xa_store() below publishes the entry, or a + * concurrent reader could resolve a stale pool_idx left by slab reuse + * to an unrelated live pool. + */ + entry->pool_idx = pool->idx; + old = xa_store(swap_zswap_tree(page_swpentry), swp_offset(page_swpentry), entry, GFP_KERNEL); @@ -1468,7 +1492,6 @@ static bool zswap_store_page(struct folio *folio, long index, * The publishing order matters to prevent writeback from seeing * an incoherent entry. */ - entry->pool = pool; entry->swpentry = page_swpentry; entry->objcg = objcg; entry->referenced = true; From 2ce6ca24ee6995aa5098e8a858a1fcd633f1b025 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 10 Sep 2026 07:22:28 -0700 Subject: [PATCH 0573/1012] mm/damon/api: introduce DAMON_FILTER_TYPE_PGIDLE_SET Patch series "mm/damon: introduce pgidle_set probe filter type". Users can do effective access monitoring with DAMON, using set_pgidle probe prep operation and pgidle_unset allowing probe filter. In the setup, high probe_hits means the region was accessed frequently. Zero probe_hits means the region was not accessed. The filter, however, matches only the memory that can show the idleness. Regions of zero probe_hits may contain memory that was just unable to show the idleness. On virtual address space based monitoring, for example, non-present pages are reported as cold. It is truly cold, but if the purpose is to make some operations like reclaiming or compressing the data, this is not really useful. Introduce a new probe filter type, pgidle_set, that confirms the page is marked as idle. Using it instead of pgidle_unset, users can find cold pages that can effectively be processed. Patch 1 introduces the pgidle_set probe filter type to DAMON kernel API. Patches 2 and 3 add support for it on physical and virtual address space operation sets, respectively. Patch 4 adds support for it on the DAMON sysfs interface. Finally, patch 5 updates the documentation for the new filter type. This patch (of 5): DAMON_FILTER_TYPE_PGIDLE_UNSET matches only memory that was able to confirm if it is marked as idle. If the memory cannot be marked as idle, it simply doesn't match. Hence, finding memory that was marked as idle with only DAMON_FILTER_TYPE_PGIDLE_UNSET is impossible. Introduce a new filter for the purpose, DAMON_FILTER_TYPE_PGIDLE_SET. Link: https://lore.kernel.org/20260910142234.171562-1-sj@kernel.org Link: https://lore.kernel.org/20260910142234.171562-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 2 ++ 1 file changed, 2 insertions(+) diff --git a/include/linux/damon.h b/include/linux/damon.h index 871d26adf6ae57..16800b3dc379b9 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -774,11 +774,13 @@ struct damon_prep { * @DAMON_FILTER_TYPE_ANON: Anonymous pages. * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages. * @DAMON_FILTER_TYPE_PGIDLE_UNSET: Pgidle is unset. + * @DAMON_FILTER_TYPE_PGIDLE_SET: Pgidle is set. */ enum damon_filter_type { DAMON_FILTER_TYPE_ANON, DAMON_FILTER_TYPE_MEMCG, DAMON_FILTER_TYPE_PGIDLE_UNSET, + DAMON_FILTER_TYPE_PGIDLE_SET, }; /** From 1e226d16f4818722112593bc307d409fc30d56cb Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 10 Sep 2026 07:22:29 -0700 Subject: [PATCH 0574/1012] mm/damon/paddr: support DAMON_FILTER_TYPE_PGIDLE_SET Implement DAMON_FILTER_TYPE_PGIDLE_SET support on the physical address space DAMON operation set (paddr). Link: https://lore.kernel.org/20260910142234.171562-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/paddr.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index d7c81829445ba1..79195026b903a6 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -151,6 +151,12 @@ static bool damon_pa_filter_match(struct damon_filter *filter, else matched = damon_folio_young(folio); break; + case DAMON_FILTER_TYPE_PGIDLE_SET: + if (!folio) + matched = false; + else + matched = damon_folio_young(folio) == false; + break; default: return damon_ops_filter_match(filter, folio); } From 1406be1f5a1ea96ab5dd05f96a7dae6b8179b6c6 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 10 Sep 2026 07:22:30 -0700 Subject: [PATCH 0575/1012] mm/damon/vaddr: support DAMON_FILTER_TYPE_PGIDLE_SET Implement DAMON_FILTER_TYPE_PGIDLE_SET support on the virtual address space DAMON operation set (vaddr). Link: https://lore.kernel.org/20260910142234.171562-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/vaddr.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 9a38dc89a156ef..7063356370c34b 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -550,6 +550,13 @@ static bool damon_va_filter_match(struct damon_filter *filter, matched = damon_va_young_addr(folio, pte, pmd, mm, addr); break; + case DAMON_FILTER_TYPE_PGIDLE_SET: + if (!folio) + matched = false; + else + matched = !damon_va_young_addr(folio, pte, pmd, mm, + addr); + break; default: return damon_ops_filter_match(filter, folio); } From e1f4528d0419622f50e45a7eba560c0ec651f137 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 10 Sep 2026 07:22:31 -0700 Subject: [PATCH 0576/1012] mm/damon/sysfs: support DAMON_FILTER_TYPE_PGIDLE_SET Extend DAMON sysfs interface to let users setup DAMON_FILTER_TYPE_PGIDLE_SET probe filter. Link: https://lore.kernel.org/20260910142234.171562-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index b576e97cbfdb8a..51fa506c879b02 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -1001,6 +1001,10 @@ damon_sysfs_filter_type_names[] = { .type = DAMON_FILTER_TYPE_PGIDLE_UNSET, .name = "pgidle_unset", }, + { + .type = DAMON_FILTER_TYPE_PGIDLE_SET, + .name = "pgidle_set", + }, }; static ssize_t type_show(struct kobject *kobj, From a878c908dc928cfe8b3be0f33d3e60d1af437cb9 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 10 Sep 2026 07:22:32 -0700 Subject: [PATCH 0577/1012] Docs/mm/damon/design: update for pgidle_set probe filter Update DAMON design document for the newly added pgidle_set probe filter type. Link: https://lore.kernel.org/20260910142234.171562-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index aac84de261aa8a..22b785cd11dfec 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -299,6 +299,7 @@ filter types. Currently below filter types are supported. - ``memcg``: Same to that for DAMOS filters. - ``pgidle_unset``: Matches if the page for the memory is marked as not access-idle. +- ``pgidle_set``: Matches if the page for the memory is marked as access-idle. If such probes are registered, DAMON executes the probes for each region's sampling memory when it does the access :ref:`sampling @@ -314,7 +315,7 @@ actions are registered, DAMON applies the actions to each region's sampling memory before starting the next sampling interval. Currently only one action, ``set_pgidle`` is supported. The action marks the page for the probing target memory as access-idle. This can be useful to be used together with -``pgidle_unset`` probe filter. +``pgidle_unset`` or ``pgidle_set`` probe filter. This is a sampling based mechanism. Hence, it is lightweight but the output may include some measurement errors. The output should be used with good From 12f144e11f04e4ea3c325abf240831f13e703a88 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Sun, 27 Sep 2026 10:54:30 +0800 Subject: [PATCH 0578/1012] mm/sparse-vmemmap: factor out shared vmemmap tail page allocation Patch series "mm: Switch device DAX to section-based vmemmap optimization", v5. This series is split out from the earlier, larger series "mm: Generalize HVO for HugeTLB and device DAX" [1]. While the parent series generalizes vmemmap optimization across HugeTLB and device DAX, this subset addresses a single, self-contained step: switching device DAX to the section-based sparse-vmemmap optimization infrastructure introduced for HugeTLB. After the HugeTLB conversion, optimized vmemmap state is described by the memory section and the sparse-vmemmap population path can allocate or reuse shared tail vmemmap pages based on that metadata. Device DAX still uses the older DAX-specific population model, including a separate tail vmemmap page reservation and architecture-specific logic to locate or populate reusable tail pages. This series makes device DAX use the same section-based model. Device DAX records the compound page order from pgmap->vmemmap_shift in section metadata before vmemmap population, uses the common per-zone shared tail vmemmap page, and drops the extra reserved tail page. The powerpc radix path is updated to use the same shared tail-page helper, so the generic and powerpc DAX paths follow the same reservation model. The first patches prepare the shared infrastructure by factoring out shared tail-page allocation, allocating the per-zone shared tail-page array dynamically, and introducing a generic CONFIG_VMEMMAP_OPTIMIZATION symbol. The middle patches move device DAX onto that infrastructure by recording the device DAX compound page order in memory-section metadata, using that metadata to back generic device DAX mappings with the common per-zone shared tail page, exposing the shared helpers so the powerpc radix path can use the same model, and dropping the extra DAX-only tail page reservation and the now-unused section accounting arguments. The final patch updates the documentation for the new DAX layout. This is intended to be the third smaller step toward the broader HVO generalization. The wider HVO consolidation between HugeTLB and device DAX is left for follow-up series. This patch (of 12): HugeTLB and sparse-vmemmap each have their own helper to allocate the shared vmemmap tail page used by vmemmap optimization. Factor that logic into a common vmemmap_shared_tail_page() helper. It allocates the page through vmemmap_alloc_block(), initializes the tail struct pages, and uses cmpxchg() to install the per-zone shared page. This removes duplicate allocation logic while handling both early boot and runtime allocation through the same helper. Link: https://lore.kernel.org/20260927025441.741633-1-songmuchun@bytedance.com Link: https://lore.kernel.org/20260927025441.741633-2-songmuchun@bytedance.com Link: https://lore.kernel.org/all/20260513130542.35604-1-songmuchun@bytedance.com/ [1] Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: Mike Rapoport (Microsoft) Cc: David Hildenbrand Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- mm/hugetlb_vmemmap.c | 29 +----------------- mm/sparse-vmemmap.c | 70 ++++++++++++++++++++------------------------ mm/sparse.h | 3 ++ 3 files changed, 36 insertions(+), 66 deletions(-) diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index f977d0a7e00274..76765c97ff68b4 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -19,7 +19,6 @@ #include #include "hugetlb_vmemmap.h" #include "sparse.h" -#include "internal.h" /** * struct vmemmap_remap_walk - walk vmemmap page table @@ -493,32 +492,6 @@ static bool vmemmap_should_optimize_folio(const struct hstate *h, struct folio * return true; } -static struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) -{ - const unsigned int idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; - struct page *tail, *p; - int node = zone_to_nid(zone); - - tail = READ_ONCE(zone->vmemmap_tails[idx]); - if (likely(tail)) - return tail; - - tail = alloc_pages_node(node, GFP_KERNEL | __GFP_ZERO, 0); - if (!tail) - return NULL; - - p = page_to_virt(tail); - for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) - init_compound_tail(p + i, NULL, order, zone); - - if (cmpxchg(&zone->vmemmap_tails[idx], NULL, tail)) { - __free_page(tail); - tail = READ_ONCE(zone->vmemmap_tails[idx]); - } - - return tail; -} - static int __hugetlb_vmemmap_optimize_folio(const struct hstate *h, struct folio *folio, struct list_head *vmemmap_pages, @@ -535,7 +508,7 @@ static int __hugetlb_vmemmap_optimize_folio(const struct hstate *h, return ret; nid = folio_nid(folio); - vmemmap_tail = vmemmap_get_tail(h->order, folio_zone(folio)); + vmemmap_tail = vmemmap_shared_tail_page(h->order, folio_zone(folio)); if (!vmemmap_tail) return -ENOMEM; diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index f22d815d7af04e..9b00085122b23f 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -42,27 +42,13 @@ #include "mm_init.h" #include "sparse.h" -/* - * Allocate a block of memory to be used to back the virtual memory map - * or to back the page tables that are used to create the mapping. - * Uses the main allocators if they are available, else bootmem. - */ - -static void * __ref __earlyonly_bootmem_alloc(int node, - unsigned long size, - unsigned long align, - unsigned long goal) -{ - return memmap_alloc(size, align, goal, node, false); -} - -void * __meminit vmemmap_alloc_block(unsigned long size, int node) +void __ref *vmemmap_alloc_block(unsigned long size, int node) { /* If the main allocator is up use that, fallback to bootmem. */ if (slab_is_available()) { gfp_t gfp_mask = GFP_KERNEL|__GFP_RETRY_MAYFAIL|__GFP_NOWARN; int order = get_order(size); - static bool warned __meminitdata; + static bool warned; struct page *page; page = alloc_pages_node(node, gfp_mask, order); @@ -76,8 +62,7 @@ void * __meminit vmemmap_alloc_block(unsigned long size, int node) } return NULL; } else - return __earlyonly_bootmem_alloc(node, size, size, - __pa(MAX_DMA_ADDRESS)); + return memmap_alloc(size, size, __pa(MAX_DMA_ADDRESS), node, false); } static void * __meminit altmap_alloc_block_buf(unsigned long size, @@ -185,34 +170,43 @@ static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) } #ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP -static __meminit struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) +struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zone) { - struct page *p, *tail; - unsigned int idx; - int node = zone_to_nid(zone); + void *addr; + struct page *page; + const unsigned int idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; - if (WARN_ON_ONCE(order < VMEMMAP_OPTIMIZATION_MIN_ORDER)) - return NULL; - if (WARN_ON_ONCE(order > MAX_FOLIO_ORDER)) + if (WARN_ON_ONCE(idx >= VMEMMAP_OPTIMIZATION_NR_ORDERS)) return NULL; - idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; - tail = zone->vmemmap_tails[idx]; - if (tail) - return tail; - p = vmemmap_alloc_block_zero(PAGE_SIZE, node); - if (!p) + page = READ_ONCE(zone->vmemmap_tails[idx]); + if (likely(page)) + return page; + + addr = vmemmap_alloc_block(PAGE_SIZE, zone_to_nid(zone)); + if (!addr) return NULL; - for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) - init_compound_tail(p + i, NULL, order, zone); - tail = virt_to_page(p); - zone->vmemmap_tails[idx] = tail; + for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) { + page = (struct page *)addr + i; + mm_zero_struct_page(page); + init_compound_tail(page, NULL, order, zone); + } - return tail; + page = virt_to_page(addr); + if (cmpxchg(&zone->vmemmap_tails[idx], NULL, page) != NULL) { + if (slab_is_available()) + __free_page(page); + else + memblock_free(addr, PAGE_SIZE); + page = READ_ONCE(zone->vmemmap_tails[idx]); + } + + return page; } #else -static inline struct page *vmemmap_get_tail(unsigned int order, struct zone *zone) +static inline struct page *vmemmap_shared_tail_page(unsigned int order, + struct zone *zone) { return NULL; } @@ -229,7 +223,7 @@ static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, return vmemmap_alloc_block_buf(PAGE_SIZE, node, altmap); zone = pfn_to_zone(pfn, node); - page = vmemmap_get_tail(order, zone); + page = vmemmap_shared_tail_page(order, zone); if (!page) return NULL; diff --git a/mm/sparse.h b/mm/sparse.h index d3a71ef4fad0fe..6e7aaeaa559471 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -142,6 +142,9 @@ static inline void sparse_sections_init(void) {} * mm/sparse-vmemmap.c */ #ifdef CONFIG_SPARSEMEM_VMEMMAP +#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +struct page *vmemmap_shared_tail_page(unsigned int order, struct zone *zone); +#endif void sparse_init_subsection_map(void); int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, struct vmem_altmap *altmap, struct dev_pagemap *pgmap); From b8375874c07c7a312fe2f342aa82d1dfe3e1db89 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Sun, 27 Sep 2026 10:54:31 +0800 Subject: [PATCH 0579/1012] mm/sparse-vmemmap: allocate shared tail page array dynamically Commit 622026e87c40 ("mm/hugetlb: remove fake head pages") added the per-zone vmemmap_tails array. Its size depends on MAX_FOLIO_ORDER, which had been moved to mmzone.h in preparation for the array. PUD_ORDER is defined by linux/pgtable.h, which cannot be included from mmzone.h without creating an include cycle. It was therefore open-coded as PUD_SHIFT - PAGE_SHIFT. This removed the dependency on PUD_ORDER, but not the underlying dependency on architecture page-table definitions. PUD_SHIFT is generally provided by architecture page-table headers, which are not guaranteed to have been included when mmzone.h is parsed. The dependency remained hidden because vmemmap_tails was originally guarded by CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP. Under that condition, MAX_FOLIO_ORDER resolves to either MAX_PAGE_ORDER or the fixed HugeTLB limit, rather than the PUD_SHIFT-based definition. Device DAX, however, does not require CONFIG_HUGETLB_PAGE. When it is converted to use section-based vmemmap optimization, MAX_FOLIO_ORDER can resolve to PUD_SHIFT - PAGE_SHIFT while it is being used to size vmemmap_tails. This would make struct zone depend on architecture page-table definitions being available when mmzone.h is parsed. Replace the embedded array with a pointer and allocate it on first use. This moves the order-count evaluation into sparse-vmemmap.c, after the architecture page-table definitions are available, and removes the dependency from mmzone.h. Removing the compile-time array also removes the original reason for keeping MAX_FOLIO_ORDER and the vmemmap optimization sizing definitions in mmzone.h. Follow-up cleanups can place each definition in the header owned by its respective subsystem. Link: https://lore.kernel.org/20260927025441.741633-3-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Qi Zheng Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- include/linux/mmzone.h | 7 +------ mm/sparse-vmemmap.c | 35 +++++++++++++++++++++++++++++++---- 2 files changed, 32 insertions(+), 10 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index acd94cecc0d399..68807ff7f9465e 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -113,11 +113,6 @@ (VMEMMAP_OPTIMIZATION_PAGES * PAGE_SIZE / sizeof(struct page)) #define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) -#define __VMEMMAP_OPTIMIZATION_NR_ORDERS \ - (MAX_FOLIO_ORDER - VMEMMAP_OPTIMIZATION_MIN_ORDER + 1) -#define VMEMMAP_OPTIMIZATION_NR_ORDERS \ - (__VMEMMAP_OPTIMIZATION_NR_ORDERS > 0 ? __VMEMMAP_OPTIMIZATION_NR_ORDERS : 0) - enum migratetype { MIGRATE_UNMOVABLE, MIGRATE_MOVABLE, @@ -1156,7 +1151,7 @@ struct zone { atomic_long_t vm_stat[NR_VM_ZONE_STAT_ITEMS]; atomic_long_t vm_numa_event[NR_VM_NUMA_EVENT_ITEMS]; #ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP - struct page *vmemmap_tails[VMEMMAP_OPTIMIZATION_NR_ORDERS]; + struct page **vmemmap_tails; #endif } ____cacheline_internodealigned_in_smp; diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 9b00085122b23f..46d9a25b327593 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -170,16 +170,43 @@ static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) } #ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#define VMEMMAP_OPTIMIZATION_NR_ORDERS (MAX_FOLIO_ORDER - VMEMMAP_OPTIMIZATION_MIN_ORDER + 1) + +static __ref struct page **vmemmap_tails_alloc(struct zone *zone) +{ + struct page **pages; + const size_t size = array_size(VMEMMAP_OPTIMIZATION_NR_ORDERS, sizeof(*pages)); + + pages = slab_is_available() ? kzalloc_objs(*pages, VMEMMAP_OPTIMIZATION_NR_ORDERS) : + memblock_alloc(size, __alignof__(*pages)); + if (!pages) + return NULL; + + if (cmpxchg(&zone->vmemmap_tails, NULL, pages) != NULL) { + if (slab_is_available()) + kfree(pages); + else + memblock_free(pages, size); + pages = READ_ONCE(zone->vmemmap_tails); + } + + return pages; +} + struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zone) { void *addr; - struct page *page; + struct page *page, **pages; const unsigned int idx = order - VMEMMAP_OPTIMIZATION_MIN_ORDER; if (WARN_ON_ONCE(idx >= VMEMMAP_OPTIMIZATION_NR_ORDERS)) return NULL; - page = READ_ONCE(zone->vmemmap_tails[idx]); + pages = READ_ONCE(zone->vmemmap_tails) ? : vmemmap_tails_alloc(zone); + if (!pages) + return NULL; + + page = READ_ONCE(pages[idx]); if (likely(page)) return page; @@ -194,12 +221,12 @@ struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zon } page = virt_to_page(addr); - if (cmpxchg(&zone->vmemmap_tails[idx], NULL, page) != NULL) { + if (cmpxchg(&pages[idx], NULL, page) != NULL) { if (slab_is_available()) __free_page(page); else memblock_free(addr, PAGE_SIZE); - page = READ_ONCE(zone->vmemmap_tails[idx]); + page = READ_ONCE(pages[idx]); } return page; From f31114c4cc2715f0311ecd53e22f2484e7ddd0a4 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Sun, 27 Sep 2026 10:54:32 +0800 Subject: [PATCH 0580/1012] mm/sparse-vmemmap: introduce CONFIG_VMEMMAP_OPTIMIZATION The section-based vmemmap optimization infrastructure is guarded by CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP, but it can also be used by ZONE_DEVICE users that set dev_pagemap::vmemmap_shift. Introduce CONFIG_VMEMMAP_OPTIMIZATION as a common config for the shared infrastructure. Select the new option from HUGETLB_PAGE_OPTIMIZE_VMEMMAP and from ZONE_DEVICE when the architecture opts in to DAX vmemmap optimization, and use it to guard the generic sparse-vmemmap state and helpers. Link: https://lore.kernel.org/20260927025441.741633-4-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- arch/x86/entry/vdso/vdso32/fake_32bit_build.h | 2 +- fs/Kconfig | 1 + include/linux/mm.h | 3 +++ include/linux/mmzone.h | 10 +++++----- include/linux/page-flags.h | 5 ++--- mm/Kconfig | 5 +++++ mm/sparse-vmemmap.c | 2 +- mm/sparse.h | 6 +++--- 8 files changed, 21 insertions(+), 13 deletions(-) diff --git a/arch/x86/entry/vdso/vdso32/fake_32bit_build.h b/arch/x86/entry/vdso/vdso32/fake_32bit_build.h index bc3e549795c3f3..72a92cb9b53d3b 100644 --- a/arch/x86/entry/vdso/vdso32/fake_32bit_build.h +++ b/arch/x86/entry/vdso/vdso32/fake_32bit_build.h @@ -11,7 +11,7 @@ #undef CONFIG_PGTABLE_LEVELS #undef CONFIG_ILLEGAL_POINTER_VALUE #undef CONFIG_SPARSEMEM_VMEMMAP -#undef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#undef CONFIG_VMEMMAP_OPTIMIZATION #undef CONFIG_NR_CPUS #undef CONFIG_PARAVIRT_XXL diff --git a/fs/Kconfig b/fs/Kconfig index d1c210c6508f0a..1454b7fe9641fd 100644 --- a/fs/Kconfig +++ b/fs/Kconfig @@ -278,6 +278,7 @@ config HUGETLB_PAGE_OPTIMIZE_VMEMMAP def_bool HUGETLB_PAGE depends on ARCH_WANT_OPTIMIZE_HUGETLB_VMEMMAP depends on SPARSEMEM_VMEMMAP + select VMEMMAP_OPTIMIZATION config HUGETLB_PMD_PAGE_TABLE_SHARING def_bool HUGETLB_PAGE diff --git a/include/linux/mm.h b/include/linux/mm.h index c49ef99b4413b4..070ce27e9cd3cd 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -5175,6 +5175,9 @@ static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap, unsigned long nr_pages; unsigned long nr_vmemmap_pages; + if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) + return false; + if (!pgmap || !is_power_of_2(sizeof(struct page))) return false; diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 68807ff7f9465e..ee9cbaaa63f4f2 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -102,9 +102,9 @@ * * HVO which is only active if the size of struct page is a power of 2. */ -#define MAX_FOLIO_VMEMMAP_ALIGN \ - (IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP) && \ - is_power_of_2(sizeof(struct page)) ? \ +#define MAX_FOLIO_VMEMMAP_ALIGN \ + (IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION) && \ + is_power_of_2(sizeof(struct page)) ? \ MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) /* The number of retained vmemmap pages with HVO enabled. */ @@ -1150,7 +1150,7 @@ struct zone { /* Zone statistics */ atomic_long_t vm_stat[NR_VM_ZONE_STAT_ITEMS]; atomic_long_t vm_numa_event[NR_VM_NUMA_EVENT_ITEMS]; -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#ifdef CONFIG_VMEMMAP_OPTIMIZATION struct page **vmemmap_tails; #endif } ____cacheline_internodealigned_in_smp; @@ -2014,7 +2014,7 @@ struct mem_section { unsigned long section_mem_map; struct mem_section_usage *usage; -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#ifdef CONFIG_VMEMMAP_OPTIMIZATION /* * Normally, sections hold regular (order-0) pages. However, for * sections with HVO enabled, this tracks the compound page order diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index 86dd0470da1173..7080a6a1a79e72 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -208,14 +208,13 @@ enum pageflags { static __always_inline bool compound_info_has_mask(void) { /* - * Limit mask usage to HugeTLB vmemmap optimization (HVO) where it - * makes a difference. + * Limit mask usage to HVO where it makes a difference. * * The approach with mask would work in the wider set of conditions, * but it requires validating that struct pages are naturally aligned * for all orders up to the MAX_FOLIO_ORDER, which can be tricky. */ - if (!IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP)) + if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) return false; return is_power_of_2(sizeof(struct page)); diff --git a/mm/Kconfig b/mm/Kconfig index bc7befafb47b57..30170a936f1fc0 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -461,6 +461,10 @@ config SPARSEMEM_VMEMMAP pfn_to_page and page_to_pfn operations. This is the most efficient option when sufficient kernel resources are available. +config VMEMMAP_OPTIMIZATION + bool + depends on SPARSEMEM_VMEMMAP + # # Select this config option from the architecture Kconfig, if it is preferred # to enable the feature of HugeTLB/dev_dax vmemmap optimization. @@ -1220,6 +1224,7 @@ config ZONE_DMA32 config ZONE_DEVICE bool "Device memory (pmem, HMM, etc...) hotplug support" depends on MEMORY_HOTREMOVE + select VMEMMAP_OPTIMIZATION if ARCH_WANT_OPTIMIZE_DAX_VMEMMAP select XARRAY_MULTI help diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 46d9a25b327593..38c36399f53e4e 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -169,7 +169,7 @@ static void * __meminit vmemmap_alloc_block_zero(unsigned long size, int node) return p; } -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#ifdef CONFIG_VMEMMAP_OPTIMIZATION #define VMEMMAP_OPTIMIZATION_NR_ORDERS (MAX_FOLIO_ORDER - VMEMMAP_OPTIMIZATION_MIN_ORDER + 1) static __ref struct page **vmemmap_tails_alloc(struct zone *zone) diff --git a/mm/sparse.h b/mm/sparse.h index 6e7aaeaa559471..326ad43bb5c37a 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -10,7 +10,7 @@ #include -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#ifdef CONFIG_VMEMMAP_OPTIMIZATION static inline unsigned int section_compound_order(const struct mem_section *section) { return section->compound_page_order; @@ -75,7 +75,7 @@ static inline bool vmemmap_optimizable_pfn(unsigned long pfn) static inline bool vmemmap_optimizable_order(unsigned int order) { - if (!IS_ENABLED(CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP)) + if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) return false; if (!is_power_of_2(sizeof(struct page))) @@ -142,7 +142,7 @@ static inline void sparse_sections_init(void) {} * mm/sparse-vmemmap.c */ #ifdef CONFIG_SPARSEMEM_VMEMMAP -#ifdef CONFIG_HUGETLB_PAGE_OPTIMIZE_VMEMMAP +#ifdef CONFIG_VMEMMAP_OPTIMIZATION struct page *vmemmap_shared_tail_page(unsigned int order, struct zone *zone); #endif void sparse_init_subsection_map(void); From e1cac862b1bf3217835b60e098b06b45e4a1b9da Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Sun, 27 Sep 2026 10:54:33 +0800 Subject: [PATCH 0581/1012] mm/sparse-vmemmap: open-code init_compound_tail() init_compound_tail() is only used by vmemmap_shared_tail_page(), where the shared tail page setup intentionally passes NULL as the compound head. Keeping this helper in mm/internal.h exposes that special case to the rest of the MM code and can make the NULL head argument look generally valid. Open-code the initialization at the only call site so the special-case use stays local to sparse vmemmap optimization. No functional change intended. Link: https://lore.kernel.org/20260927025441.741633-5-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- mm/internal.h | 9 --------- mm/sparse-vmemmap.c | 5 ++++- 2 files changed, 4 insertions(+), 10 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index da14c56fb24e11..0dca33db068f6b 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -786,15 +786,6 @@ static inline void prep_compound_tail(struct page *tail, VM_WARN_ON_ONCE(tail->private); } -static inline void init_compound_tail(struct page *tail, - const struct page *head, unsigned int order, struct zone *zone) -{ - atomic_set(&tail->_mapcount, -1); - set_page_node(tail, zone_to_nid(zone)); - set_page_zone(tail, zone_idx(zone)); - prep_compound_tail(tail, head, order); -} - #if defined CONFIG_COMPACTION || defined CONFIG_CMA /* diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 38c36399f53e4e..e4dae98ba7f888 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -217,7 +217,10 @@ struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zon for (int i = 0; i < PAGE_SIZE / sizeof(struct page); i++) { page = (struct page *)addr + i; mm_zero_struct_page(page); - init_compound_tail(page, NULL, order, zone); + atomic_set(&page->_mapcount, -1); + set_page_node(page, zone_to_nid(zone)); + set_page_zone(page, zone_idx(zone)); + prep_compound_tail(page, NULL, order); } page = virt_to_page(addr); From db0b26fc8330fe9e9219a28b178043badbd9ff30 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Sun, 27 Sep 2026 10:54:34 +0800 Subject: [PATCH 0582/1012] mm/sparse-vmemmap: prepare DAX vmemmap population for compound page orders Device DAX still uses vmemmap_populate_compound_pages() to populate its compound-page vmemmap mappings. That helper allocates the head and first tail vmemmap pages explicitly, then reuses the first tail page for the remaining tail page mappings. Device DAX is being moved to the section-based vmemmap optimization infrastructure, but it cannot switch to the generic section-based population path yet. Once a later patch records the DAX compound page order in section metadata, DAX head and first-tail PFNs can look optimizable to the generic helpers as well. Add a DAX-specific population flag for this transition. It keeps DAX head/first-tail allocations on the normal vmemmap allocation path, while preserving the existing page reference for reused DAX tail mappings. Link: https://lore.kernel.org/20260927025441.741633-6-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Cc: David Hildenbrand Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- mm/sparse-vmemmap.c | 27 +++++++++++++++------------ 1 file changed, 15 insertions(+), 12 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index e4dae98ba7f888..2457ea2c6dca1f 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -35,8 +35,8 @@ /* * Flags for vmemmap_populate_range and friends. */ -/* Get a ref on the head page struct page, for ZONE_DEVICE compound pages */ -#define VMEMMAP_POPULATE_PAGEREF 0x0001 +/* Vmemmap population for ZONE_DEVICE compound pages */ +#define VMEMMAP_POPULATE_DAX 0x0001 #include "internal.h" #include "mm_init.h" @@ -243,13 +243,17 @@ static inline struct page *vmemmap_shared_tail_page(unsigned int order, #endif static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, - struct vmem_altmap *altmap) + struct vmem_altmap *altmap, unsigned long flags) { struct zone *zone; struct page *page; const unsigned int order = pfn_to_section_compound_order(pfn); - if (!vmemmap_optimizable_pfn(pfn)) + /* + * Device DAX still relies on vmemmap_populate_compound_pages() for + * head/first-tail allocation and tail-page reuse. + */ + if (!vmemmap_optimizable_pfn(pfn) || flags & VMEMMAP_POPULATE_DAX) return vmemmap_alloc_block_buf(PAGE_SIZE, node, altmap); zone = pfn_to_zone(pfn, node); @@ -271,7 +275,7 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in pte_t entry; if (ptpfn == (unsigned long)-1) { - void *p = vmemmap_alloc_pte(pfn, node, altmap); + void *p = vmemmap_alloc_pte(pfn, node, altmap, flags); if (!p) return NULL; @@ -286,7 +290,7 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in * and through vmemmap_populate_compound_pages() when * slab is available. */ - if (flags & VMEMMAP_POPULATE_PAGEREF) + if (flags & VMEMMAP_POPULATE_DAX) get_page(pfn_to_page(ptpfn)); } entry = pfn_pte(ptpfn, PAGE_KERNEL); @@ -546,6 +550,7 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, unsigned long size, addr; pte_t *pte; int rc; + unsigned long flags = VMEMMAP_POPULATE_DAX; if (reuse_compound_section(start_pfn, pgmap)) { pte = compound_section_tail_page(start); @@ -557,8 +562,7 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, * with just tail struct pages. */ return vmemmap_populate_range(start, end, node, NULL, - pte_pfn(ptep_get(pte)), - VMEMMAP_POPULATE_PAGEREF); + pte_pfn(ptep_get(pte)), flags); } size = min(end - start, pgmap_vmemmap_nr(pgmap) * sizeof(struct page)); @@ -566,13 +570,13 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, unsigned long next, last = addr + size; /* Populate the head page vmemmap page */ - pte = vmemmap_populate_address(addr, node, NULL, -1, 0); + pte = vmemmap_populate_address(addr, node, NULL, -1, flags); if (!pte) return -ENOMEM; /* Populate the tail pages vmemmap page */ next = addr + PAGE_SIZE; - pte = vmemmap_populate_address(next, node, NULL, -1, 0); + pte = vmemmap_populate_address(next, node, NULL, -1, flags); if (!pte) return -ENOMEM; @@ -582,8 +586,7 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, */ next += PAGE_SIZE; rc = vmemmap_populate_range(next, last, node, NULL, - pte_pfn(ptep_get(pte)), - VMEMMAP_POPULATE_PAGEREF); + pte_pfn(ptep_get(pte)), flags); if (rc) return -ENOMEM; } From 6a86e87a6c471ad30d49f05f9d34e72fe61375f6 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Sun, 27 Sep 2026 10:54:35 +0800 Subject: [PATCH 0583/1012] mm/sparse-vmemmap: set compound page order for device DAX Device DAX can use vmemmap optimization only when a full section is populated with a compound-page geometry. Record that geometry as the compound page order in section metadata before populating the section, so later vmemmap accounting and population decisions can use the section state directly. Clear the compound page order when the section becomes empty again. Also reject partial additions to a section that already has optimized vmemmap mappings. compound_nr_pages() determines how many struct pages to initialize with a section as the smallest granularity. A section therefore cannot safely mix optimized and ordinary vmemmap layouts. Partial additions continue to use ordinary vmemmap population, so they do not save vmemmap memory. Such additions are uncommon, and the lost saving is negligible. Link: https://lore.kernel.org/20260927025441.741633-7-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- mm/mm_init.c | 15 +++++---------- mm/sparse-vmemmap.c | 16 ++++++++++++---- 2 files changed, 17 insertions(+), 14 deletions(-) diff --git a/mm/mm_init.c b/mm/mm_init.c index 97e0158d2aca5b..efffa8609b8592 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1049,16 +1049,11 @@ static void zone_device_page_init_from_template(struct page *page, * of an altmap. See vmemmap_populate_compound_pages(). */ static inline unsigned long compound_nr_pages(unsigned long pfn, - struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { - /* - * If DAX memory is hot-plugged into an unoccupied subsection - * of an early section, the unoptimized boot memmap is reused. - * See section_activate(). - */ - if (early_section(__pfn_to_section(pfn)) || - !vmemmap_can_optimize(altmap, pgmap)) + const struct mem_section *ms = __pfn_to_section(pfn); + + if (!section_vmemmap_optimizable(ms)) return pgmap_vmemmap_nr(pgmap); return VMEMMAP_RESERVE_NR * (PAGE_SIZE / sizeof(struct page)); @@ -1144,7 +1139,7 @@ void __ref memmap_init_zone_device(struct zone *zone, memcpy(&template, page, sizeof(*page)); if (pfns_per_compound != 1) memmap_init_compound(page, pfn, zone_idx, nid, pgmap, - compound_nr_pages(pfn, altmap, pgmap)); + compound_nr_pages(pfn, pgmap)); pfn += pfns_per_compound; /* Initialize the remaining head pages from template. */ @@ -1160,7 +1155,7 @@ void __ref memmap_init_zone_device(struct zone *zone, continue; memmap_init_compound(page, pfn, zone_idx, nid, pgmap, - compound_nr_pages(pfn, altmap, pgmap)); + compound_nr_pages(pfn, pgmap)); } pageblock_migratetype_init_range(start_pfn, nr_pages, MIGRATE_MOVABLE, diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 2457ea2c6dca1f..ca2470e96a74e9 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -135,14 +135,14 @@ int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages struct vmem_altmap *altmap, struct dev_pagemap *pgmap) { const struct mem_section *ms = __pfn_to_section(pfn); - const int order = pgmap ? pgmap->vmemmap_shift : section_compound_order(ms); + const int order = section_compound_order(ms); const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; const unsigned long pages_per_compound = 1UL << order; VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); VM_WARN_ON_ONCE(nr_pages > PAGES_PER_SECTION); - if (!vmemmap_can_optimize(altmap, pgmap) && !section_vmemmap_optimizable(ms)) + if (!section_vmemmap_optimizable(ms)) return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); if (order < PFN_SECTION_SHIFT) { @@ -608,7 +608,7 @@ struct page * __meminit __populate_section_memmap(unsigned long pfn, !IS_ALIGNED(nr_pages, PAGES_PER_SUBSECTION))) return NULL; - if (vmemmap_can_optimize(altmap, pgmap)) + if (pgmap && section_vmemmap_optimizable(__pfn_to_section(pfn))) r = vmemmap_populate_compound_pages(pfn, start, end, nid, pgmap); else r = vmemmap_populate(start, end, nid, altmap); @@ -827,8 +827,10 @@ static void section_deactivate(unsigned long pfn, unsigned long nr_pages, else if (memmap) free_map_bootmem(memmap); - if (empty) + if (empty) { ms->section_mem_map = (unsigned long)NULL; + section_set_compound_order(ms, 0); + } } static struct page * __meminit section_activate(int nid, unsigned long pfn, @@ -838,8 +840,13 @@ static struct page * __meminit section_activate(int nid, unsigned long pfn, struct mem_section *ms = __pfn_to_section(pfn); struct mem_section_usage *usage = NULL; struct page *memmap; + unsigned int order; int rc; + order = vmemmap_can_optimize(altmap, pgmap) ? pgmap->vmemmap_shift : 0; + if (nr_pages < PAGES_PER_SECTION && section_compound_order(ms)) + return ERR_PTR(-EOPNOTSUPP); + if (!ms->usage) { usage = kzalloc(mem_section_usage_size(), GFP_KERNEL); if (!usage) @@ -865,6 +872,7 @@ static struct page * __meminit section_activate(int nid, unsigned long pfn, if (nr_pages < PAGES_PER_SECTION && early_section(ms)) return pfn_to_page(pfn); + section_set_compound_order_range(pfn, nr_pages, order); memmap = populate_section_memmap(pfn, nr_pages, nid, altmap, pgmap); if (!memmap) { section_deactivate(pfn, nr_pages, altmap, pgmap); From 987cc97e1dadae67bf74acd189429745aa0d0c76 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Sun, 27 Sep 2026 10:54:36 +0800 Subject: [PATCH 0584/1012] mm/sparse-vmemmap: switch device DAX to shared tail vmemmap pages HugeTLB vmemmap optimization now uses per-zone shared tail vmemmap pages. Device DAX has not been switched to that mechanism yet. Switch device DAX to vmemmap_shared_tail_page() as well. This aligns DAX with HugeTLB by using the common per-zone shared tail vmemmap page. The optimization is enabled only for DEV-DAX through pgmap->vmemmap_shift, which supplies the compound page order recorded in section metadata before vmemmap population. Unlike FS-DAX, DEV-DAX does not modify tail struct pages, so sharing them is safe. Since the shared tail page can now back ZONE_DEVICE vmemmap mappings, initialize its entries with PG_reserved for device zones. Also skip poisoning vmemmap-optimizable sections while their struct pages may be shared. Link: https://lore.kernel.org/20260927025441.741633-8-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Cc: David Hildenbrand Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- include/linux/mmzone.h | 10 +++++++++ mm/memory_hotplug.c | 6 ++++-- mm/sparse-vmemmap.c | 47 ++++++++++++++---------------------------- 3 files changed, 29 insertions(+), 34 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index ee9cbaaa63f4f2..cd68c1904c9118 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2143,11 +2143,21 @@ static inline int online_device_section(const struct mem_section *section) return section && ((section->section_mem_map & flags) == flags); } + +static inline struct zone *device_zone(int nid) +{ + return &NODE_DATA(nid)->node_zones[ZONE_DEVICE]; +} #else static inline int online_device_section(const struct mem_section *section) { return 0; } + +static inline struct zone *device_zone(int nid) +{ + return NULL; +} #endif static inline int online_section_nr(unsigned long nr) diff --git a/mm/memory_hotplug.c b/mm/memory_hotplug.c index b428da66d279c0..d7a59167bec43c 100644 --- a/mm/memory_hotplug.c +++ b/mm/memory_hotplug.c @@ -43,6 +43,7 @@ #include "mm_init.h" #include "page_alloc.h" #include "shuffle.h" +#include "sparse.h" enum { MEMMAP_ON_MEMORY_DISABLE = 0, @@ -554,8 +555,9 @@ void remove_pfn_range_from_zone(struct zone *zone, /* Select all remaining pages up to the next section boundary */ cur_nr_pages = min(end_pfn - pfn, SECTION_ALIGN_UP(pfn + 1) - pfn); - page_init_poison(pfn_to_page(pfn), - sizeof(struct page) * cur_nr_pages); + if (!section_vmemmap_optimizable(__pfn_to_section(pfn))) + page_init_poison(pfn_to_page(pfn), + sizeof(struct page) * cur_nr_pages); } /* diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index ca2470e96a74e9..2bc78aa053a1ad 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -221,6 +221,8 @@ struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zon set_page_node(page, zone_to_nid(zone)); set_page_zone(page, zone_idx(zone)); prep_compound_tail(page, NULL, order); + if (zone_is_zone_device(zone)) + __SetPageReserved(page); } page = virt_to_page(addr); @@ -525,23 +527,6 @@ static bool __meminit reuse_compound_section(unsigned long start_pfn, return !IS_ALIGNED(offset, nr_pages) && nr_pages > PAGES_PER_SUBSECTION; } -static pte_t * __meminit compound_section_tail_page(unsigned long addr) -{ - pte_t *pte; - - addr -= PAGE_SIZE; - - /* - * Assuming sections are populated sequentially, the previous section's - * page data can be reused. - */ - pte = pte_offset_kernel(pmd_off_k(addr), addr); - if (!pte) - return NULL; - - return pte; -} - static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, unsigned long start, unsigned long end, int node, @@ -551,21 +536,18 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, pte_t *pte; int rc; unsigned long flags = VMEMMAP_POPULATE_DAX; + struct page *page; + unsigned int order = pfn_to_section_compound_order(start_pfn); - if (reuse_compound_section(start_pfn, pgmap)) { - pte = compound_section_tail_page(start); - if (!pte) - return -ENOMEM; + page = vmemmap_shared_tail_page(order, device_zone(node)); + if (!page) + return -ENOMEM; - /* - * Reuse the page that was populated in the prior iteration - * with just tail struct pages. - */ + if (reuse_compound_section(start_pfn, pgmap)) return vmemmap_populate_range(start, end, node, NULL, - pte_pfn(ptep_get(pte)), flags); - } + page_to_pfn(page), flags); - size = min(end - start, pgmap_vmemmap_nr(pgmap) * sizeof(struct page)); + size = min(end - start, (1UL << order) * sizeof(struct page)); for (addr = start; addr < end; addr += size) { unsigned long next, last = addr + size; @@ -581,12 +563,12 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, return -ENOMEM; /* - * Reuse the previous page for the rest of tail pages + * Reuse the shared page for the rest of tail pages * See layout diagram in Documentation/mm/vmemmap_dedup.rst */ next += PAGE_SIZE; rc = vmemmap_populate_range(next, last, node, NULL, - pte_pfn(ptep_get(pte)), flags); + page_to_pfn(page), flags); if (rc) return -ENOMEM; } @@ -918,13 +900,14 @@ int __meminit sparse_add_section(int nid, unsigned long start_pfn, if (IS_ERR(memmap)) return PTR_ERR(memmap); + ms = __nr_to_section(section_nr); /* * Poison uninitialized struct pages in order to catch invalid flags * combinations. */ - page_init_poison(memmap, sizeof(struct page) * nr_pages); + if (!section_vmemmap_optimizable(ms)) + page_init_poison(memmap, sizeof(struct page) * nr_pages); - ms = __nr_to_section(section_nr); __section_mark_present(ms, section_nr); /* Align memmap to section boundary in the subsection case */ From 3b6da79e1d9b8821f88159f18b80c9e2f208b6d1 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Mon, 28 Sep 2026 12:41:48 +0800 Subject: [PATCH 0585/1012] mm-sparse-vmemmap-switch-device-dax-to-shared-tail-vmemmap-pages-fix Each PTE mapping the shared device DAX tail page takes a page reference. A sufficiently large range could therefore cycle the reference count back to zero if population were allowed to continue after it became non-positive. Use try_get_page() so further mappings fail once the reference count is no longer positive. The section population error path tears down mappings created for the failed section, while the warning makes this currently impractical limit visible. Link: https://lore.kernel.org/20260928044148.3300333-1-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Qi Zheng Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- mm/sparse-vmemmap.c | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 2bc78aa053a1ad..1874cdad2d05f3 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -286,14 +286,18 @@ static pte_t * __meminit vmemmap_pte_populate(pmd_t *pmd, unsigned long addr, in /* * When a PTE/PMD entry is freed from the init_mm * there's a free_pages() call to this page allocated - * above. Thus this get_page() is paired with the + * above. Thus this try_get_page() is paired with the * put_page_testzero() on the freeing path. * This can only called by certain ZONE_DEVICE path, * and through vmemmap_populate_compound_pages() when * slab is available. + * + * Use try_get_page() to prevent the shared page refcount + * from overflowing. */ - if (flags & VMEMMAP_POPULATE_DAX) - get_page(pfn_to_page(ptpfn)); + if ((flags & VMEMMAP_POPULATE_DAX) && + !try_get_page(pfn_to_page(ptpfn))) + return NULL; } entry = pfn_pte(ptpfn, PAGE_KERNEL); set_pte_at(&init_mm, addr, pte, entry); From 24ec847db4c6eea16077bca53673165f1e7aa9c6 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Sun, 27 Sep 2026 10:54:37 +0800 Subject: [PATCH 0586/1012] mm/sparse-vmemmap: move vmemmap optimization helpers to a public header The vmemmap optimization helpers currently live in mm/sparse.h, which is an internal MM header. That works for MM code, but prevents powerpc from using the same interfaces without including a private header. Move the declarations and inline helpers to vmemmap-optimization.h. This is a preparatory change for powerpc, which has its own vmemmap optimization implementation and needs to use the common vmemmap optimization interfaces from architecture code. Link: https://lore.kernel.org/20260927025441.741633-9-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- MAINTAINERS | 1 + arch/loongarch/include/asm/pgtable.h | 1 + arch/riscv/mm/init.c | 1 + include/linux/mmzone.h | 17 ----- include/linux/vmemmap-optimization.h | 109 +++++++++++++++++++++++++++ mm/hugetlb.c | 2 +- mm/hugetlb_vmemmap.c | 2 +- mm/sparse.h | 78 +------------------ 8 files changed, 115 insertions(+), 96 deletions(-) create mode 100644 include/linux/vmemmap-optimization.h diff --git a/MAINTAINERS b/MAINTAINERS index fe1d70ed5100b5..4689a021006002 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -12097,6 +12097,7 @@ F: Documentation/mm/hugetlbfs_reserv.rst F: Documentation/mm/vmemmap_dedup.rst F: fs/hugetlbfs/ F: include/linux/hugetlb.h +F: include/linux/vmemmap-optimization.h F: include/trace/events/hugetlbfs.h F: mm/hugetlb.c F: mm/hugetlb_cgroup.c diff --git a/arch/loongarch/include/asm/pgtable.h b/arch/loongarch/include/asm/pgtable.h index cf29a4c8ac593a..f876031351319a 100644 --- a/arch/loongarch/include/asm/pgtable.h +++ b/arch/loongarch/include/asm/pgtable.h @@ -72,6 +72,7 @@ #include #include +#include #include #include diff --git a/arch/riscv/mm/init.c b/arch/riscv/mm/init.c index fb37b0b67efec6..857f9a55039ce7 100644 --- a/arch/riscv/mm/init.c +++ b/arch/riscv/mm/init.c @@ -22,6 +22,7 @@ #include #include #include +#include #include #include diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index cd68c1904c9118..65de3bb13eb3ae 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -96,23 +96,6 @@ #define MAX_FOLIO_NR_PAGES (1UL << MAX_FOLIO_ORDER) -/* - * HugeTLB Vmemmap Optimization (HVO) requires struct pages of the head page to - * be naturally aligned with regard to the folio size. - * - * HVO which is only active if the size of struct page is a power of 2. - */ -#define MAX_FOLIO_VMEMMAP_ALIGN \ - (IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION) && \ - is_power_of_2(sizeof(struct page)) ? \ - MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) - -/* The number of retained vmemmap pages with HVO enabled. */ -#define VMEMMAP_OPTIMIZATION_PAGES 1 -#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES \ - (VMEMMAP_OPTIMIZATION_PAGES * PAGE_SIZE / sizeof(struct page)) -#define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) - enum migratetype { MIGRATE_UNMOVABLE, MIGRATE_MOVABLE, diff --git a/include/linux/vmemmap-optimization.h b/include/linux/vmemmap-optimization.h new file mode 100644 index 00000000000000..bd0974b262a4d1 --- /dev/null +++ b/include/linux/vmemmap-optimization.h @@ -0,0 +1,109 @@ +/* SPDX-License-Identifier: GPL-2.0-or-later */ +/* + * vmemmap-optimization.h + * + * Generic vmemmap optimization declarations. + * + * Author: Muchun Song + */ +#ifndef _LINUX_VMEMMAP_OPTIMIZATION_H +#define _LINUX_VMEMMAP_OPTIMIZATION_H + +#include +#include +#include +#include + +/* + * HugeTLB Vmemmap Optimization (HVO) requires struct pages of the head page to + * be naturally aligned with regard to the folio size. + * + * HVO which is only active if the size of struct page is a power of 2. + */ +#define MAX_FOLIO_VMEMMAP_ALIGN \ + (IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION) && \ + is_power_of_2(sizeof(struct page)) ? \ + MAX_FOLIO_NR_PAGES * sizeof(struct page) : 0) + +/* The number of retained vmemmap pages with HVO enabled. */ +#define VMEMMAP_OPTIMIZATION_PAGES 1 +#define VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES \ + (VMEMMAP_OPTIMIZATION_PAGES * PAGE_SIZE / sizeof(struct page)) +#define VMEMMAP_OPTIMIZATION_MIN_ORDER (ilog2(VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES) + 1) + +#ifdef CONFIG_VMEMMAP_OPTIMIZATION +static inline unsigned int section_compound_order(const struct mem_section *section) +{ + return section->compound_page_order; +} + +static inline void section_set_compound_order(struct mem_section *section, + unsigned int order) +{ + VM_WARN_ON(section_compound_order(section) && order && + section_compound_order(section) != order); + section->compound_page_order = order; +} + +static inline void section_set_compound_order_range(unsigned long pfn, + unsigned long nr_pages, unsigned int order) +{ + unsigned long section_nr = pfn_to_section_nr(pfn); + + if (!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)) + return; + + for (unsigned long i = 0; i < nr_pages / PAGES_PER_SECTION; i++) + section_set_compound_order(__nr_to_section(section_nr + i), order); +} + +static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) +{ + return section_compound_order(__pfn_to_section(pfn)); +} + +struct page *vmemmap_shared_tail_page(unsigned int order, struct zone *zone); +#else +static inline unsigned int section_compound_order(const struct mem_section *section) +{ + return 0; +} + +static inline void section_set_compound_order(struct mem_section *section, + unsigned int order) +{ +} + +static inline void section_set_compound_order_range(unsigned long pfn, + unsigned long nr_pages, unsigned int order) +{ +} + +static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) +{ + return 0; +} +#endif /* CONFIG_VMEMMAP_OPTIMIZATION */ + +static inline bool vmemmap_optimizable_pfn(unsigned long pfn) +{ + const unsigned int order = pfn_to_section_compound_order(pfn); + const unsigned long nr_pages = 1UL << order; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES; +} + +static inline bool vmemmap_optimizable_order(unsigned int order) +{ + if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) + return false; + + if (!is_power_of_2(sizeof(struct page))) + return false; + + return order >= VMEMMAP_OPTIMIZATION_MIN_ORDER; +} +#endif /* _LINUX_VMEMMAP_OPTIMIZATION_H */ diff --git a/mm/hugetlb.c b/mm/hugetlb.c index fd00141b089a98..2003439ea13c6f 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -38,6 +38,7 @@ #include #include #include +#include #include #include @@ -52,7 +53,6 @@ #include "hugetlb_cma.h" #include "hugetlb_internal.h" #include "mm_init.h" -#include "sparse.h" #include #define HUGE_BOOTMEM_ZONES_VALID BIT(0) diff --git a/mm/hugetlb_vmemmap.c b/mm/hugetlb_vmemmap.c index 76765c97ff68b4..0057fa2a16a891 100644 --- a/mm/hugetlb_vmemmap.c +++ b/mm/hugetlb_vmemmap.c @@ -15,10 +15,10 @@ #include #include #include +#include #include #include "hugetlb_vmemmap.h" -#include "sparse.h" /** * struct vmemmap_remap_walk - walk vmemmap page table diff --git a/mm/sparse.h b/mm/sparse.h index 326ad43bb5c37a..a5111087ee3a3d 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -9,80 +9,7 @@ #define __MM_SPARSE_H #include - -#ifdef CONFIG_VMEMMAP_OPTIMIZATION -static inline unsigned int section_compound_order(const struct mem_section *section) -{ - return section->compound_page_order; -} - -static inline void section_set_compound_order(struct mem_section *section, - unsigned int order) -{ - VM_WARN_ON(section_compound_order(section) && order && - section_compound_order(section) != order); - section->compound_page_order = order; -} - -static inline void section_set_compound_order_range(unsigned long pfn, - unsigned long nr_pages, unsigned int order) -{ - unsigned long section_nr = pfn_to_section_nr(pfn); - - if (!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)) - return; - - for (unsigned long i = 0; i < nr_pages / PAGES_PER_SECTION; i++) - section_set_compound_order(__nr_to_section(section_nr + i), order); -} - -static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) -{ - return section_compound_order(__pfn_to_section(pfn)); -} -#else -static inline unsigned int section_compound_order(const struct mem_section *section) -{ - return 0; -} - -static inline void section_set_compound_order(struct mem_section *section, - unsigned int order) -{ -} - -static inline void section_set_compound_order_range(unsigned long pfn, - unsigned long nr_pages, unsigned int order) -{ -} - -static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) -{ - return 0; -} -#endif - -static inline bool vmemmap_optimizable_pfn(unsigned long pfn) -{ - const unsigned int order = pfn_to_section_compound_order(pfn); - const unsigned long nr_pages = 1UL << order; - - if (!is_power_of_2(sizeof(struct page))) - return false; - - return (pfn & (nr_pages - 1)) >= VMEMMAP_OPTIMIZATION_NR_STRUCT_PAGES; -} - -static inline bool vmemmap_optimizable_order(unsigned int order) -{ - if (!IS_ENABLED(CONFIG_VMEMMAP_OPTIMIZATION)) - return false; - - if (!is_power_of_2(sizeof(struct page))) - return false; - - return order >= VMEMMAP_OPTIMIZATION_MIN_ORDER; -} +#include /* * mm/sparse.c @@ -142,9 +69,6 @@ static inline void sparse_sections_init(void) {} * mm/sparse-vmemmap.c */ #ifdef CONFIG_SPARSEMEM_VMEMMAP -#ifdef CONFIG_VMEMMAP_OPTIMIZATION -struct page *vmemmap_shared_tail_page(unsigned int order, struct zone *zone); -#endif void sparse_init_subsection_map(void); int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, struct vmem_altmap *altmap, struct dev_pagemap *pgmap); From 75551e3a23c39e0d496a2341e8c47e735427bf6d Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Sun, 27 Sep 2026 10:54:38 +0800 Subject: [PATCH 0587/1012] powerpc/mm: switch device DAX to shared tail vmemmap pages The powerpc radix compound vmemmap population path still finds a reusable tail page by walking the vmemmap page tables. Switch it to the common vmemmap_shared_tail_page() helper instead, so it can use the shared vmemmap page directly to simplify the code. This removes the powerpc-specific tail-page lookup and its fallback path and aligns the device DAX vmemmap optimization path with HugeTLB. Link: https://lore.kernel.org/20260927025441.741633-10-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Qi Zheng Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- arch/powerpc/mm/book3s64/radix_pgtable.c | 80 +++--------------------- include/linux/vmemmap-optimization.h | 6 ++ mm/sparse-vmemmap.c | 6 -- 3 files changed, 15 insertions(+), 77 deletions(-) diff --git a/arch/powerpc/mm/book3s64/radix_pgtable.c b/arch/powerpc/mm/book3s64/radix_pgtable.c index cf692b2b5f7bc5..ee068f24a79f07 100644 --- a/arch/powerpc/mm/book3s64/radix_pgtable.c +++ b/arch/powerpc/mm/book3s64/radix_pgtable.c @@ -19,6 +19,7 @@ #include #include #include +#include #include #include @@ -1250,59 +1251,6 @@ static pte_t * __meminit radix__vmemmap_populate_address(unsigned long addr, int return pte; } -static pte_t * __meminit vmemmap_compound_tail_page(unsigned long addr, - unsigned long pfn_offset, int node) -{ - pgd_t *pgd; - p4d_t *p4d; - pud_t *pud; - pmd_t *pmd; - pte_t *pte; - unsigned long map_addr; - - /* the second vmemmap page which we use for duplication */ - map_addr = addr - pfn_offset * sizeof(struct page) + PAGE_SIZE; - pgd = pgd_offset_k(map_addr); - p4d = p4d_offset(pgd, map_addr); - pud = vmemmap_pud_alloc(p4d, node, map_addr); - if (!pud) - return NULL; - pmd = vmemmap_pmd_alloc(pud, node, map_addr); - if (!pmd) - return NULL; - if (pmd_leaf(*pmd)) - /* - * The second page is mapped as a hugepage due to a nearby request. - * Force our mapping to page size without deduplication - */ - return NULL; - pte = vmemmap_pte_alloc(pmd, node, map_addr); - if (!pte) - return NULL; - /* - * Check if there exist a mapping to the left - */ - if (pte_none(*pte)) { - /* - * Populate the head page vmemmap page. - * It can fall in different pmd, hence - * vmemmap_populate_address() - */ - pte = radix__vmemmap_populate_address(map_addr - PAGE_SIZE, node, NULL, NULL); - if (!pte) - return NULL; - /* - * Populate the tail pages vmemmap page - */ - pte = radix__vmemmap_pte_populate(pmd, map_addr, node, NULL, NULL); - if (!pte) - return NULL; - vmemmap_verify(pte, node, map_addr, map_addr + PAGE_SIZE); - return pte; - } - return pte; -} - int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, unsigned long start, unsigned long end, int node, @@ -1320,6 +1268,12 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, pud_t *pud; pmd_t *pmd; pte_t *pte; + struct page *tail_page; + unsigned int order = pfn_to_section_compound_order(start_pfn); + + tail_page = vmemmap_shared_tail_page(order, device_zone(node)); + if (!tail_page) + return -ENOMEM; for (addr = start; addr < end; addr = next) { @@ -1349,10 +1303,9 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, next = addr + PAGE_SIZE; continue; } else { - unsigned long nr_pages = pgmap_vmemmap_nr(pgmap); + unsigned long nr_pages = 1UL << order; unsigned long addr_pfn = page_to_pfn((struct page *)addr); unsigned long pfn_offset = addr_pfn - ALIGN_DOWN(addr_pfn, nr_pages); - pte_t *tail_page_pte; /* * if the address is aligned to huge page size it is the @@ -1377,23 +1330,8 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, next = addr + 2 * PAGE_SIZE; continue; } - /* - * get the 2nd mapping details - * Also create it if that doesn't exist - */ - tail_page_pte = vmemmap_compound_tail_page(addr, pfn_offset, node); - if (!tail_page_pte) { - - pte = radix__vmemmap_pte_populate(pmd, addr, node, NULL, NULL); - if (!pte) - return -ENOMEM; - vmemmap_verify(pte, node, addr, addr + PAGE_SIZE); - - next = addr + PAGE_SIZE; - continue; - } - pte = radix__vmemmap_pte_populate(pmd, addr, node, NULL, pte_page(*tail_page_pte)); + pte = radix__vmemmap_pte_populate(pmd, addr, node, NULL, tail_page); if (!pte) return -ENOMEM; vmemmap_verify(pte, node, addr, addr + PAGE_SIZE); diff --git a/include/linux/vmemmap-optimization.h b/include/linux/vmemmap-optimization.h index bd0974b262a4d1..fa9e9abd66560f 100644 --- a/include/linux/vmemmap-optimization.h +++ b/include/linux/vmemmap-optimization.h @@ -83,6 +83,12 @@ static inline unsigned int pfn_to_section_compound_order(unsigned long pfn) { return 0; } + +static inline struct page *vmemmap_shared_tail_page(unsigned int order, + struct zone *zone) +{ + return NULL; +} #endif /* CONFIG_VMEMMAP_OPTIMIZATION */ static inline bool vmemmap_optimizable_pfn(unsigned long pfn) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 1874cdad2d05f3..a501b5f4311b7a 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -236,12 +236,6 @@ struct page __ref *vmemmap_shared_tail_page(unsigned int order, struct zone *zon return page; } -#else -static inline struct page *vmemmap_shared_tail_page(unsigned int order, - struct zone *zone) -{ - return NULL; -} #endif static __meminit void *vmemmap_alloc_pte(unsigned long pfn, int node, From c6da9e9e24d66723dec318ca5c2eca64f249ee26 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Sun, 27 Sep 2026 10:54:39 +0800 Subject: [PATCH 0588/1012] mm/sparse-vmemmap: drop the extra tail page from device DAX reservation The device DAX vmemmap population still reserves one extra tail vmemmap page after the head page. Drop that extra reservation and let the shared tail page cover all tail vmemmap pages after the head page, so DAX follows the same reservation model as HugeTLB. This reduces the reserved vmemmap pages for optimized DAX mappings to one and removes the now-unneeded first-tail population from the generic and powerpc paths to simplify the code as well. Link: https://lore.kernel.org/20260927025441.741633-11-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- arch/powerpc/mm/book3s64/radix_pgtable.c | 46 ++---------------------- include/linux/mm.h | 4 +-- mm/mm_init.c | 2 +- mm/sparse-vmemmap.c | 13 ++----- 4 files changed, 8 insertions(+), 57 deletions(-) diff --git a/arch/powerpc/mm/book3s64/radix_pgtable.c b/arch/powerpc/mm/book3s64/radix_pgtable.c index ee068f24a79f07..9ca28e4a610a26 100644 --- a/arch/powerpc/mm/book3s64/radix_pgtable.c +++ b/arch/powerpc/mm/book3s64/radix_pgtable.c @@ -1218,39 +1218,6 @@ int __meminit radix__vmemmap_populate(unsigned long start, unsigned long end, in return 0; } -static pte_t * __meminit radix__vmemmap_populate_address(unsigned long addr, int node, - struct vmem_altmap *altmap, - struct page *reuse) -{ - pgd_t *pgd; - p4d_t *p4d; - pud_t *pud; - pmd_t *pmd; - pte_t *pte; - - pgd = pgd_offset_k(addr); - p4d = p4d_offset(pgd, addr); - pud = vmemmap_pud_alloc(p4d, node, addr); - if (!pud) - return NULL; - pmd = vmemmap_pmd_alloc(pud, node, addr); - if (!pmd) - return NULL; - if (pmd_leaf(*pmd)) - /* - * The second page is mapped as a hugepage due to a nearby request. - * Force our mapping to page size without deduplication - */ - return NULL; - pte = vmemmap_pte_alloc(pmd, node, addr); - if (!pte) - return NULL; - radix__vmemmap_pte_populate(pmd, addr, node, NULL, NULL); - vmemmap_verify(pte, node, addr, addr + PAGE_SIZE); - - return pte; -} - int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, unsigned long start, unsigned long end, int node, @@ -1297,7 +1264,7 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, if (!pte_none(*pte)) { /* * This could be because we already have a compound - * page whose VMEMMAP_RESERVE_NR pages were mapped and + * page whose retained vmemmap page was mapped and * this request fall in those pages. */ next = addr + PAGE_SIZE; @@ -1318,16 +1285,7 @@ int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, return -ENOMEM; vmemmap_verify(pte, node, addr, addr + PAGE_SIZE); - /* - * Populate the tail pages vmemmap page - * It can fall in different pmd, hence - * vmemmap_populate_address() - */ - pte = radix__vmemmap_populate_address(addr + PAGE_SIZE, node, NULL, NULL); - if (!pte) - return -ENOMEM; - - next = addr + 2 * PAGE_SIZE; + next = addr + PAGE_SIZE; continue; } diff --git a/include/linux/mm.h b/include/linux/mm.h index 070ce27e9cd3cd..30a3365bca8271 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -38,6 +38,7 @@ #include #include #include +#include struct mempolicy; struct anon_vma; @@ -5167,7 +5168,6 @@ static inline void vmem_altmap_free(struct vmem_altmap *altmap, } #endif -#define VMEMMAP_RESERVE_NR 2 #ifdef CONFIG_ARCH_WANT_OPTIMIZE_DAX_VMEMMAP static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap, struct dev_pagemap *pgmap) @@ -5187,7 +5187,7 @@ static inline bool __vmemmap_can_optimize(struct vmem_altmap *altmap, * For vmemmap optimization with DAX we need minimum 2 vmemmap * pages. See layout diagram in Documentation/mm/vmemmap_dedup.rst */ - return !altmap && (nr_vmemmap_pages > VMEMMAP_RESERVE_NR); + return !altmap && (nr_vmemmap_pages > VMEMMAP_OPTIMIZATION_PAGES); } /* * If we don't have an architecture override, use the generic rule diff --git a/mm/mm_init.c b/mm/mm_init.c index efffa8609b8592..56bb4567a49405 100644 --- a/mm/mm_init.c +++ b/mm/mm_init.c @@ -1056,7 +1056,7 @@ static inline unsigned long compound_nr_pages(unsigned long pfn, if (!section_vmemmap_optimizable(ms)) return pgmap_vmemmap_nr(pgmap); - return VMEMMAP_RESERVE_NR * (PAGE_SIZE / sizeof(struct page)); + return VMEMMAP_OPTIMIZATION_PAGES * (PAGE_SIZE / sizeof(struct page)); } static void __ref memmap_init_compound(struct page *head, diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index a501b5f4311b7a..2930dc2cf86211 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -136,7 +136,6 @@ int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages { const struct mem_section *ms = __pfn_to_section(pfn); const int order = section_compound_order(ms); - const int vmemmap_pages = pgmap ? VMEMMAP_RESERVE_NR : VMEMMAP_OPTIMIZATION_PAGES; const unsigned long pages_per_compound = 1UL << order; VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SUBSECTION)); @@ -147,13 +146,13 @@ int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages if (order < PFN_SECTION_SHIFT) { VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, pages_per_compound)); - return vmemmap_pages * nr_pages / pages_per_compound; + return VMEMMAP_OPTIMIZATION_PAGES * nr_pages / pages_per_compound; } VM_WARN_ON_ONCE(!IS_ALIGNED(pfn | nr_pages, PAGES_PER_SECTION)); if (IS_ALIGNED(pfn, pages_per_compound)) - return vmemmap_pages; + return VMEMMAP_OPTIMIZATION_PAGES; return 0; } @@ -554,17 +553,11 @@ static int __meminit vmemmap_populate_compound_pages(unsigned long start_pfn, if (!pte) return -ENOMEM; - /* Populate the tail pages vmemmap page */ - next = addr + PAGE_SIZE; - pte = vmemmap_populate_address(next, node, NULL, -1, flags); - if (!pte) - return -ENOMEM; - /* * Reuse the shared page for the rest of tail pages * See layout diagram in Documentation/mm/vmemmap_dedup.rst */ - next += PAGE_SIZE; + next = addr + PAGE_SIZE; rc = vmemmap_populate_range(next, last, node, NULL, page_to_pfn(page), flags); if (rc) From c79b575b9233704e0986a0ae294a2e74fd4f986d Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Sun, 27 Sep 2026 10:54:40 +0800 Subject: [PATCH 0589/1012] mm/sparse-vmemmap: drop unused section_nr_vmemmap_pages() arguments section_nr_vmemmap_pages() no longer uses the altmap or pgmap arguments, so drop them from the helper and its callers. Link: https://lore.kernel.org/20260927025441.741633-12-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- mm/sparse-vmemmap.c | 10 ++++------ mm/sparse.c | 3 +-- mm/sparse.h | 6 ++---- 3 files changed, 7 insertions(+), 12 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 2930dc2cf86211..66de04f8863bfb 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -131,8 +131,7 @@ void __meminit vmemmap_verify(pte_t *pte, int node, start, end - 1); } -int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, - struct vmem_altmap *altmap, struct dev_pagemap *pgmap) +int __meminit section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages) { const struct mem_section *ms = __pfn_to_section(pfn); const int order = section_compound_order(ms); @@ -670,7 +669,7 @@ static struct page * __meminit populate_section_memmap(unsigned long pfn, struct page *page = __populate_section_memmap(pfn, nr_pages, nid, altmap, pgmap); - memmap_pages_add(section_nr_vmemmap_pages(pfn, nr_pages, altmap, pgmap)); + memmap_pages_add(section_nr_vmemmap_pages(pfn, nr_pages)); return page; } @@ -681,7 +680,7 @@ static void depopulate_section_memmap(unsigned long pfn, unsigned long nr_pages, unsigned long start = (unsigned long) pfn_to_page(pfn); unsigned long end = start + nr_pages * sizeof(struct page); - memmap_pages_add(-section_nr_vmemmap_pages(pfn, nr_pages, altmap, pgmap)); + memmap_pages_add(-section_nr_vmemmap_pages(pfn, nr_pages)); vmemmap_free(start, end, altmap); } @@ -691,8 +690,7 @@ static void free_map_bootmem(struct page *memmap) unsigned long end = (unsigned long)(memmap + PAGES_PER_SECTION); unsigned long pfn = page_to_pfn(memmap); - memmap_boot_pages_add(-section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, - NULL, NULL)); + memmap_boot_pages_add(-section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION)); vmemmap_free(start, end, NULL); } diff --git a/mm/sparse.c b/mm/sparse.c index cc28bb41fdb1f1..b75921c622edef 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -250,8 +250,7 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, nid, NULL, NULL); if (!map) panic("Failed to allocate memmap for section %lu\n", pnum); - memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION, - NULL, NULL)); + memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION)); sparse_init_one_section(__nr_to_section(pnum), pnum, map, usage, SECTION_IS_EARLY); usage = (void *)usage + mem_section_usage_size(); diff --git a/mm/sparse.h b/mm/sparse.h index a5111087ee3a3d..530692cdd516fe 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -70,12 +70,10 @@ static inline void sparse_sections_init(void) {} */ #ifdef CONFIG_SPARSEMEM_VMEMMAP void sparse_init_subsection_map(void); -int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, - struct vmem_altmap *altmap, struct dev_pagemap *pgmap); +int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages); #else static inline void sparse_init_subsection_map(void) {} -static inline int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages, - struct vmem_altmap *altmap, struct dev_pagemap *pgmap) +static inline int section_nr_vmemmap_pages(unsigned long pfn, unsigned long nr_pages) { return DIV_ROUND_UP(nr_pages * sizeof(struct page), PAGE_SIZE); } From 66ff9bfda186d5c0044a200ab1fada284c28dcb0 Mon Sep 17 00:00:00 2001 From: Muchun Song Date: Sun, 27 Sep 2026 10:54:41 +0800 Subject: [PATCH 0590/1012] Documentation/mm: update DAX vmemmap deduplication docs Device DAX now uses the common per-zone shared tail page for vmemmap deduplication. The old documentation still described a DAX-specific layout with a separately populated tail vmemmap page and half the HugeTLB savings. Update the generic and powerpc documentation to describe the shared layout. In the powerpc document, keep the radix and 64K-specific details, drop the duplicated 4K PUD arithmetic, and replace the repeated device-dax diagrams with a single parameterized PMD/PUD diagram. Link: https://lore.kernel.org/20260927025441.741633-13-songmuchun@bytedance.com Signed-off-by: Muchun Song Signed-off-by: Andrew Morton Acked-by: Qi Zheng Acked-by: David Hildenbrand (Arm) Cc: Oscar Salvador Cc: Madhavan Srinivasan Cc: Michael Ellerman Cc: Jonathan Corbet Cc: Lorenzo Stoakes Cc: Mike Rapoport Cc: Nicholas Piggin Cc: Christophe Leroy Cc: Randy Dunlap Cc: Lance Yang --- Documentation/arch/powerpc/vmemmap_dedup.rst | 90 ++++---------------- Documentation/mm/vmemmap_dedup.rst | 32 +------ 2 files changed, 21 insertions(+), 101 deletions(-) diff --git a/Documentation/arch/powerpc/vmemmap_dedup.rst b/Documentation/arch/powerpc/vmemmap_dedup.rst index dc4db59fdf87b4..8286acbca9bc57 100644 --- a/Documentation/arch/powerpc/vmemmap_dedup.rst +++ b/Documentation/arch/powerpc/vmemmap_dedup.rst @@ -19,82 +19,28 @@ With 1G PUD level mapping, we require 16384 struct pages and a single 64K vmemmap page can contain 1024 struct pages (64K/sizeof(struct page)). Hence we require 16 64K pages in vmemmap to map the struct page for 1G PUD level mapping. -Here's how things look like on device-dax after the sections are populated:: - +-----------+ ---virt_to_page---> +-----------+ mapping to +-----------+ - | | | 0 | -------------> | 0 | - | | +-----------+ +-----------+ - | | | 1 | -------------> | 1 | - | | +-----------+ +-----------+ - | | | 2 | ----------------^ ^ ^ ^ ^ ^ - | | +-----------+ | | | | | - | | | 3 | ------------------+ | | | | - | | +-----------+ | | | | - | | | 4 | --------------------+ | | | - | PUD | +-----------+ | | | - | level | | . | ----------------------+ | | - | mapping | +-----------+ | | - | | | . | ------------------------+ | - | | +-----------+ | - | | | 15 | --------------------------+ - | | +-----------+ - | | - | | - | | - +-----------+ - - With 4K page size, 2M PMD level mapping requires 512 struct pages and a single 4K vmemmap page contains 64 struct pages(4K/sizeof(struct page)). Hence we require 8 4K pages in vmemmap to map the struct page for 2M pmd level mapping. -Here's how things look like on device-dax after the sections are populated:: - - +-----------+ ---virt_to_page---> +-----------+ mapping to +-----------+ - | | | 0 | -------------> | 0 | - | | +-----------+ +-----------+ - | | | 1 | -------------> | 1 | - | | +-----------+ +-----------+ - | | | 2 | ----------------^ ^ ^ ^ ^ ^ - | | +-----------+ | | | | | - | | | 3 | ------------------+ | | | | - | | +-----------+ | | | | - | | | 4 | --------------------+ | | | - | PMD | +-----------+ | | | - | level | | 5 | ----------------------+ | | - | mapping | +-----------+ | | - | | | 6 | ------------------------+ | - | | +-----------+ | - | | | 7 | --------------------------+ - | | +-----------+ - | | - | | - | | - +-----------+ - -With 1G PUD level mapping, we require 262144 struct pages and a single 4K -vmemmap page can contain 64 struct pages (4K/sizeof(struct page)). Hence we -require 4096 4K pages in vmemmap to map the struct pages for 1G PUD level -mapping. - -Here's how things look like on device-dax after the sections are populated:: - - +-----------+ ---virt_to_page---> +-----------+ mapping to +-----------+ - | | | 0 | -------------> | 0 | - | | +-----------+ +-----------+ - | | | 1 | -------------> | 1 | - | | +-----------+ +-----------+ - | | | 2 | ----------------^ ^ ^ ^ ^ ^ - | | +-----------+ | | | | | - | | | 3 | ------------------+ | | | | - | | +-----------+ | | | | - | | | 4 | --------------------+ | | | - | PUD | +-----------+ | | | - | level | | . | ----------------------+ | | - | mapping | +-----------+ | | - | | | . | ------------------------+ | - | | +-----------+ | - | | | 4095 | --------------------------+ - | | +-----------+ +Here's how things look on device-dax after vmemmap-optimized sections are +populated. ``N`` is the number of vmemmap pages required by the DAX mapping +above:: + + Device DAX vmemmap pages (N pages) backing page frames + +-----------+ ---virt_to_page---> +-----------+ mapping to +-------------+ + | | | 0 | -------------> | 0 | + | | +-----------+ +-------------+ + | | | 1 | ------+ + | | +-----------+ | + | | | 2 | ------+ + | | +-----------+ | + | | | . | ------+ +-------------+ + | PMD/PUD | +-----------+ | | A single, | + | level | | . | ------+------> | per-zone | + | mapping | +-----------+ | | shared tail | + | | | N - 1 | ------+ | page | + | | +-----------+ +-------------+ | | | | | | diff --git a/Documentation/mm/vmemmap_dedup.rst b/Documentation/mm/vmemmap_dedup.rst index 9fa8642ded483b..8c287ae3f86cc8 100644 --- a/Documentation/mm/vmemmap_dedup.rst +++ b/Documentation/mm/vmemmap_dedup.rst @@ -1,4 +1,3 @@ - .. SPDX-License-Identifier: GPL-2.0 ========================================= @@ -192,32 +191,7 @@ to 4 on HugeTLB pages. There's no remapping of vmemmap given that device-dax memory is not part of System RAM ranges initialized at boot. Thus the tail page deduplication -happens at a later stage when we populate the sections. HugeTLB reuses the -the head vmemmap page representing, whereas device-dax reuses the tail -vmemmap page. This results in only half of the savings compared to HugeTLB. - -Deduplicated tail pages are not mapped read-only. +happens at a later stage when we populate the sections. -Here's how things look like on device-dax after the sections are populated:: - - +-----------+ ---virt_to_page---> +-----------+ mapping to +-----------+ - | | | 0 | -------------> | 0 | - | | +-----------+ +-----------+ - | | | 1 | -------------> | 1 | - | | +-----------+ +-----------+ - | | | 2 | ----------------^ ^ ^ ^ ^ ^ - | | +-----------+ | | | | | - | | | 3 | ------------------+ | | | | - | | +-----------+ | | | | - | | | 4 | --------------------+ | | | - | PMD | +-----------+ | | | - | level | | 5 | ----------------------+ | | - | mapping | +-----------+ | | - | | | 6 | ------------------------+ | - | | +-----------+ | - | | | 7 | --------------------------+ - | | +-----------+ - | | - | | - | | - +-----------+ +Deduplicated tail pages are not mapped read-only. The mapping layout is the same +as HugeTLB. From 67b2a6ff31e2119f3a2013f035052b803671462c Mon Sep 17 00:00:00 2001 From: Yury Norov Date: Fri, 11 Sep 2026 11:52:43 -0400 Subject: [PATCH 0591/1012] lib: fix lock initialization in region allocation benchmark The Maple Tree benchmark uses MTREE_INIT() for a stack-allocated tree. Its static spinlock initializer leaves lockdep to use the lock address as the class key. Since the address is on the stack, the first allocation triggers "INFO: trying to register non-static key" and disables lockdep. The IDA benchmark has the same problem through IDA_INIT(), but runs after Maple Tree and therefore encounters an already disabled lockdep. Use mt_init_flags() and ida_init() to initialize the locks with persistent lock-class keys. Keep initialization outside the timed allocation paths. Link: https://lore.kernel.org/20260911155244.1406122-1-ynorov@nvidia.com Fixes: f4806cc63cc6 ("lib: test bitmap vs IDA vs Maple Tree performance for region allocations") Signed-off-by: Yury Norov Signed-off-by: Andrew Morton Closes: https://lore.kernel.org/oe-lkp/202609101106.771b567e-lkp@intel.com Acked-by: Liam R. Howlett (Oracle) Cc: Alice Ryhl Cc: Andrew Ballance Cc: Matthew Wilcox (Oracle) Cc: Rasmus Villemoes --- lib/region_alloc_benchmark.c | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/lib/region_alloc_benchmark.c b/lib/region_alloc_benchmark.c index e88b4cf55c629c..a644f3d5431a0c 100644 --- a/lib/region_alloc_benchmark.c +++ b/lib/region_alloc_benchmark.c @@ -78,11 +78,13 @@ static size_t __init ida_size(unsigned long nr_ids) static unsigned long __init benchmark_ida(unsigned long cap) { - struct ida ida = IDA_INIT(ida); + struct ida ida; unsigned long cnt, idx, off, nr_ids = 0; ktime_t alloc_time, free_time; int id = -ENOSPC; + ida_init(&ida); + alloc_time = ktime_get(); for (cnt = 0; cnt <= cap; cnt++) { for (off = 0; off < reg_sz[cnt]; off++) { @@ -125,12 +127,14 @@ static unsigned long __init benchmark_ida(unsigned long cap) static unsigned long __init benchmark_maple_tree(unsigned long cap) { - struct maple_tree mt = MTREE_INIT(mt, MT_FLAGS_ALLOC_RANGE); + struct maple_tree mt; unsigned long cnt, idx; ktime_t alloc_time, free_time; size_t sz; int ret; + mt_init_flags(&mt, MT_FLAGS_ALLOC_RANGE); + alloc_time = ktime_get(); for (cnt = 0; cnt <= cap; cnt++) { ret = mtree_alloc_range(&mt, &idx, xa_mk_value(cnt + 1), From 0924a82fadb90a8bae7d287fc6fc392198aefd7c Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Fri, 11 Sep 2026 15:29:03 +0100 Subject: [PATCH 0592/1012] kselftest: mm: fix potential failure for merged VMA in guard-regions check_vmflag_guard() uses /proc/self/smaps to retrieve the VMA flags, but this can fail if the mapping is merged with an adjacent VMA. To avoid this potential failure, first allocate a temporary region with extra pages at both ends, unmap it, and then map the test region within the temporary address range, leaving an unmapped page on each side to prevent VMA merging. Link: https://lore.kernel.org/20260911142904.1825452-1-yeoreum.yun@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Cc: Liam R. Howlett Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/mm/guard-regions.c | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/tools/testing/selftests/mm/guard-regions.c b/tools/testing/selftests/mm/guard-regions.c index 5c8ec3ca75d7d2..b724d62d2b7555 100644 --- a/tools/testing/selftests/mm/guard-regions.c +++ b/tools/testing/selftests/mm/guard-regions.c @@ -2257,8 +2257,18 @@ TEST_F(guard_regions, smaps) char *ptr, *ptr2; int i; - /* Map a region. */ - ptr = mmap_(self, variant, NULL, 10 * page_size, PROT_READ | PROT_WRITE, 0, 0); + /* Map then unmap placeholder to avoid adjacent merges */ + ptr = mmap_(self, variant, NULL, 12 * page_size, PROT_NONE, 0, 0); + ASSERT_NE(ptr, MAP_FAILED); + ASSERT_EQ(munmap(ptr, 12 * page_size), 0); + + /* + * Map a region for the test. Since the preceding temporary mapping + * succeeded, this mapping should also succeed without merging with + * adjacent VMAs. + */ + ptr = mmap_(self, variant, ptr + page_size, 10 * page_size, + PROT_READ | PROT_WRITE, MAP_FIXED, 0); ASSERT_NE(ptr, MAP_FAILED); /* We shouldn't yet see a guard flag. */ From cdace658d18b4d85e924ec58c34ae80d0623792b Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:03 -0700 Subject: [PATCH 0593/1012] mm/damon/api: introduce DAMOS_FILTER_TYPE_PROBE_HITS_WSUM Patch series "mm/damon: introduce probe_hits_wsum DAMOS core filter", v2. DAMON can do flexible data attributes monitoring. The classical data access monitoring can also be done using the attributes monitoring. The access event is just one of the data attributes that DAMON supports. Users do monitoring to make some actions based on it. DAMOS is a feature for automating that. However, DAMOS cannot utilize the data attributes monitoring results. It is still Data "Access" Monitoring-based Operation Schemes. It requires users to set the target "access" pattern. DAMOS core filter is effectively the same as the target access pattern. It is just a more generalized and flexible way of describing the operation action target region. Introduce a new DAMOS core filter type, probe_hits_wsum. It specifies the filter target based on a range of the probe hits weighted sum. Using this, users can apply DAMOS actions to regions of specific data attributes pattern. Note that the classic target access pattern still works. Hence the target nr_accesses range should still be properly configured. The new filter would be used in only data attributes-only mode. In the mode, classic access monitoring is just turned off, and therefore nr_accesses of regions are always zero. Users could simply set the target nr_accesses range to include the zero nr_Accesses regions. Patches Sequence ================ Patch 1 updates the DAMON kernel API for the new filter type. Patch 2 extends damon_probe_hits_wsum() to do the calculation based on moving sum. Patch 3 implements the filter type in the core layer. Patch 4 refactors DAMON sysfs interface internal data structure for efficient reuse of data structure for the probe hits weighted sum range user inputs. Patch 5 updates DAMON sysfs interface to support the new filter type. Patches 6 and 7 update design and usage documents for the new filter type, respectively. This patch (of 7): Update DAMON kernel API to introduce new DAMOS core filter type, PROBE_HITS_WSUM. It will allow API callers to describe the DAMOS action target regions based on their probe_hits weighted sum. For describing the filtering target weighted sum range, add two type-dependent union fields to damos_filter. Link: https://lore.kernel.org/20260911135510.96914-1-sj@kernel.org Link: https://lore.kernel.org/20260911135510.96914-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 20 ++++++++++++++------ 1 file changed, 14 insertions(+), 6 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 16800b3dc379b9..9e84bdeb5616df 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -399,14 +399,15 @@ struct damos_stat { * @DAMOS_FILTER_TYPE_UNMAPPED: Unmapped pages. * @DAMOS_FILTER_TYPE_ADDR: Address range. * @DAMOS_FILTER_TYPE_TARGET: Data Access Monitoring target. + * @DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: probe_hits weighted sum range. * @NR_DAMOS_FILTER_TYPES: Number of filter types. * - * All types except &DAMOS_FILTER_TYPE_ADDR and &DAMOS_FILTER_TYPE_TARGET - * are handled by the underlying &struct damon_operations as a part of scheme - * action trying, and therefore accounted as 'tried'. In contrast, - * &DAMOS_FILTER_TYPE_ADDR and &DAMOS_FILTER_TYPE_TARGET filters are handled - * by the core layer before trying of the action, and therefore not accounted - * as 'tried'. + * All types except &DAMOS_FILTER_TYPE_ADDR, &DAMOS_FILTER_TYPE_TARGET and + * &DAMOS_FILTER_TYPE_PROBE_HITS_WSUM are handled by the underlying &struct + * damon_operations as a part of scheme action trying, and therefore accounted + * as 'tried'. In contrast, &DAMOS_FILTER_TYPE_ADDR and + * &DAMOS_FILTER_TYPE_TARGET filters are handled by the core layer before + * trying of the action, and therefore not accounted as 'tried'. * * Support for the operations-handled filters depends on the running * &struct damon_operations. @@ -420,6 +421,7 @@ enum damos_filter_type { DAMOS_FILTER_TYPE_UNMAPPED, DAMOS_FILTER_TYPE_ADDR, DAMOS_FILTER_TYPE_TARGET, + DAMOS_FILTER_TYPE_PROBE_HITS_WSUM, NR_DAMOS_FILTER_TYPES, }; @@ -434,6 +436,8 @@ enum damos_filter_type { * &damon_ctx->adaptive_targets if @type is * DAMOS_FILTER_TYPE_TARGET. * @sz_range: Size range if @type is DAMOS_FILTER_TYPE_HUGEPAGE_SIZE. + * @range_min: Minimum value of range arguments. + * @range_max: Maximum value of range arguments. * * Before applying the &damos->action to a memory region, DAMOS checks if each * byte of the region matches to this given condition and avoid applying the @@ -450,6 +454,10 @@ struct damos_filter { struct damon_addr_range addr_range; int target_idx; struct damon_size_range sz_range; + struct { + unsigned long range_min; + unsigned long range_max; + }; }; /* private: */ /* List head for siblings. */ From 1b3c0b294abd11b0599eb9bd39d64de04e63b22a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 07:20:01 -0700 Subject: [PATCH 0594/1012] mm/damon/api: clarify DAMOS_FILTER_TYPE_PROBE_HITS_WSUM behavior clarify DAMOS_FILTER_TYPE_PROBE_HITS_WSUM behavior Link: https://lore.kernel.org/20260911142522.98013-1-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 9e84bdeb5616df..e27ed156e7657e 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -405,9 +405,9 @@ struct damos_stat { * All types except &DAMOS_FILTER_TYPE_ADDR, &DAMOS_FILTER_TYPE_TARGET and * &DAMOS_FILTER_TYPE_PROBE_HITS_WSUM are handled by the underlying &struct * damon_operations as a part of scheme action trying, and therefore accounted - * as 'tried'. In contrast, &DAMOS_FILTER_TYPE_ADDR and - * &DAMOS_FILTER_TYPE_TARGET filters are handled by the core layer before - * trying of the action, and therefore not accounted as 'tried'. + * as 'tried'. In contrast, &DAMOS_FILTER_TYPE_ADDR, &DAMOS_FILTER_TYPE_TARGET + * and &DAMOS_FILTER_TYPE_PROBE_HITS_WSUM filters are handled by the core layer + * before trying of the action, and therefore not accounted as 'tried'. * * Support for the operations-handled filters depends on the running * &struct damon_operations. From 8fc1b331861c787c16e2306201c5796a48006392 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:04 -0700 Subject: [PATCH 0595/1012] mm/damon/core: extend probe_hits_wsum() for moving sum based calculation damon_probe_hits_wsum() is being called only in aggregation time. In future, it could also be used by DAMOS. In this case, since DAMOS uses its own apply_interval, it could be called in sampling time. Then using the not yet fully aggregated probe_hits could result in suboptimum outcomes. Extend damon_probe_hits_wsum() to get the weighted sum based on moving sum to prepare the DAMOS usage. Link: https://lore.kernel.org/20260911135510.96914-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 2 +- mm/damon/core.c | 8 ++++++-- mm/damon/paddr.c | 2 +- mm/damon/vaddr.c | 2 +- 4 files changed, 9 insertions(+), 5 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index e27ed156e7657e..4be7d1df8e71fa 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -1075,7 +1075,7 @@ unsigned int damon_nr_accesses_mvsum(struct damon_region *r, struct damon_ctx *ctx); unsigned char damon_probe_hits_mvsum(int probe_idx, struct damon_region *r, struct damon_ctx *ctx); -unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, +unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, bool mv, struct damon_ctx *ctx); int damon_set_regions(struct damon_target *t, struct damon_addr_range *ranges, diff --git a/mm/damon/core.c b/mm/damon/core.c index ea6df4311ceb7f..2b361ab2407887 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -469,11 +469,12 @@ static bool damon_is_last_region(struct damon_region *r, * damon_probe_hits_wsum() - Returns probe hits weighted sum of a region. * @r: region to get the weighted sum of. * @last: if the request is for last-window aggregated probe hits. + * @mv: use moving sum. * @ctx: context of &r. * * Return: the weighted sum of probe hits of the region. */ -unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, +unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, bool mv, struct damon_ctx *ctx) { struct damon_probe *probe; @@ -483,6 +484,9 @@ unsigned int damon_probe_hits_wsum(struct damon_region *r, bool last, damon_for_each_probe(probe, ctx) { if (last) sum += r->last_probe_hits[i++] * probe->weight; + else if (mv) + sum += damon_probe_hits_mvsum(i++, r, ctx) * + probe->weight; else sum += r->probe_hits[i++] * probe->weight; } @@ -3472,7 +3476,7 @@ static unsigned int damon_merge_score(struct damon_region *r, bool last, struct damon_ctx *ctx, bool use_probe_hits) { if (use_probe_hits) - return damon_probe_hits_wsum(r, last, ctx); + return damon_probe_hits_wsum(r, last, false, ctx); if (last) return r->last_nr_accesses; return r->nr_accesses; diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 79195026b903a6..5abfabaa339e0e 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -208,7 +208,7 @@ static unsigned int damon_pa_apply_probes(struct damon_ctx *ctx, folio_put(folio); if (return_max_wsum) max_wsum = max(damon_probe_hits_wsum(r, false, - ctx), max_wsum); + false, ctx), max_wsum); } } return max_wsum; diff --git a/mm/damon/vaddr.c b/mm/damon/vaddr.c index 7063356370c34b..d5dde97b3cd0d0 100644 --- a/mm/damon/vaddr.c +++ b/mm/damon/vaddr.c @@ -711,7 +711,7 @@ static unsigned int damon_va_apply_probes(struct damon_ctx *ctx, __damon_va_apply_probes(ctx, mm, r); if (return_max_wsum) max_wsum = max(damon_probe_hits_wsum(r, false, - ctx), max_wsum); + false, ctx), max_wsum); } if (mm) mmput(mm); From be79cfed0322ab517e526d60aae7c47f0033d2de Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:05 -0700 Subject: [PATCH 0596/1012] mm/damon/core: support probe_hits_wsum damos core filter Implement probe_hits_wsum DAMOS core filter support in the core layer. Make three small changes for the support. First, update damos_filter_for_ops() to treat probe_hits_wsum filter as core filter. Second, Update destination damos_filter->range_{min,max} for probe_hits_wsum type damos filter commits. Third, extend damos_filter_match() to handle probe_hits_wsum type filter. Calculate the weighted sum of the given region and compare it with the given filter's target weighted sum range. Link: https://lore.kernel.org/20260911135510.96914-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 2b361ab2407887..38383ced3dd6c0 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -658,6 +658,7 @@ bool damos_filter_for_ops(enum damos_filter_type type) switch (type) { case DAMOS_FILTER_TYPE_ADDR: case DAMOS_FILTER_TYPE_TARGET: + case DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: return false; default: break; @@ -1332,6 +1333,10 @@ static void damos_commit_filter_arg( case DAMOS_FILTER_TYPE_HUGEPAGE_SIZE: dst->sz_range = src->sz_range; break; + case DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: + dst->range_min = src->range_min; + dst->range_max = src->range_max; + break; default: break; } @@ -2510,7 +2515,7 @@ static bool damos_filter_match(struct damon_ctx *ctx, struct damon_target *t, bool matched = false; struct damon_target *ti; int target_idx = 0; - unsigned long start, end; + unsigned long start, end, wsum; switch (filter->type) { case DAMOS_FILTER_TYPE_TARGET: @@ -2545,6 +2550,11 @@ static bool damos_filter_match(struct damon_ctx *ctx, struct damon_target *t, damon_split_region_at(t, r, end - r->ar.start); matched = true; break; + case DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: + wsum = damon_probe_hits_wsum(r, false, true, ctx); + matched = filter->range_min <= wsum && + wsum <= filter->range_max; + break; default: return false; } From 9d0dd37fa7b930a70648e38803123591c2ba7f80 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:06 -0700 Subject: [PATCH 0597/1012] mm/damon/sysfs-schemes: rename sysfs_filter->sz_range to range_{min,max} DAMON sysfs interface provides 'min' and 'max' files under the DAMOS filter directory. The purpose is setting the general range arguments for hugepage_size like filters that require range arguments. So far, hugepage_size was the only filter using it. Hence sz_range field of damon_sysfs_scheme_filter struct was connected to the files. In future, we could add a new filter that can reuse the 'min' and 'max' files. And the filter might use a range of a type that is not size. For example, probe_hits weighted sum. In this case, simply reusing the sz_range field would make it a little confusing. Rename sz_range to range_{min,max} to avoid such confusion. Link: https://lore.kernel.org/20260911135510.96914-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs-schemes.c | 19 +++++++++++-------- 1 file changed, 11 insertions(+), 8 deletions(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index d9b81d7b5910ed..4d9147d2a269ea 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -534,7 +534,8 @@ struct damon_sysfs_scheme_filter { bool allow; char *memcg_path; struct damon_addr_range addr_range; - struct damon_size_range sz_range; + unsigned long range_min; + unsigned long range_max; int target_idx; }; @@ -588,6 +589,7 @@ damos_sysfs_filter_type_names[] = { .type = DAMOS_FILTER_TYPE_TARGET, .name = "target", }, + }; static ssize_t type_show(struct kobject *kobj, @@ -778,7 +780,7 @@ static ssize_t min_show(struct kobject *kobj, struct damon_sysfs_scheme_filter *filter = container_of(kobj, struct damon_sysfs_scheme_filter, kobj); - return sysfs_emit(buf, "%lu\n", filter->sz_range.min); + return sysfs_emit(buf, "%lu\n", filter->range_min); } static ssize_t min_store(struct kobject *kobj, @@ -786,7 +788,7 @@ static ssize_t min_store(struct kobject *kobj, { struct damon_sysfs_scheme_filter *filter = container_of(kobj, struct damon_sysfs_scheme_filter, kobj); - int err = kstrtoul(buf, 0, &filter->sz_range.min); + int err = kstrtoul(buf, 0, &filter->range_min); return err ? err : count; } @@ -797,7 +799,7 @@ static ssize_t max_show(struct kobject *kobj, struct damon_sysfs_scheme_filter *filter = container_of(kobj, struct damon_sysfs_scheme_filter, kobj); - return sysfs_emit(buf, "%lu\n", filter->sz_range.max); + return sysfs_emit(buf, "%lu\n", filter->range_max); } static ssize_t max_store(struct kobject *kobj, @@ -805,7 +807,7 @@ static ssize_t max_store(struct kobject *kobj, { struct damon_sysfs_scheme_filter *filter = container_of(kobj, struct damon_sysfs_scheme_filter, kobj); - int err = kstrtoul(buf, 0, &filter->sz_range.max); + int err = kstrtoul(buf, 0, &filter->range_max); return err ? err : count; } @@ -2835,12 +2837,13 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, } else if (filter->type == DAMOS_FILTER_TYPE_TARGET) { filter->target_idx = sysfs_filter->target_idx; } else if (filter->type == DAMOS_FILTER_TYPE_HUGEPAGE_SIZE) { - if (sysfs_filter->sz_range.min > - sysfs_filter->sz_range.max) { + if (sysfs_filter->range_min > + sysfs_filter->range_max) { damos_destroy_filter(filter); return -EINVAL; } - filter->sz_range = sysfs_filter->sz_range; + filter->sz_range.min = sysfs_filter->range_min; + filter->sz_range.max = sysfs_filter->range_max; } damos_add_filter(scheme, filter); From b54c6f8024eeac7ab4f13113e43e0afd9b44d6e3 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:07 -0700 Subject: [PATCH 0598/1012] mm/damon/sysfs-schemes: support probe_hits_wsum damos core filter Extend DAMON sysfs interface to support probe_hits_wsum input. Also update sysfs input based scheme build logic to setup the min/max probe hits weighted sum range as user provided. Link: https://lore.kernel.org/20260911135510.96914-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs-schemes.c | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index 4d9147d2a269ea..3de4d804e049f8 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -589,7 +589,10 @@ damos_sysfs_filter_type_names[] = { .type = DAMOS_FILTER_TYPE_TARGET, .name = "target", }, - + { + .type = DAMOS_FILTER_TYPE_PROBE_HITS_WSUM, + .name = "probe_hits_wsum", + }, }; static ssize_t type_show(struct kobject *kobj, @@ -2844,6 +2847,13 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, } filter->sz_range.min = sysfs_filter->range_min; filter->sz_range.max = sysfs_filter->range_max; + } else if (filter->type == DAMOS_FILTER_TYPE_PROBE_HITS_WSUM) { + filter->range_min = sysfs_filter->range_min; + filter->range_max = sysfs_filter->range_max; + if (filter->range_min > filter->range_max) { + damos_destroy_filter(filter); + return -EINVAL; + } } damos_add_filter(scheme, filter); From d270f78376e7114bcc73dc2193b329aff924db42 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:08 -0700 Subject: [PATCH 0599/1012] Docs/mm/damon/design: update for probe_hits_wsum DAMOS core filter Update DAMON design document for the newly added probe hits weighted sum based DAMOS core filter type. Link: https://lore.kernel.org/20260911135510.96914-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 3 +++ 1 file changed, 3 insertions(+) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 22b785cd11dfec..cb116a82ff20b4 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -823,6 +823,9 @@ Below ``type`` of filters are currently supported. - Applied to pages that belonging to a given address range. - target - Applied to pages that belonging to a given DAMON monitoring target. + - probe_hits_wsum + - Matches to monitoring regions having a given range of :ref:`probe + hits weighted sum ` value. - Operations layer handled, supported by only ``paddr`` operations set. - anon - Applied to pages that containing data that not stored in files. From 384f585da032bd288aba106f608474f7e28725ee Mon Sep 17 00:00:00 2001 From: SJ Park Date: Fri, 11 Sep 2026 06:55:09 -0700 Subject: [PATCH 0600/1012] Docs/admin-guide/mm/damon/usage: update for probe_hits_wsum DAMOS filter Update DAMON usage document for the newly added probe hits weighted sum based DAMOS core filter type. Link: https://lore.kernel.org/20260911135510.96914-8-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/admin-guide/mm/damon/usage.rst | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/Documentation/admin-guide/mm/damon/usage.rst b/Documentation/admin-guide/mm/damon/usage.rst index 023c6334024f8e..d3e37400367bdd 100644 --- a/Documentation/admin-guide/mm/damon/usage.rst +++ b/Documentation/admin-guide/mm/damon/usage.rst @@ -551,6 +551,10 @@ and ``damon_target_idx``. To ``type`` file, you can write the type of the filter. Refer to :ref:`the design doc ` for available type names, their meaning and on what layer those are handled. +For ``probe_hits_wsum`` type, you can specify the minimum and maximum probe +hits weighted sum value for the filter to ``min`` and ``max`` files, +respectively. + For ``memcg`` type, you can specify the memory cgroup of the interest by writing the path of the memory cgroup from the cgroups mount point to ``memcg_path`` file. For ``addr`` type, you can specify the start and end From 02b5b8971aae18ced516e45dbf65d2faa9206594 Mon Sep 17 00:00:00 2001 From: Jaeyeon Lee Date: Sat, 12 Sep 2026 22:29:03 +0200 Subject: [PATCH 0601/1012] selftests/mm: skip khugepaged file tests if mkfs.xfs is unavailable The XFS setup for ./khugepaged all:file checks that the kernel supports XFS but not that mkfs.xfs is installed. When CONFIG_XFS_FS=y and xfsprogs is missing, mkfs.xfs and mount both fail, but SPLIT_HUGE_PAGE_TEST_XFS_PATH is assigned from mktemp -d and stays set. The test then runs against plain tmpfs and six subtests fail. The script already has a skip path for this test, but it never runs because the path variable is always set. Assign SPLIT_HUGE_PAGE_TEST_XFS_PATH only once mkfs.xfs and mount have both succeeded, and remove the image and directory otherwise. Link: https://lore.kernel.org/20260912202903.16157-1-jaeyeon.lee.dev@gmail.com Signed-off-by: Jaeyeon Lee Signed-off-by: Andrew Morton Suggested-by: Zi Yan Reviewed-by: Zi Yan Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Assisted-by: LLM Cc: Shuah Khan Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko --- tools/testing/selftests/mm/run_vmtests.sh | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 9bbef9410ccc35..2e5e7975ff4d49 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -435,11 +435,15 @@ if [ -z "${SPLIT_HUGE_PAGE_TEST_XFS_PATH}" ]; then if test_selected "thp"; then if grep xfs /proc/filesystems &>/dev/null; then XFS_IMG=$(mktemp /tmp/xfs_img_XXXXXX) - SPLIT_HUGE_PAGE_TEST_XFS_PATH=$(mktemp -d /tmp/xfs_dir_XXXXXX) + XFS_DIR=$(mktemp -d /tmp/xfs_dir_XXXXXX) truncate -s 314572800 ${XFS_IMG} - mkfs.xfs -q ${XFS_IMG} - mount -o loop ${XFS_IMG} ${SPLIT_HUGE_PAGE_TEST_XFS_PATH} - MOUNTED_XFS=1 + if mkfs.xfs -q ${XFS_IMG} && mount -t xfs -o loop ${XFS_IMG} ${XFS_DIR}; then + SPLIT_HUGE_PAGE_TEST_XFS_PATH=${XFS_DIR} + MOUNTED_XFS=1 + else + rmdir ${XFS_DIR} + rm -f ${XFS_IMG} + fi fi fi fi From fe1dc9c84b5c59779dccae0b065a3006c74c0d3e Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Sat, 26 Sep 2026 06:51:09 -0400 Subject: [PATCH 0602/1012] mm/mempolicy: use vm_normal_folio_pmd() in queue_folios_pmd() Patch series "mm: stop calling pmd_folio() on special PMDs", v3. Two page table walkers resolve the folio behind a PMD with pmd_folio(), which is only valid for a PMD mapping a refcounted struct page: madvise_cold_or_pageout_pte_range() mm/madvise.c queue_folios_pmd() mm/mempolicy.c vmf_insert_pfn_pmd() installs special PMDs holding a raw pfn that need not have a memmap entry at all. Both walkers can reach one and fault on the first folio field read. The PTE halves of both already use vm_normal_folio(); these two patches make the PMD halves match. The four callers of vmf_insert_pfn_pmd(), and which walker each reaches: drivers/vfio/pci/vfio_pci_core.c VM_PFNMAP mempolicy drivers/gpu/drm/drm_gem_shmem_helper.c VM_PFNMAP mempolicy drivers/gpu/drm/panthor/panthor_gem.c VM_PFNMAP mempolicy drivers/hv/mshv_vtl_main.c VM_MIXEDMAP both can_madv_lru_vma() rejects VM_PFNMAP, so only mshv_vtl_low reaches the madvise walker, and that needs CAP_SYS_ADMIN. queue_pages_walk_ops supplies its own ->test_walk, so walk_page_test()'s generic VM_PFNMAP skip never runs and vfio-pci is reachable by any process holding the device fd. Hence the different stable tags. One behaviour change: mbind(MPOL_MF_STRICT) over a PMD mapped VM_PFNMAP region now returns 0 rather than -EIO. The PTE loop already returned 0 there. drm_gem_shmem and panthor are where this is observable, since they PMD map pages that do have a memmap entry and so never faulted. Reproducer ========== No hardware needed. An out of tree module stands in for the drivers above: three misc devices, each with a ->huge_fault calling vmf_insert_pfn_pmd(), plus VM_HUGEPAGE so the fault path takes the PMD branch. /dev/pmdspec_mixed VM_MIXEDMAP, pfn at the 1 TiB mark, no memmap /dev/pmdspec_pfnmap VM_PFNMAP, pfn at the 1 TiB mark, no memmap /dev/pmdspec_real VM_PFNMAP, real alloc_pages(PMD_ORDER) on node 0 Userspace maps the device into a PMD aligned window, reads one byte to fault the PMD in, checks a module parameter to confirm it went in, then issues the operation. vng --run --user root --memory 4G --verbose \ --append "numa=fake=2" \ --exec "insmod pmdspec.ko && ./pmdspec_test " numa=fake=2 gives a node 1 to bind to; the module allocates its real page on node 0, which is what makes queue_folio_required() true. subtest operation parent series -------------------------------------------------------------------- madv_cold madvise(MADV_COLD) oops ret=0 madv_pageout madvise(MADV_PAGEOUT) oops ret=0 mbind_mixed mbind(MPOL_BIND, n1, MPOL_MF_MOVE) oops ret=0 mbind_pfnmap mbind(MPOL_BIND, n1, MPOL_MF_STRICT) oops ret=0 mbind_real mbind(MPOL_BIND, n1, MPOL_MF_STRICT) -EIO ret=0 Two things the table shows that are easy to miss in the code: - mbind_mixed passes only MPOL_MF_MOVE. MPOL_MF_STRICT is not needed for a VM_MIXEDMAP vma: walk_page_test() only skips VM_PFNMAP, and vma_migratable() is true for VM_MIXEDMAP. - mbind_real demonstrates the user visible change (-EIO -> 0) This patch (of 2): mmap a VM_PFNMAP region whose ->huge_fault installs a PMD through vmf_insert_pfn_pmd() - a vfio-pci MMIO BAR does this - then mbind(p, len, MPOL_BIND, &mask, maxnode, MPOL_MF_STRICT); With a stand-in module for the driver: BUG: unable to handle page fault for address: fffff96dc0000008 RIP: 0010:queue_folios_pte_range+0xaf/0x440 walk_pgd_range+0x52b/0xaf0 __walk_page_range+0x6a/0x1d0 walk_page_range_mm_unsafe+0x193/0x230 queue_pages_range+0x64/0xa0 do_mbind+0x25e/0x640 queue_folios_pmd(), inlined above, calls pmd_folio() on that PMD. The pfn is raw MMIO with no memmap entry, so the folio lands in unpopulated vmemmap. Neither guard stops the walk: walk_page_test() skips VM_PFNMAP, but queue_pages_walk_ops supplies ->test_walk, so it never runs queue_pages_test_walk() honours vma_migratable(), but only while MPOL_MF_STRICT is clear A VM_MIXEDMAP vma needs neither flag, being vma_migratable(), so plain mbind(MPOL_MF_MOVE) reaches this too - and there the bad folio carries on into migrate_folio_add() and folio_isolate_lru(). mshv_vtl_low is such a mapping. Use vm_normal_folio_pmd() and skip on NULL, as the PTE loop in queue_folios_pte_range() already does with vm_normal_folio(). This also filters the huge zero PMD, so its separate check is no longer needed. mbind(MPOL_MF_STRICT) over a PMD mapped VM_PFNMAP region now returns 0 rather than -EIO. The PTE loop already returned 0 there. Link: https://lore.kernel.org/20260926105110.2156652-1-gourry@gourry.net Link: https://lore.kernel.org/20260926105110.2156652-2-gourry@gourry.net Fixes: 3c8e44c9b369 ("mm: mark special bits for huge pfn mappings when inject") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260817220810.1175596-1-gourry%40gourry.net Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Assisted-by: LLM Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Vlastimil Babka Cc: Jann Horn Cc: Joshua Hahn Cc: Rakie Kim Cc: Ying Huang Cc: Peter Xu Cc: Jason Gunthorpe Cc: --- mm/mempolicy.c | 13 +++++-------- 1 file changed, 5 insertions(+), 8 deletions(-) diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 95dba5d919e925..e9860fb9f73f8d 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -667,7 +667,8 @@ static inline bool queue_folio_required(struct folio *folio, return node_isset(nid, *qp->nmask) == !(flags & MPOL_MF_INVERT); } -static void queue_folios_pmd(pmd_t *pmd, struct mm_walk *walk) +static void queue_folios_pmd(pmd_t *pmd, unsigned long addr, + struct mm_walk *walk) { struct folio *folio; struct queue_pages *qp = walk->private; @@ -678,13 +679,9 @@ static void queue_folios_pmd(pmd_t *pmd, struct mm_walk *walk) qp->nr_failed++; return; } - folio = pmd_folio(pmdval); - if (folio_is_zone_device(folio)) + folio = vm_normal_folio_pmd(walk->vma, addr, pmdval); + if (!folio || folio_is_zone_device(folio)) return; - if (is_huge_zero_folio(folio)) { - walk->action = ACTION_CONTINUE; - return; - } if (!queue_folio_required(folio, qp)) return; if (!(qp->flags & (MPOL_MF_MOVE | MPOL_MF_MOVE_ALL)) || @@ -717,7 +714,7 @@ static int queue_folios_pte_range(pmd_t *pmd, unsigned long addr, ptl = pmd_trans_huge_lock(pmd, vma); if (ptl) { - queue_folios_pmd(pmd, walk); + queue_folios_pmd(pmd, addr, walk); spin_unlock(ptl); goto out; } From 1f396d32f63c62f949dd64b0e3846182cfbc3bda Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Sat, 26 Sep 2026 06:51:10 -0400 Subject: [PATCH 0603/1012] mm/madvise: use vm_normal_folio_pmd() in cold/pageout PMD range mmap a VM_MIXEDMAP region whose ->huge_fault installs a PMD through vmf_insert_pfn_pmd() - mshv_vtl_low does this, and needs CAP_SYS_ADMIN to open - then: madvise(p, PMD_SIZE, MADV_PAGEOUT); With a stand-in module for the driver: BUG: unable to handle page fault for address: fffff587c0000008 RIP: 0010:madvise_cold_or_pageout_pte_range+0x410/0x9b0 walk_pgd_range+0x52b/0xaf0 __walk_page_range+0x6a/0x1d0 walk_page_range_vma_unsafe+0x8e/0x120 madvise_pageout+0xb2/0x180 madvise_vma_behavior+0x46b/0xa90 do_madvise+0x108/0x190 __x64_sys_madvise+0x26/0x30 Nothing validates the pfn on the way in: can_madv_lru_vma() rejects VM_PFNMAP, but not VM_MIXEDMAP can_fault() *pfn = vmf->pgoff & ~(mask >> PAGE_SHIFT); vmf_insert_pfn_pmd() no pfn_valid() check pmd_folio() pfn_to_page() -> unpopulated vmemmap Even with a valid pfn the path is wrong. The mapping carries no rmap, so folio_maybe_mapped_shared() sees mapcount 0, and the walker goes on to folio_deactivate(), or folio_isolate_lru() plus reclaim_pages(), against a folio this mapping does not own. Use vm_normal_folio_pmd() and skip on NULL, as the PTE half of this same walker already does with vm_normal_folio(). This also filters the huge zero PMD, so its separate check is no longer needed. Link: https://lore.kernel.org/20260926105110.2156652-3-gourry@gourry.net Fixes: 3c8e44c9b369 ("mm: mark special bits for huge pfn mappings when inject") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260817220810.1175596-1-gourry%40gourry.net Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Assisted-by: LLM Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Jann Horn Cc: Joshua Hahn Cc: Rakie Kim Cc: Ying Huang Cc: Peter Xu Cc: Jason Gunthorpe Cc: # v6.19+ --- mm/madvise.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/mm/madvise.c b/mm/madvise.c index acb5215e2f079f..1cfb0433229310 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -395,16 +395,15 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd, return 0; orig_pmd = *pmd; - if (is_huge_zero_pmd(orig_pmd)) - goto huge_unlock; - if (unlikely(!pmd_present(orig_pmd))) { VM_WARN_ON_ONCE(!pmd_is_migration_entry(orig_pmd) && !pmd_is_device_private_entry(orig_pmd)); goto huge_unlock; } - folio = pmd_folio(orig_pmd); + folio = vm_normal_folio_pmd(vma, addr, orig_pmd); + if (!folio) + goto huge_unlock; if (folio_is_zone_device(folio)) goto huge_unlock; From 01eb9488121e6f3cbb9b8eec169ac567d13a4c16 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Tue, 22 Sep 2026 22:29:01 -0400 Subject: [PATCH 0604/1012] mm: refactor find_next_best_node to find_next_best_node_in Patch series "mm: refactor zonelist constructors and iterators", v3. find_next_best_node() picks the next-closest node when building a fallback list, and hardcodes N_MEMORY as the set it picks from. Refactor it into find_next_best_node_in(), which takes the candidate set explicitly. This makes the existing behaviour explicit at both mm/memory-tiers.c call sites - they select demotion targets in fallback order from N_MEMORY - and lets callers narrow that set. Then extract the per-node construction loop out of build_zonelists() into build_node_zonelist(), parameterised on the candidate nodemask and destination zonelist index. Together these allow a zonelist to be built over a candidate set other than N_MEMORY, into a zonelist other than FALLBACK, and iterated in fallback order over a caller-defined subset. These are prerequisites for generating a private node zonelist (nodes unreachable by default), but are otherwise general improvements to the existing interfaces so I'm proposing them separately. No functional change intended - purely refactor commits. This patch (of 2): find_next_best_node() picks the next-closest node for a fallback list from the full N_MEMORY set. Refactor it into find_next_best_node_in(), which takes an explicit candidates nodemask. This enables building fallback lists with non-N_MEMORY candidates. No functional change: every caller still selects from N_MEMORY. Link: https://lore.kernel.org/20260923022902.2433614-1-gourry@gourry.net Link: https://lore.kernel.org/20260923022902.2433614-2-gourry@gourry.net Signed-off-by: Gregory Price Signed-off-by: Andrew Morton Reviewed-by: Vlastimil Babka (SUSE) Acked-by: Balbir Singh Reviewed-by: Zi Yan Reviewed-by: Zenghui Yu (Huawei) Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Joshua Hahn Cc: Rakie Kim Cc: Ying Huang Cc: Brendan Jackman Cc: Johannes Weiner --- mm/internal.h | 6 ++++-- mm/memory-tiers.c | 7 ++++--- mm/page_alloc.c | 13 ++++++++----- 3 files changed, 16 insertions(+), 10 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 0dca33db068f6b..05179c4b2090ef 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -1124,7 +1124,8 @@ extern int node_reclaim_mode; extern unsigned long node_reclaim(struct pglist_data *pgdat, gfp_t gfp_mask, unsigned int order); -extern int find_next_best_node(int node, nodemask_t *used_node_mask); +int find_next_best_node_in(int node, nodemask_t *used_node_mask, + const nodemask_t *candidates); #else #define node_reclaim_mode 0 @@ -1133,7 +1134,8 @@ static inline unsigned long node_reclaim(struct pglist_data *pgdat, { return 0; } -static inline int find_next_best_node(int node, nodemask_t *used_node_mask) +static inline int find_next_best_node_in(int node, nodemask_t *used_node_mask, + const nodemask_t *candidates) { return NUMA_NO_NODE; } diff --git a/mm/memory-tiers.c b/mm/memory-tiers.c index 54851d8a195b03..25e121851b5862 100644 --- a/mm/memory-tiers.c +++ b/mm/memory-tiers.c @@ -370,7 +370,7 @@ int next_demotion_node(int node, const nodemask_t *allowed_mask) * closest demotion target. */ nodes_complement(mask, *allowed_mask); - return find_next_best_node(node, &mask); + return find_next_best_node_in(node, &mask, &node_states[N_MEMORY]); } static void disable_all_demotion_targets(void) @@ -450,7 +450,7 @@ static void establish_demotion_targets(void) memtier = list_next_entry(memtier, list); tier_nodes = get_memtier_nodemask(memtier); /* - * find_next_best_node, use 'used' nodemask as a skip list. + * find_next_best_node_in, use 'used' nodemask as a skip list. * Add all memory nodes except the selected memory tier * nodelist to skip list so that we find the best node from the * memtier nodelist. @@ -463,7 +463,8 @@ static void establish_demotion_targets(void) * in the preferred mask when allocating pages during demotion. */ do { - target = find_next_best_node(node, &tier_nodes); + target = find_next_best_node_in(node, &tier_nodes, + &node_states[N_MEMORY]); if (target == NUMA_NO_NODE) break; diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 5dc259788bcc12..df45d4f3c48070 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -5820,9 +5820,10 @@ static int numa_zonelist_order_handler(const struct ctl_table *table, int write, static int node_load[MAX_NUMNODES]; /** - * find_next_best_node - find the next node that should appear in a given node's fallback list + * find_next_best_node_in - find the next node that should appear in a given node's fallback list * @node: node whose fallback list we're appending * @used_node_mask: nodemask_t of already used nodes + * @candidates: nodemask_t of nodes eligible for selection * * We use a number of factors to determine which is the next node that should * appear on a given node's fallback list. The node should not have appeared @@ -5834,7 +5835,8 @@ static int node_load[MAX_NUMNODES]; * * Return: node id of the found node or %NUMA_NO_NODE if no node is found. */ -int find_next_best_node(int node, nodemask_t *used_node_mask) +int find_next_best_node_in(int node, nodemask_t *used_node_mask, + const nodemask_t *candidates) { int n, val; int min_val = INT_MAX; @@ -5844,12 +5846,12 @@ int find_next_best_node(int node, nodemask_t *used_node_mask) * Use the local node if we haven't already, but for memoryless local * node, we should skip it and fall back to other nodes. */ - if (!node_isset(node, *used_node_mask) && node_state(node, N_MEMORY)) { + if (!node_isset(node, *used_node_mask) && node_isset(node, *candidates)) { node_set(node, *used_node_mask); return node; } - for_each_node_state(n, N_MEMORY) { + for_each_node_mask(n, *candidates) { /* Don't want a node to appear more than once */ if (node_isset(n, *used_node_mask)) @@ -5934,7 +5936,8 @@ static void build_zonelists(pg_data_t *pgdat) prev_node = local_node; memset(node_order, 0, sizeof(node_order)); - while ((node = find_next_best_node(local_node, &used_mask)) >= 0) { + while ((node = find_next_best_node_in(local_node, &used_mask, + &node_states[N_MEMORY])) >= 0) { /* * We don't want to pressure a particular node. * So adding penalty to the first node in same From 0ece837b9cd8874b81f1b1a8a66f1c4dfead28da Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Tue, 22 Sep 2026 22:29:02 -0400 Subject: [PATCH 0605/1012] mm/page_alloc: refactor build_node_zonelist() out of build_zonelists() Extract per-node fallback-list construction into build_node_zonelist(). Build each selected node directly into the destination zonelist so no intermediate node_order array or node count is needed. Print the fallback order as each node is added. This lets us build new zonelists from candidate nodemasks instead of just the default N_MEMORY node state list. Add a zlidx argument so callers explicitly select the destination zonelist. The existing caller continues to use ZONELIST_FALLBACK. No functional change: build_zonelists() builds and prints the same FALLBACK list over N_MEMORY with node_load updates as before. Link: https://lore.kernel.org/20260923022902.2433614-3-gourry@gourry.net Signed-off-by: Gregory Price Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Vlastimil Babka (SUSE) Reviewed-by: Zenghui Yu (Huawei) Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Joshua Hahn Cc: Rakie Kim Cc: Ying Huang Cc: Brendan Jackman Cc: Johannes Weiner --- mm/page_alloc.c | 63 ++++++++++++++++++------------------------------- 1 file changed, 23 insertions(+), 40 deletions(-) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index df45d4f3c48070..306e3d34adf45a 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -5884,31 +5884,6 @@ int find_next_best_node_in(int node, nodemask_t *used_node_mask, } -/* - * Build zonelists ordered by node and zones within node. - * This results in maximum locality--normal zone overflows into local - * DMA zone, if any--but risks exhausting DMA zone. - */ -static void build_zonelists_in_node_order(pg_data_t *pgdat, int *node_order, - unsigned nr_nodes) -{ - struct zoneref *zonerefs; - int i; - - zonerefs = pgdat->node_zonelists[ZONELIST_FALLBACK]._zonerefs; - - for (i = 0; i < nr_nodes; i++) { - int nr_zones; - - pg_data_t *node = NODE_DATA(node_order[i]); - - nr_zones = build_zonerefs_node(node, zonerefs); - zonerefs += nr_zones; - } - zonerefs->zone = NULL; - zonerefs->zone_idx = 0; -} - /* * Build __GFP_THISNODE zonelists */ @@ -5924,20 +5899,24 @@ static void build_thisnode_zonelists(pg_data_t *pgdat) zonerefs->zone_idx = 0; } -static void build_zonelists(pg_data_t *pgdat) +/* + * Build one zonelist ordered by node and zones within node. This results in + * maximum locality--normal zone overflows into local DMA zone, if any--but + * risks exhausting DMA zone. + */ +static void build_node_zonelist(pg_data_t *pgdat, const nodemask_t *candidates, + int zlidx) { - static int node_order[MAX_NUMNODES]; - int node, nr_nodes = 0; + struct zoneref *zonerefs = pgdat->node_zonelists[zlidx]._zonerefs; nodemask_t used_mask = NODE_MASK_NONE; - int local_node, prev_node; + int local_node = pgdat->node_id; + int prev_node = local_node; + int node; - /* NUMA-aware ordering of nodes */ - local_node = pgdat->node_id; - prev_node = local_node; + pr_info("Fallback order for Node %d: ", local_node); - memset(node_order, 0, sizeof(node_order)); while ((node = find_next_best_node_in(local_node, &used_mask, - &node_states[N_MEMORY])) >= 0) { + candidates)) >= 0) { /* * We don't want to pressure a particular node. * So adding penalty to the first node in same @@ -5947,18 +5926,22 @@ static void build_zonelists(pg_data_t *pgdat) node_distance(local_node, prev_node)) node_load[node] += 1; - node_order[nr_nodes++] = node; + zonerefs += build_zonerefs_node(NODE_DATA(node), zonerefs); + pr_cont("%d ", node); prev_node = node; } - build_zonelists_in_node_order(pgdat, node_order, nr_nodes); - build_thisnode_zonelists(pgdat); - pr_info("Fallback order for Node %d: ", local_node); - for (node = 0; node < nr_nodes; node++) - pr_cont("%d ", node_order[node]); + zonerefs->zone = NULL; + zonerefs->zone_idx = 0; pr_cont("\n"); } +static void build_zonelists(pg_data_t *pgdat) +{ + build_node_zonelist(pgdat, &node_states[N_MEMORY], ZONELIST_FALLBACK); + build_thisnode_zonelists(pgdat); +} + #ifdef CONFIG_HAVE_MEMORYLESS_NODES /* * Return node id of node used for "local" allocations. From 5beb2fee9a1974a8fe3794bb7259076f9039c21a Mon Sep 17 00:00:00 2001 From: Yury Norov Date: Fri, 11 Sep 2026 18:14:41 -0400 Subject: [PATCH 0606/1012] compiler.h: add ASSERT_STATIC_STORAGE() Patch series "Catch automatic storage in IDA and Maple Tree definitions". A 0day report [1] from the region allocation benchmark exposed a lockdep initialization bug: a stack-local Maple Tree used MTREE_INIT(), whose embedded lock has a static initializer. On the first allocation, lockdep rejected the lock address as a non-static class key and disabled locking validation. The IDA benchmark had the same issue, masked because it ran after Maple Tree had already disabled lockdep. The fix [2] switches the test to using mt_init_flags() and ida_init(). This series adds a compile-time check to the related DEFINE_IDA() and DEFINE_MTREE() declaration macros to catch the same class of mistake earlier. Patch 1 introduces ASSERT_STATIC_STORAGE(). It declares an unused static pointer initialized with the object's address, requiring that address to be a valid static initializer. Patches 2 and 3 apply the helper to IDA and Maple Tree definitions, respectively. The helper is mirrored in the tools compiler header. The existing automatic local IDAs and Maple Trees in the userspace radix-tree tests are converted to runtime initialization. The interval-tree span test keeps its existing mt_init_flags() call and uses a plain Maple Tree declaration. For example, an automatic local definition: void example(void) { DEFINE_IDA(ida); ida_destroy(&ida); } now produces: error: initializer element is not constant note: in expansion of macro 'ASSERT_STATIC_STORAGE' note: in expansion of macro 'DEFINE_IDA' File-scope definitions and static local definitions remain valid. Automatic local objects should use ida_init(), mt_init(), or mt_init_flags(). The check is limited to declaration macros. Direct uses of IDA_INIT(), MTREE_INIT(), and MTREE_INIT_EXT() remain unchanged. The helper cannot be inserted directly into those initializer expressions because it expands to a declaration. Validated by GCC and Clang checks accepting static storage and rejecting automatic storage The userspace IDR/IDA and Maple Tree test are passed as well. This patch (of 3): Static lock initializers rely on a persistent object address when lockdep assigns a lock-class key. Using such an initializer for an automatic local object can compile successfully but disable lockdep on the first lock acquisition. Add ASSERT_STATIC_STORAGE() for declaration macros that require static storage duration. It declares an unused static pointer initialized with the object's address. An automatic local object's address is not a valid static initializer, so the compiler rejects it. Mirror the helper in tools/include/linux/compiler.h because the userspace radix-tree tests include the kernel IDA and Maple Tree headers with the tools compiler definitions. For example: void example(void) { int object; ASSERT_STATIC_STORAGE(object); } GCC reports: error: initializer element is not constant name##_storage_check = &(name) ^ note: in expansion of macro 'ASSERT_STATIC_STORAGE' ASSERT_STATIC_STORAGE(object); File-scope objects and static local objects remain valid. The helper takes an object identifier and must be used as a declaration after that object has been declared. Link: https://lore.kernel.org/all/20260911155244.1406122-1-ynorov@nvidia.com/ Link: https://lore.kernel.org/20260911221444.1523311-2-ynorov@nvidia.com Link: https://download.01.org/0day-ci/archive/20260910/202609101106.771b567e-lkp@intel.com/ [1] Link: https://lore.kernel.org/all/20260911155244.1406122-1-ynorov@nvidia.com/ [2] Signed-off-by: Yury Norov Signed-off-by: Andrew Morton Cc: Alice Ryhl Cc: Andrew Ballance Cc: Christopher Li Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) --- include/linux/compiler.h | 5 +++++ tools/include/linux/compiler.h | 5 +++++ 2 files changed, 10 insertions(+) diff --git a/include/linux/compiler.h b/include/linux/compiler.h index cb2f6050bdf7dc..ef9036fa5413d6 100644 --- a/include/linux/compiler.h +++ b/include/linux/compiler.h @@ -275,6 +275,11 @@ static inline void *offset_to_ptr(const int *off) #define __ADDRESSABLE(sym) \ ___ADDRESSABLE(sym, __section(".discard.addressable")) +/* Enforce static storage duration. */ +#define ASSERT_STATIC_STORAGE(name) \ + static typeof(name) * const __always_unused \ + name##_storage_check = &(name) + /* * This returns a constant expression while determining if an argument is * a constant expression, most importantly without evaluating the argument. diff --git a/tools/include/linux/compiler.h b/tools/include/linux/compiler.h index f2f54b0381680b..03ecf90866436b 100644 --- a/tools/include/linux/compiler.h +++ b/tools/include/linux/compiler.h @@ -73,6 +73,11 @@ # define __same_type(a, b) __builtin_types_compatible_p(typeof(a), typeof(b)) #endif +/* Enforce static storage duration. */ +#define ASSERT_STATIC_STORAGE(name) \ + static typeof(name) * const __always_unused \ + name##_storage_check = &(name) + /* * This returns a constant expression while determining if an argument is * a constant expression, most importantly without evaluating the argument. From 0142c1327b6bcc38f4ee930e8887fe2f81d69738 Mon Sep 17 00:00:00 2001 From: Yury Norov Date: Fri, 11 Sep 2026 18:14:42 -0400 Subject: [PATCH 0607/1012] idr: assert static storage for DEFINE_IDA() DEFINE_IDA() uses IDA_INIT(), which initializes the embedded XArray lock with a static spinlock initializer. For an automatic local IDA, lockdep cannot use the lock address as a persistent class key and reports "INFO: trying to register non-static key" before disabling itself. Apply ASSERT_STATIC_STORAGE() to DEFINE_IDA() so that this misuse is rejected at compile time. For example: void example(void) { DEFINE_IDA(ida); ida_destroy(&ida); } GCC reports: error: initializer element is not constant name##_storage_check = &(name) ^ note: in expansion of macro 'ASSERT_STATIC_STORAGE' ASSERT_STATIC_STORAGE(name) note: in expansion of macro 'DEFINE_IDA' DEFINE_IDA(ida); File-scope definitions and static DEFINE_IDA() within a function remain valid. Automatic local IDAs must instead be initialized with ida_init(). Direct uses of IDA_INIT() are not covered by this declaration check. Convert the five automatic local IDAs in the userspace radix-tree tests to ida_init() so they satisfy the new requirement. Validated file-scope and static local definitions with a kernel object build, and confirmed that an automatic local definition fails to compile. Link: https://lore.kernel.org/20260911221444.1523311-3-ynorov@nvidia.com Signed-off-by: Yury Norov Signed-off-by: Andrew Morton Cc: Alice Ryhl Cc: Andrew Ballance Cc: Christopher Li Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) --- include/linux/idr.h | 5 ++++- tools/testing/radix-tree/idr-test.c | 20 +++++++++++++++----- 2 files changed, 19 insertions(+), 6 deletions(-) diff --git a/include/linux/idr.h b/include/linux/idr.h index 789e23e6744410..e2a4b6298511c7 100644 --- a/include/linux/idr.h +++ b/include/linux/idr.h @@ -16,6 +16,7 @@ #include #include #include +#include struct idr { struct radix_tree_root idr_rt; @@ -269,7 +270,9 @@ struct ida { #define IDA_INIT(name) { \ .xa = XARRAY_INIT(name, IDA_INIT_FLAGS) \ } -#define DEFINE_IDA(name) struct ida name = IDA_INIT(name) +#define DEFINE_IDA(name) \ + struct ida name = IDA_INIT(name); \ + ASSERT_STATIC_STORAGE(name) int ida_alloc_range(struct ida *, unsigned int min, unsigned int max, gfp_t); void ida_free(struct ida *, unsigned int id); diff --git a/tools/testing/radix-tree/idr-test.c b/tools/testing/radix-tree/idr-test.c index 945144e9850724..6fcba5b5870b4e 100644 --- a/tools/testing/radix-tree/idr-test.c +++ b/tools/testing/radix-tree/idr-test.c @@ -460,9 +460,11 @@ void ida_dump(struct ida *); */ void ida_check_nomem(void) { - DEFINE_IDA(ida); + struct ida ida; int id; + ida_init(&ida); + id = ida_alloc_min(&ida, 256, GFP_NOWAIT); IDA_BUG_ON(&ida, id != -ENOMEM); id = ida_alloc_min(&ida, 1UL << 30, GFP_NOWAIT); @@ -475,9 +477,11 @@ void ida_check_nomem(void) */ void ida_check_conv_user(void) { - DEFINE_IDA(ida); + struct ida ida; unsigned long i; + ida_init(&ida); + for (i = 0; i < 1000000; i++) { int id = ida_alloc(&ida, GFP_NOWAIT); if (id == -ENOMEM) { @@ -496,11 +500,13 @@ void ida_check_conv_user(void) void ida_check_random(void) { - DEFINE_IDA(ida); + struct ida ida; DECLARE_BITMAP(bitmap, 2048); unsigned int i; time_t s = time(NULL); + ida_init(&ida); + repeat: memset(bitmap, 0, sizeof(bitmap)); for (i = 0; i < 100000; i++) { @@ -522,9 +528,11 @@ void ida_check_random(void) void ida_alloc_free_test(void) { - DEFINE_IDA(ida); + struct ida ida; unsigned long i; + ida_init(&ida); + for (i = 0; i < 10000; i++) assert(ida_alloc_max(&ida, 20000, GFP_KERNEL) == i); assert(ida_alloc_range(&ida, 5, 30, GFP_KERNEL) < 0); @@ -576,10 +584,12 @@ static void *ida_leak_fn(void *arg) void ida_thread_tests(void) { - DEFINE_IDA(ida); + struct ida ida; pthread_t threads[20]; int i; + ida_init(&ida); + for (i = 0; i < ARRAY_SIZE(threads); i++) if (pthread_create(&threads[i], NULL, ida_random_fn, NULL)) { perror("creating ida thread"); From 926223398765f7c9c5a58902fc5d8a89e014e1f6 Mon Sep 17 00:00:00 2001 From: Yury Norov Date: Fri, 11 Sep 2026 18:14:43 -0400 Subject: [PATCH 0608/1012] maple_tree: assert static storage for DEFINE_MTREE() DEFINE_MTREE() uses MTREE_INIT(), which initializes the tree's embedded lock with a static spinlock initializer. If the tree is an automatic local object, lockdep rejects its address as a non-static class key and disables locking validation on the first lock acquisition. Apply ASSERT_STATIC_STORAGE() to DEFINE_MTREE() to catch automatic local definitions at compile time. For example: void example(void) { DEFINE_MTREE(mt); mtree_destroy(&mt); } GCC reports: error: initializer element is not constant name##_storage_check = &(name) ^ note: in expansion of macro 'ASSERT_STATIC_STORAGE' ASSERT_STATIC_STORAGE(name) note: in expansion of macro 'DEFINE_MTREE' DEFINE_MTREE(mt); File-scope definitions and static local trees remain valid. Automatic local trees must instead use mt_init() or mt_init_flags(). Direct uses of MTREE_INIT() and MTREE_INIT_EXT() are unchanged. Replace the local DEFINE_MTREE() in the interval-tree span test with a plain declaration; the test already initializes the tree with mt_init_flags() before use. Convert the three local Maple Trees in the userspace radix-tree tests to mt_init(). Validated file-scope and static local definitions with a kernel object build, and confirmed that an automatic local definition fails to compile. The Maple Tree test and region allocation benchmark objects also build with lockdep enabled. Link: https://lore.kernel.org/20260911221444.1523311-4-ynorov@nvidia.com Signed-off-by: Yury Norov Signed-off-by: Andrew Morton Cc: Alice Ryhl Cc: Andrew Ballance Cc: Christopher Li Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) --- include/linux/maple_tree.h | 4 +++- lib/interval_tree_test.c | 2 +- tools/testing/radix-tree/maple.c | 12 +++++++++--- 3 files changed, 13 insertions(+), 5 deletions(-) diff --git a/include/linux/maple_tree.h b/include/linux/maple_tree.h index e595ae5cd0eed4..e30e57f5f3df9f 100644 --- a/include/linux/maple_tree.h +++ b/include/linux/maple_tree.h @@ -9,6 +9,7 @@ */ #include +#include #include #include @@ -297,7 +298,8 @@ struct maple_tree { #endif #define DEFINE_MTREE(name) \ - struct maple_tree name = MTREE_INIT(name, 0) + struct maple_tree name = MTREE_INIT(name, 0); \ + ASSERT_STATIC_STORAGE(name) #define mtree_lock(mt) spin_lock((&(mt)->ma_lock)) #define mtree_lock_nested(mas, subclass) \ diff --git a/lib/interval_tree_test.c b/lib/interval_tree_test.c index b0b07270ce7c3c..06f77fb3179fc8 100644 --- a/lib/interval_tree_test.c +++ b/lib/interval_tree_test.c @@ -244,7 +244,7 @@ static int span_iteration_check(void) unsigned long start, last; struct interval_tree_span_iter span, mas_span; - DEFINE_MTREE(tree); + struct maple_tree tree; MA_STATE(mas, &tree, 0, 0); diff --git a/tools/testing/radix-tree/maple.c b/tools/testing/radix-tree/maple.c index d967e76a3c0650..bfe4b8626c429d 100644 --- a/tools/testing/radix-tree/maple.c +++ b/tools/testing/radix-tree/maple.c @@ -36022,10 +36022,12 @@ static noinline void __init check_erase_rebalance(struct maple_tree *mt) static noinline void __init check_mtree_dup(struct maple_tree *mt) { - DEFINE_MTREE(new); + struct maple_tree new; int i, j, ret, count = 0; unsigned int rand_seed = 17, rand; + mt_init(&new); + /* store a value at [0, 0] */ mt_init_flags(mt, 0); mtree_store_range(mt, 0, 0, xa_mk_value(0), GFP_KERNEL); @@ -36319,7 +36321,9 @@ static inline int check_vma_modification(struct maple_tree *mt) void farmer_tests(void) { struct maple_node *node; - DEFINE_MTREE(tree); + struct maple_tree tree; + + mt_init(&tree); mt_dump(&tree, mt_dump_dec); @@ -36432,9 +36436,11 @@ static unsigned long get_last_index(struct ma_state *mas) static void test_spanning_store_regression(void) { unsigned long from = 0, to = 0; - DEFINE_MTREE(tree); + struct maple_tree tree; MA_STATE(mas, &tree, 0, 0); + mt_init(&tree); + /* * Build a 3-level tree. We require a parent node below the root node * and 2 leaf nodes under it, so we can span the entirety of the right From b6ce99652033a5a96aac0732431cb9bd462e3d05 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Fri, 11 Sep 2026 22:06:11 +0100 Subject: [PATCH 0609/1012] kselftest: mm: remove exclusion of building soft-dirty test in arm64 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Since commit 0389c305ef56 (“selftests/mm: skip soft-dirty tests when CONFIG_MEM_SOFT_DIRTY is disabled”), the soft-dirty test is skipped when soft-dirty is not supported. There is therefore no reason to exclude the test from being built on arm64. Remove the arm64-specific exclusion. Link: https://lore.kernel.org/20260911210611.4001419-1-yeoreum.yun@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: David Hildenbrand (Arm) Acked-by: Lorenzo Stoakes (ARM) Tested-by: Zenghui Yu (Huawei) Cc: Alice Ryhl Cc: Andrew Ballance Cc: Christopher Li Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Yury Norov (NVIDIA) --- tools/testing/selftests/mm/Makefile | 3 --- tools/testing/selftests/mm/run_vmtests.sh | 5 +---- 2 files changed, 1 insertion(+), 7 deletions(-) diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index 2d5366196e3091..d3e9bd67904aa0 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -104,10 +104,7 @@ TEST_GEN_FILES += guard-regions TEST_GEN_FILES += merge TEST_GEN_FILES += rmap TEST_GEN_FILES += folio_split_race_test - -ifneq ($(ARCH),arm64) TEST_GEN_FILES += soft-dirty -endif ifeq ($(ARCH),x86_64) CAN_BUILD_I386 := $(shell ./../x86/check_cc.sh "$(CC)" ../x86/trivial_32bit_program.c -m32) diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 2e5e7975ff4d49..a1db516557023f 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -408,10 +408,7 @@ then CATEGORY="pkey" run_test ./protection_keys_64 fi -if [ -x ./soft-dirty ] -then - CATEGORY="soft_dirty" run_test ./soft-dirty -fi +CATEGORY="soft_dirty" run_test ./soft-dirty CATEGORY="pagemap" run_test ./pagemap_ioctl From 0294e2743a4dbe682e40b9c35dc1c18358cd9e5a Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Sun, 13 Sep 2026 14:30:31 +0800 Subject: [PATCH 0610/1012] mm: zswap: return -ENOENT when the swap device is gone zswap_writeback_entry() returns -EEXIST when get_swap_device() finds no device. -EEXIST is the shrinker's "page already in swap cache" signal, which makes zswap_shrinker_scan() stop shrinking entirely. A NULL get_swap_device() instead means the device is being swapped off, so the entry is simply stale. Return -ENOENT so the shrinker skips the stale entry and keeps scanning. It affects all swap devices. Link: https://lore.kernel.org/20260913063031.1689420-1-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Acked-by: Nhat Pham Cc: Chengming Zhou Cc: Chris Li Cc: Johannes Weiner Cc: Kairui Song --- mm/zswap.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/zswap.c b/mm/zswap.c index ff7c6742af50f5..584dd306376943 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1016,7 +1016,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry, /* try to allocate swap cache folio */ si = get_swap_device(swpentry); if (IS_ERR_OR_NULL(si)) - return -EEXIST; + return -ENOENT; mpol = get_task_policy(current); folio = swap_cache_alloc_folio(swpentry, GFP_KERNEL, BIT(0), NULL, mpol, From 92cc0ffd8447bd713271767f5172128e589af66a Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:27:57 -0400 Subject: [PATCH 0611/1012] mm/zsmalloc: replace PG_private with pointer comparison Patch series "Remove PG_private by using page/folio->private checks instead", v5. This patchset removes PG_private to make space for upcoming PG_folio for identifying pages from a folio (more details in Note below). Instead of checking PG_private, all code is changed to check page/folio->private != NULL instead. Overview === Most code uses folio_attach/detach/change_private() functions, so folio refcount is increased and decreased when folio->private is set and reset, respectively. There is no need to change them. Changes are needed for exceptional users: 1. zsmalloc uses PG_private to indicate first component zpdesc page and page->private is used to store zspage in zpdesc. To remove PG_private, is_first_zpdesc() is replaced by pointer comparison. 2. kernel/events/ring_buffer.c stores page order in page->private. Replacing PG_private with page->private != NULL works. 3. drivers/xen/grant-table.c stores xen_page_foreign in page->private, where on 32-bit, a pointer to xen_page_foreign is stored; on 64-bit, page->private is used as xen_page_foreign. PG_private check is replaced by page->private != NULL on 32-bit for xen_page_foreign deallocation. On 64-bit, page->private is cleared unconditionally since {domid=0, gref=0} (xen_page_foreign can be 0) is valid. 4. fs/crypto/crypto.c stores a folio pointer in page->private, PG_private checks are replaced by page->private != NULL. 5. fs/erofs has two different uses: 5a. folio->private is used to form a reversed list of the outputs of readahead_folio(). readahead_folio_last() is added to output folios in reversed order, so that ->private is no longer needed. 5b. folio->private is used as an in-flight I/O counter. Convert the code to use folio_attach/detach/get_private() and add bias==1 to the counter to avoid folio->private being zero. 6. fs/nfs/write.c: folio refcount maintenance is in a bigger scope than folio->private. So folio_attach/detach/get_private() is not used. Nothing to change. 7. fs/f2fs uses attach_page_private() to first reset folio->private then immediately sets PAGE_PRIVATE_NOT_POINTER bit on it. Change it to use attach_page_private() to set PAGE_PRIVATE_NOT_POINTER bit directly to avoid folio->private == NULL gap inside set_page_private_##name(). 8. hugetlb uses folio_change_private(folio, NULL) without folio refcount maintenance. Change it to folio->private = NULL. After the above changes, PG_private ops are converted to page/folio->private ops. folio_has_attached_private() is added to check filesystem-only private data by excluding swapcache and hugetlb folios, because swapcache folios overlap swp_entry_t swap with ->private and hugetlb sets its own flags in ->private. Note === 1. KPF_PRIVATE is removed after PG_private is removed. 2. Documentation/mm/hugetlbfs_reserv.rst is outdated, so I did not remove PG_private related text. It should be rewritten. 3. PG_folio is planned to be set on every page from a folio in page_rmappable_folio(), so folios with any order (currently PG_large_rmappable is used to identify >0 order folios, but not order-0 folios) can be identified. Then vm_insert_*() can correctly reject all folios and rmap code will only see folios. Eventually, page_folio() will return NULL for non-folio pages by checking PG_folio, but before that all existing users that treat compound pages as folios will need to be converted. This patch (of 17): zsmalloc uses PG_private to indicate first zpdesc in a zspage chain. It is equivalent to check zpdesc == zspage->first_zpdesc. Replace is_first_zpdesc() with zpdesc == zspage->first_zpdesc in obj_allocated(). For get_first_zpdesc(), first_zpdesc is from zspage->first_zpdesc, so replace is_first_zpdesc() with first_zpdesc->zspage == zspage, the second requirement of a zspage chain, where all zpdescs point to the same zspage. is_first_zpdesc(), is only used in VM_BUG_ON_PAGE(), so performance impact should be negligible. While at it, change VM_BUG_ON() to VM_WARN_ON_ONCE_PAGE(). It prepares for a future commit that remove PG_private. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-0-bb68b6a21869@nvidia.com Link: https://lore.kernel.org/20260920-remove-pg_private-v5-1-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: Johannes Weiner Reviewed-by: Sergey Senozhatsky Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Assisted-by: LLM Cc: Minchan Kim --- mm/zpdesc.h | 2 +- mm/zsmalloc.c | 24 ++++++------------------ 2 files changed, 7 insertions(+), 19 deletions(-) diff --git a/mm/zpdesc.h b/mm/zpdesc.h index b8258dc78548d1..4fd81c2e80769f 100644 --- a/mm/zpdesc.h +++ b/mm/zpdesc.h @@ -26,8 +26,8 @@ * with memcg_data. * * Page flags used: - * * PG_private identifies the first component page. * * PG_locked is used by page migration code. + * The first component page has zpdesc->zspage->first_zpdesc == zpdesc */ struct zpdesc { unsigned long flags; diff --git a/mm/zsmalloc.c b/mm/zsmalloc.c index 11be37c4317189..7ef80e0da62676 100644 --- a/mm/zsmalloc.c +++ b/mm/zsmalloc.c @@ -290,11 +290,6 @@ struct zs_pool { atomic_t compaction_in_progress; }; -static inline void zpdesc_set_first(struct zpdesc *zpdesc) -{ - SetPagePrivate(zpdesc_page(zpdesc)); -} - static inline void zpdesc_inc_zone_page_state(struct zpdesc *zpdesc) { inc_zone_page_state(zpdesc_page(zpdesc), NR_ZSPAGES); @@ -476,11 +471,6 @@ static void record_obj(unsigned long handle, unsigned long obj) WRITE_ONCE(*(unsigned long *)handle, obj); } -static inline bool __maybe_unused is_first_zpdesc(struct zpdesc *zpdesc) -{ - return PagePrivate(zpdesc_page(zpdesc)); -} - /* Protected by class->lock */ static inline int get_zspage_inuse(struct zspage *zspage) { @@ -496,7 +486,8 @@ static struct zpdesc *get_first_zpdesc(struct zspage *zspage) { struct zpdesc *first_zpdesc = zspage->first_zpdesc; - VM_BUG_ON_PAGE(!is_first_zpdesc(first_zpdesc), zpdesc_page(first_zpdesc)); + /* the first zpdesc must point back to this zspage */ + VM_WARN_ON_ONCE_PAGE(first_zpdesc->zspage != zspage, zpdesc_page(first_zpdesc)); return first_zpdesc; } @@ -838,7 +829,8 @@ static inline bool obj_allocated(struct zpdesc *zpdesc, void *obj, struct zspage *zspage = get_zspage(zpdesc); if (unlikely(ZsHugePage(zspage))) { - VM_BUG_ON_PAGE(!is_first_zpdesc(zpdesc), zpdesc_page(zpdesc)); + /* only first zpdesc holds the handle */ + VM_WARN_ON_ONCE_PAGE(zspage->first_zpdesc != zpdesc, zpdesc_page(zpdesc)); handle = zpdesc->handle; } else handle = *(unsigned long *)obj; @@ -853,9 +845,6 @@ static inline bool obj_allocated(struct zpdesc *zpdesc, void *obj, static void reset_zpdesc(struct zpdesc *zpdesc) { - struct page *page = zpdesc_page(zpdesc); - - ClearPagePrivate(page); zpdesc->zspage = NULL; zpdesc->next = NULL; /* PageZsmalloc is sticky until the page is freed to the buddy. */ @@ -1006,8 +995,8 @@ static void create_page_chain(struct size_class *class, struct zspage *zspage, * 1. all pages are linked together using zpdesc->next * 2. each sub-page point to zspage using zpdesc->zspage * - * we set PG_private to identify the first zpdesc (i.e. no other zpdesc - * has this flag set). + * The first zpdesc has its zspage->first_zpdesc set to itself, no + * other zpdesc has this set. */ for (i = 0; i < nr_zpdescs; i++) { zpdesc = zpdescs[i]; @@ -1015,7 +1004,6 @@ static void create_page_chain(struct size_class *class, struct zspage *zspage, zpdesc->next = NULL; if (i == 0) { zspage->first_zpdesc = zpdesc; - zpdesc_set_first(zpdesc); if (unlikely(class->objs_per_zspage == 1 && class->pages_per_zspage == 1)) SetZsHugePage(zspage); From 23af0cc5fa3327737e77fee66957af7d7ac3c85c Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:27:58 -0400 Subject: [PATCH 0612/1012] perf/ring_buffer: stop using PG_private as AUX page high-order marker A high-order AUX page sets PG_private on its first page and stores the order in first_page->private. Stop using PG_private and check first_page->private for AUX page order only in the ring buffer and its users. This is fine because page->private is 0 for order-0 AUX pages, matching the page order. It prepares for a future commit that remove PG_private. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-2-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: Usama Arif Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Peter Zijlstra Cc: Ingo Molnar Cc: Arnaldo Carvalho de Melo Cc: Namhyung Kim Cc: Thomas Gleixner Cc: Borislav Petkov Cc: Dave Hansen Cc: Mark Rutland Cc: Alexander Shishkin Cc: Jiri Olsa Cc: Ian Rogers Cc: Adrian Hunter Cc: James Clark Cc: "H. Peter Anvin" --- arch/x86/events/intel/bts.c | 3 --- arch/x86/events/intel/pt.c | 6 ++---- kernel/events/ring_buffer.c | 7 +++---- 3 files changed, 5 insertions(+), 11 deletions(-) diff --git a/arch/x86/events/intel/bts.c b/arch/x86/events/intel/bts.c index cbac54cb3a9ec5..5849392cf26d5b 100644 --- a/arch/x86/events/intel/bts.c +++ b/arch/x86/events/intel/bts.c @@ -66,9 +66,6 @@ static struct pmu bts_pmu; static int buf_nr_pages(struct page *page) { - if (!PagePrivate(page)) - return 1; - return 1 << page_private(page); } diff --git a/arch/x86/events/intel/pt.c b/arch/x86/events/intel/pt.c index 5754cd40556281..49349afee6119f 100644 --- a/arch/x86/events/intel/pt.c +++ b/arch/x86/events/intel/pt.c @@ -781,8 +781,7 @@ static int topa_insert_pages(struct pt_buffer *buf, int cpu, gfp_t gfp) struct page *p; p = virt_to_page(buf->data_pages[buf->nr_pages]); - if (PagePrivate(p)) - order = page_private(p); + order = page_private(p); if (topa_table_full(topa)) { topa = topa_alloc(cpu, gfp); @@ -1296,8 +1295,7 @@ static int pt_buffer_try_single(struct pt_buffer *buf, int nr_pages) if (!intel_pt_validate_hw_cap(PT_CAP_single_range_output)) goto out; - if (PagePrivate(p)) - order = page_private(p); + order = page_private(p); if (1 << order != nr_pages) goto out; diff --git a/kernel/events/ring_buffer.c b/kernel/events/ring_buffer.c index 1b1ffe0533e58a..9d3d324f512720 100644 --- a/kernel/events/ring_buffer.c +++ b/kernel/events/ring_buffer.c @@ -635,11 +635,10 @@ static struct page *rb_alloc_aux_page(int node, int order) /* * Communicate the allocation size to the driver: * if we managed to secure a high-order allocation, - * set its first page's private to this order; - * !PagePrivate(page) means it's just a normal page. + * set its first page's private to this order, otherwise page's + * private remains zero. */ split_page(page, order); - SetPagePrivate(page); set_page_private(page, order); } @@ -650,7 +649,7 @@ static void rb_free_aux_page(struct perf_buffer *rb, int idx) { struct page *page = virt_to_page(rb->aux_pages[idx]); - ClearPagePrivate(page); + set_page_private(page, 0); __free_page(page); } From 36d6e6c9bcd7c20ac071c5f7198fba57fc3f9e2e Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:27:59 -0400 Subject: [PATCH 0613/1012] xen/grant-table: stop setting PG_private on pages for grant mapping gnttab_alloc_pages() stores xen_page_foreign in page->private. On 32-bit, a pointer to an allocated xen_page_foreign is stored; on 64-bit, xen_page_forCcgn is stored inline. Checking page->private != NULL is enough to tell whether a xen_page_foreign needs to be freed on 32-bit and page->private is zeroed unconditionally on 64-bit. It prepares for a future commit that remove PG_private. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-3-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Juergen Gross Cc: Stefano Stabellini Cc: Oleksandr Tyshchenko --- drivers/xen/grant-table.c | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/drivers/xen/grant-table.c b/drivers/xen/grant-table.c index 076c1b0ab87fd1..f75b5cc71cfc5a 100644 --- a/drivers/xen/grant-table.c +++ b/drivers/xen/grant-table.c @@ -863,10 +863,10 @@ EXPORT_SYMBOL_GPL(gnttab_free_auto_xlat_frames); int gnttab_pages_set_private(int nr_pages, struct page **pages) { +#if BITS_PER_LONG < 64 int i; for (i = 0; i < nr_pages; i++) { -#if BITS_PER_LONG < 64 struct xen_page_foreign *foreign; foreign = kzalloc_obj(*foreign); @@ -874,9 +874,9 @@ int gnttab_pages_set_private(int nr_pages, struct page **pages) return -ENOMEM; set_page_private(pages[i], (unsigned long)foreign); -#endif - SetPagePrivate(pages[i]); } +#endif + /* Data is stored in page->private on 64-bit */ return 0; } @@ -1031,12 +1031,11 @@ void gnttab_pages_clear_private(int nr_pages, struct page **pages) int i; for (i = 0; i < nr_pages; i++) { - if (PagePrivate(pages[i])) { #if BITS_PER_LONG < 64 + if (page_private(pages[i])) kfree((void *)page_private(pages[i])); #endif - ClearPagePrivate(pages[i]); - } + set_page_private(pages[i], 0); } } EXPORT_SYMBOL_GPL(gnttab_pages_clear_private); From 319f4057d357836c7f47cbe071e2509d0bb504f0 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:00 -0400 Subject: [PATCH 0614/1012] fscrypt: stop setting PG_private on bounce page The pointer to a plain text folio is stored in bound_page->private and cannot be NULL until the bounce_page is freed, making PG_private redundant. It prepares for a future commit that remove PG_private. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-4-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: Eric Biggers Acked-by: Usama Arif Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Assisted-by: LLM Cc: "Theodore Y. Ts'o" Cc: Jaegeuk Kim --- fs/crypto/crypto.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/fs/crypto/crypto.c b/fs/crypto/crypto.c index 5286a124b0d982..aced5c50a46011 100644 --- a/fs/crypto/crypto.c +++ b/fs/crypto/crypto.c @@ -65,7 +65,6 @@ void fscrypt_free_bounce_page(struct page *bounce_page) if (!bounce_page) return; set_page_private(bounce_page, (unsigned long)NULL); - ClearPagePrivate(bounce_page); mempool_free(bounce_page, fscrypt_bounce_page_pool); } EXPORT_SYMBOL(fscrypt_free_bounce_page); @@ -210,7 +209,6 @@ struct page *fscrypt_encrypt_pagecache_blocks(struct folio *folio, return ERR_PTR(err); } } - SetPagePrivate(ciphertext_page); set_page_private(ciphertext_page, (unsigned long)folio); return ciphertext_page; } From 030bb09cd62517c1df7a01b2857d7f9522d0f334 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:01 -0400 Subject: [PATCH 0615/1012] mm/hugetlb: use direct assignment instead of folio_change_private() folio_change_private() should be used along with folio_attach_private() and folio_detach_private(), where adding and remove ->private content requires folio refcount change. add_hugetlb_folio() simply sets folio->private to NULL without refcount manipulation. Change it to direct assignment to avoid semantic confusion. It prepares for a future commit that remove PG_private. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-5-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: Usama Arif Reviewed-by: Gregory Price (Meta) Reviewed-by: Muchun Song Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Assisted-by: LLM Cc: Oscar Salvador --- mm/hugetlb.c | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 2003439ea13c6f..7b27c3c5c3e58e 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -1446,11 +1446,8 @@ void add_hugetlb_folio(struct hstate *h, struct folio *folio, } __folio_set_hugetlb(folio); - folio_change_private(folio, NULL); - /* - * We have to set hugetlb_vmemmap_optimized again as above - * folio_change_private(folio, NULL) cleared it. - */ + /* Clear all folio->private flags except hugetlb_vmemmap_optimized. */ + folio->private = NULL; folio_set_hugetlb_vmemmap_optimized(folio); arch_clear_hugetlb_flags(folio); From f755e623e6a9089c3474265dedf3c63f38a67c56 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:02 -0400 Subject: [PATCH 0616/1012] f2fs: stop using PG_private f2fs sets its PAGE_PRIVATE_* flags in page->private and checking page->private != NULL is equivalent to checking PG_private. Change PagePrivate() to page_private(). Meanwhile, in set_page_private_##name(), page->private is first set to 0/NULL before an PAGE_PRIVATE_* flag is set, but it can cause confusion when PG_private is removed and page->private != NULL is used instead. Change it to initialize page->private to PAGE_PRIVATE_NOT_POINTER instead and retain the original semantics. It prepares for a future commit that removes PG_private. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-6-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: Usama Arif Acked-by: Chao Yu Reviewed-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Jaegeuk Kim --- fs/f2fs/f2fs.h | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/fs/f2fs/f2fs.h b/fs/f2fs/f2fs.h index 9940a6cecf1a2c..2f7ab5888b078d 100644 --- a/fs/f2fs/f2fs.h +++ b/fs/f2fs/f2fs.h @@ -2691,7 +2691,7 @@ static inline bool folio_test_f2fs_##name(const struct folio *folio) \ } \ static inline bool page_private_##name(struct page *page) \ { \ - return PagePrivate(page) && \ + return page_private(page) && \ test_bit(PAGE_PRIVATE_NOT_POINTER, &page_private(page)) && \ test_bit(PAGE_PRIVATE_##flagname, &page_private(page)); \ } @@ -2710,9 +2710,9 @@ static inline void folio_set_f2fs_##name(struct folio *folio) \ } \ static inline void set_page_private_##name(struct page *page) \ { \ - if (!PagePrivate(page)) \ - attach_page_private(page, (void *)0); \ - set_bit(PAGE_PRIVATE_NOT_POINTER, &page_private(page)); \ + if (!page_private(page)) \ + attach_page_private(page, \ + (void *)BIT(PAGE_PRIVATE_NOT_POINTER)); \ set_bit(PAGE_PRIVATE_##flagname, &page_private(page)); \ } From daa261be002f5ef963fb85805292d4f026a7eebb Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:03 -0400 Subject: [PATCH 0617/1012] f2fs: convert the ->private flag helpers to folio-only page-based ->private flag helpers are used in the compression path, where large folios are not enabled. They can use folio versions with page_folio(). The two remaining users in data.c and segment.c can use fio->folio instead of fio->page (two are in a union). Drop page-based helpers after the conversion and rename PAGE_PRIVATE_{GET,SET,CLEAR}_FUNC() and the PAGE_PRIVATE_* flags to F2FS_FOLIO_PRIVATE_* to match. Convert the folio/page union from f2fs_io_info union to folio only, since no page user is left. The folio helpers do a plain read-modify-write where the page ones used set_bit()/clear_bit(). It is fine because the converted code either holds folio lock or, in f2fs_compress_write_end_io(), matches what the non-compressed code does in f2fs_write_end_bio(). Link: https://lore.kernel.org/20260920-remove-pg_private-v5-7-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Co-developed-by: David Hildenbrand (Arm) Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Suggested-by: Tal Zussman Reviewed-by: Tal Zussman Reviewed-by: Lance Yang Assisted-by: LLM Cc: Jaegeuk Kim Cc: Chao Yu --- fs/f2fs/compress.c | 35 ++++++++++------ fs/f2fs/data.c | 2 +- fs/f2fs/f2fs.h | 99 ++++++++++++++++++---------------------------- fs/f2fs/segment.c | 2 +- 4 files changed, 63 insertions(+), 75 deletions(-) diff --git a/fs/f2fs/compress.c b/fs/f2fs/compress.c index ce88092d9ce26d..09d9b8d0fdcce0 100644 --- a/fs/f2fs/compress.c +++ b/fs/f2fs/compress.c @@ -1064,13 +1064,15 @@ static void cancel_cluster_writeback(struct compress_ctx *cc, /* Cancel writeback and stay locked. */ for (i = 0; i < cc->cluster_size; i++) { + struct folio *folio = page_folio(cc->rpages[i]); + if (i < submitted) { inode_inc_dirty_pages(cc->inode); - lock_page(cc->rpages[i]); + folio_lock(folio); } - clear_page_private_gcing(cc->rpages[i]); - if (folio_test_writeback(page_folio(cc->rpages[i]))) - end_page_writeback(cc->rpages[i]); + folio_clear_f2fs_gcing(folio); + if (folio_test_writeback(folio)) + folio_end_writeback(folio); } } @@ -1078,11 +1080,15 @@ static void set_cluster_dirty(struct compress_ctx *cc) { int i; - for (i = 0; i < cc->cluster_size; i++) - if (cc->rpages[i]) { - set_page_dirty(cc->rpages[i]); - set_page_private_gcing(cc->rpages[i]); - } + for (i = 0; i < cc->cluster_size; i++) { + struct folio *folio; + + if (!cc->rpages[i]) + continue; + folio = page_folio(cc->rpages[i]); + folio_mark_dirty(folio); + folio_set_f2fs_gcing(folio); + } } static int prepare_compress_overwrite(struct compress_ctx *cc, @@ -1281,7 +1287,7 @@ static int f2fs_write_compressed_pages(struct compress_ctx *cc, .op = REQ_OP_WRITE, .op_flags = wbc_to_write_flags(wbc), .old_blkaddr = NEW_ADDR, - .page = NULL, + .folio = NULL, .encrypted_page = NULL, .compressed_page = NULL, .io_type = io_type, @@ -1370,7 +1376,7 @@ static int f2fs_write_compressed_pages(struct compress_ctx *cc, block_t blkaddr; blkaddr = f2fs_data_blkaddr(&dn); - fio.page = cc->rpages[i]; + fio.folio = page_folio(cc->rpages[i]); fio.old_blkaddr = blkaddr; /* cluster header */ @@ -1476,9 +1482,12 @@ void f2fs_compress_write_end_io(struct bio *bio, struct folio *folio) } for (i = 0; i < cic->nr_rpages; i++) { + struct folio *rfolio; + WARN_ON(!cic->rpages[i]); - clear_page_private_gcing(cic->rpages[i]); - end_page_writeback(cic->rpages[i]); + rfolio = page_folio(cic->rpages[i]); + folio_clear_f2fs_gcing(rfolio); + folio_end_writeback(rfolio); } page_array_free(sbi, cic->rpages, cic->nr_rpages); diff --git a/fs/f2fs/data.c b/fs/f2fs/data.c index 21f396ebe22ca9..ca8232a9095f89 100644 --- a/fs/f2fs/data.c +++ b/fs/f2fs/data.c @@ -2923,7 +2923,7 @@ bool f2fs_should_update_outplace(struct inode *inode, struct f2fs_io_info *fio) return true; if (fio) { - if (page_private_gcing(fio->page)) + if (folio_test_f2fs_gcing(fio->folio)) return true; if (unlikely(is_sbi_flag_set(sbi, SBI_CP_DISABLED) && f2fs_is_checkpointed_data(sbi, fio->old_blkaddr))) diff --git a/fs/f2fs/f2fs.h b/fs/f2fs/f2fs.h index 2f7ab5888b078d..85937de3d7016e 100644 --- a/fs/f2fs/f2fs.h +++ b/fs/f2fs/f2fs.h @@ -1357,10 +1357,7 @@ struct f2fs_io_info { blk_opf_t op_flags; /* req_flag_bits */ block_t new_blkaddr; /* new block address to be written */ block_t old_blkaddr; /* old block address before Cow */ - union { - struct page *page; /* page to be written */ - struct folio *folio; - }; + struct folio *folio; /* folio to be written */ struct page *encrypted_page; /* encrypted page */ struct page *compressed_page; /* compressed page */ struct list_head list; /* serialize IOs */ @@ -1613,27 +1610,27 @@ static inline void f2fs_set_bit(unsigned int nr, char *addr); static inline void f2fs_clear_bit(unsigned int nr, char *addr); /* - * Layout of f2fs page.private: + * Layout of f2fs folio->private: * * Layout A: lowest bit should be 1 * | bit0 = 1 | bit1 | bit2 | ... | bit MAX | private data .... | - * bit 0 PAGE_PRIVATE_NOT_POINTER - * bit 1 PAGE_PRIVATE_ONGOING_MIGRATION - * bit 2 PAGE_PRIVATE_INLINE_INODE - * bit 3 PAGE_PRIVATE_REF_RESOURCE - * bit 4 PAGE_PRIVATE_ATOMIC_WRITE + * bit 0 F2FS_FOLIO_PRIVATE_NOT_POINTER + * bit 1 F2FS_FOLIO_PRIVATE_ONGOING_MIGRATION + * bit 2 F2FS_FOLIO_PRIVATE_INLINE_INODE + * bit 3 F2FS_FOLIO_PRIVATE_REF_RESOURCE + * bit 4 F2FS_FOLIO_PRIVATE_ATOMIC_WRITE * bit 5- f2fs private data * * Layout B: lowest bit should be 0 - * page.private is a wrapped pointer. + * folio->private is a wrapped pointer. */ enum { - PAGE_PRIVATE_NOT_POINTER, /* private contains non-pointer data */ - PAGE_PRIVATE_ONGOING_MIGRATION, /* data page which is on-going migrating */ - PAGE_PRIVATE_INLINE_INODE, /* inode page contains inline data */ - PAGE_PRIVATE_REF_RESOURCE, /* dirty page has referenced resources */ - PAGE_PRIVATE_ATOMIC_WRITE, /* data page from atomic write path */ - PAGE_PRIVATE_MAX + F2FS_FOLIO_PRIVATE_NOT_POINTER, /* private contains non-pointer data */ + F2FS_FOLIO_PRIVATE_ONGOING_MIGRATION, /* data page which is on-going migrating */ + F2FS_FOLIO_PRIVATE_INLINE_INODE, /* inode page contains inline data */ + F2FS_FOLIO_PRIVATE_REF_RESOURCE, /* dirty page has referenced resources */ + F2FS_FOLIO_PRIVATE_ATOMIC_WRITE, /* data page from atomic write path */ + F2FS_FOLIO_PRIVATE_MAX }; /* For compression */ @@ -2681,86 +2678,68 @@ static inline int inc_valid_block_count(struct f2fs_sb_info *sbi, return -ENOSPC; } -#define PAGE_PRIVATE_GET_FUNC(name, flagname) \ +#define F2FS_FOLIO_PRIVATE_GET_FUNC(name, flagname) \ static inline bool folio_test_f2fs_##name(const struct folio *folio) \ { \ unsigned long priv = (unsigned long)folio->private; \ - unsigned long v = (1UL << PAGE_PRIVATE_NOT_POINTER) | \ - (1UL << PAGE_PRIVATE_##flagname); \ + unsigned long v = (1UL << F2FS_FOLIO_PRIVATE_NOT_POINTER) | \ + (1UL << F2FS_FOLIO_PRIVATE_##flagname); \ return (priv & v) == v; \ -} \ -static inline bool page_private_##name(struct page *page) \ -{ \ - return page_private(page) && \ - test_bit(PAGE_PRIVATE_NOT_POINTER, &page_private(page)) && \ - test_bit(PAGE_PRIVATE_##flagname, &page_private(page)); \ } -#define PAGE_PRIVATE_SET_FUNC(name, flagname) \ +#define F2FS_FOLIO_PRIVATE_SET_FUNC(name, flagname) \ static inline void folio_set_f2fs_##name(struct folio *folio) \ { \ - unsigned long v = (1UL << PAGE_PRIVATE_NOT_POINTER) | \ - (1UL << PAGE_PRIVATE_##flagname); \ + unsigned long v = (1UL << F2FS_FOLIO_PRIVATE_NOT_POINTER) | \ + (1UL << F2FS_FOLIO_PRIVATE_##flagname); \ if (!folio->private) \ folio_attach_private(folio, (void *)v); \ else { \ v |= (unsigned long)folio->private; \ folio->private = (void *)v; \ } \ -} \ -static inline void set_page_private_##name(struct page *page) \ -{ \ - if (!page_private(page)) \ - attach_page_private(page, \ - (void *)BIT(PAGE_PRIVATE_NOT_POINTER)); \ - set_bit(PAGE_PRIVATE_##flagname, &page_private(page)); \ } -#define PAGE_PRIVATE_CLEAR_FUNC(name, flagname) \ +#define F2FS_FOLIO_PRIVATE_CLEAR_FUNC(name, flagname) \ static inline void folio_clear_f2fs_##name(struct folio *folio) \ { \ unsigned long v = (unsigned long)folio->private; \ \ - v &= ~(1UL << PAGE_PRIVATE_##flagname); \ - if (v == (1UL << PAGE_PRIVATE_NOT_POINTER)) \ + v &= ~(1UL << F2FS_FOLIO_PRIVATE_##flagname); \ + if (v == (1UL << F2FS_FOLIO_PRIVATE_NOT_POINTER)) \ folio_detach_private(folio); \ else \ folio->private = (void *)v; \ -} \ -static inline void clear_page_private_##name(struct page *page) \ -{ \ - clear_bit(PAGE_PRIVATE_##flagname, &page_private(page)); \ - if (page_private(page) == BIT(PAGE_PRIVATE_NOT_POINTER)) \ - detach_page_private(page); \ } -PAGE_PRIVATE_GET_FUNC(nonpointer, NOT_POINTER); -PAGE_PRIVATE_GET_FUNC(inline, INLINE_INODE); -PAGE_PRIVATE_GET_FUNC(gcing, ONGOING_MIGRATION); -PAGE_PRIVATE_GET_FUNC(atomic, ATOMIC_WRITE); +F2FS_FOLIO_PRIVATE_GET_FUNC(nonpointer, NOT_POINTER); +F2FS_FOLIO_PRIVATE_GET_FUNC(inline, INLINE_INODE); +F2FS_FOLIO_PRIVATE_GET_FUNC(gcing, ONGOING_MIGRATION); +F2FS_FOLIO_PRIVATE_GET_FUNC(atomic, ATOMIC_WRITE); -PAGE_PRIVATE_SET_FUNC(reference, REF_RESOURCE); -PAGE_PRIVATE_SET_FUNC(inline, INLINE_INODE); -PAGE_PRIVATE_SET_FUNC(gcing, ONGOING_MIGRATION); -PAGE_PRIVATE_SET_FUNC(atomic, ATOMIC_WRITE); +F2FS_FOLIO_PRIVATE_SET_FUNC(reference, REF_RESOURCE); +F2FS_FOLIO_PRIVATE_SET_FUNC(inline, INLINE_INODE); +F2FS_FOLIO_PRIVATE_SET_FUNC(gcing, ONGOING_MIGRATION); +F2FS_FOLIO_PRIVATE_SET_FUNC(atomic, ATOMIC_WRITE); -PAGE_PRIVATE_CLEAR_FUNC(reference, REF_RESOURCE); -PAGE_PRIVATE_CLEAR_FUNC(inline, INLINE_INODE); -PAGE_PRIVATE_CLEAR_FUNC(gcing, ONGOING_MIGRATION); -PAGE_PRIVATE_CLEAR_FUNC(atomic, ATOMIC_WRITE); +F2FS_FOLIO_PRIVATE_CLEAR_FUNC(reference, REF_RESOURCE); +F2FS_FOLIO_PRIVATE_CLEAR_FUNC(inline, INLINE_INODE); +F2FS_FOLIO_PRIVATE_CLEAR_FUNC(gcing, ONGOING_MIGRATION); +F2FS_FOLIO_PRIVATE_CLEAR_FUNC(atomic, ATOMIC_WRITE); static inline unsigned long folio_get_f2fs_data(struct folio *folio) { unsigned long data = (unsigned long)folio->private; - if (!test_bit(PAGE_PRIVATE_NOT_POINTER, &data)) + if (!test_bit(F2FS_FOLIO_PRIVATE_NOT_POINTER, &data)) return 0; - return data >> PAGE_PRIVATE_MAX; + return data >> F2FS_FOLIO_PRIVATE_MAX; } static inline void folio_set_f2fs_data(struct folio *folio, unsigned long data) { - data = (1UL << PAGE_PRIVATE_NOT_POINTER) | (data << PAGE_PRIVATE_MAX); + data = (1UL << F2FS_FOLIO_PRIVATE_NOT_POINTER) | + (data << F2FS_FOLIO_PRIVATE_MAX); if (!folio_test_private(folio)) folio_attach_private(folio, (void *)data); diff --git a/fs/f2fs/segment.c b/fs/f2fs/segment.c index 63b712d3d599ed..8c156e1fd37d06 100644 --- a/fs/f2fs/segment.c +++ b/fs/f2fs/segment.c @@ -3803,7 +3803,7 @@ static int __get_segment_type_6(struct f2fs_io_info *fio) if (is_inode_flag_set(inode, FI_ALIGNED_WRITE)) return CURSEG_COLD_DATA_PINNED; - if (page_private_gcing(fio->page)) { + if (folio_test_f2fs_gcing(fio->folio)) { if (fio->sbi->am.atgc_enabled && (fio->io_type == FS_DATA_IO) && (fio->sbi->gc_mode != GC_URGENT_HIGH) && From 854217dc9fbfc97bcc30c0f7ce05186aa9f81f41 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:04 -0400 Subject: [PATCH 0618/1012] erofs: mm/pagemap: add readahead_folio_last() to avoid folio->private erofs needs to traverse readahead folios in reverse order to achieve maximum performance by 1. reading all folios from readahead_folio(); 2. storing the prior folio pointer in folio->private; 3. traverse from the last folio to the first one. Add readahead_folio_last() to achieve the same function without using folio->private. __readahead_advance() helper shares readahead_control adjustment code among __readahead_folio(), readahead_folio_last(), and __readahead_batch() by checking new private member, _forward, of readahead_control. It prepares for a future commit that replaces PG_private checks with !folio->private checks. After switching the checks, erofs's use of folio->private without bumping folio refcount can cause unexpected outcomes, e.g., in filemap_release_folio(), try_to_free_buffers() becomes reachable. No functional change intended. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-8-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Assisted-by: LLM Cc: Yue Hu Cc: Jeffle Xu Cc: Sandeep Dhavale Cc: Hongbo Li Cc: Chunhai Guo Cc: Gao Xiang Cc: Chao Yu Cc: "Matthew Wilcox (Oracle)" Cc: Jan Kara --- fs/erofs/zdata.c | 13 +++------ include/linux/pagemap.h | 60 +++++++++++++++++++++++++++++++++++------ 2 files changed, 55 insertions(+), 18 deletions(-) diff --git a/fs/erofs/zdata.c b/fs/erofs/zdata.c index 6b07e73ee2aaa0..e981e371d6c294 100644 --- a/fs/erofs/zdata.c +++ b/fs/erofs/zdata.c @@ -1886,21 +1886,14 @@ static void z_erofs_readahead(struct readahead_control *rac) struct inode *realinode = erofs_real_inode(sharedinode, &need_iput); Z_EROFS_DEFINE_FRONTEND(f, realinode, sharedinode, readahead_pos(rac)); unsigned int nrpages = readahead_count(rac); - struct folio *head = NULL, *folio; + struct folio *folio; int err; trace_erofs_readahead(realinode, readahead_index(rac), nrpages, false); z_erofs_pcluster_readmore(&f, rac, true); - while ((folio = readahead_folio(rac))) { - folio->private = head; - head = folio; - } - - /* traverse in reverse order for best metadata I/O performance */ - while (head) { - folio = head; - head = folio_get_private(folio); + /* traverse from last to first for best metadata I/O performance */ + while ((folio = readahead_folio_last(rac))) { err = z_erofs_scan_folio(&f, folio, true); if (err && err != -EINTR) erofs_err(realinode->i_sb, "readahead error at folio %lu @ nid %llu", diff --git a/include/linux/pagemap.h b/include/linux/pagemap.h index 939f3a5e973f6b..1e3462357aaa45 100644 --- a/include/linux/pagemap.h +++ b/include/linux/pagemap.h @@ -1415,6 +1415,7 @@ struct readahead_control { bool dropbehind; bool _workingset; unsigned long _pflags; + bool _forward; }; #define DEFINE_READAHEAD(ractl, f, r, m, i) \ @@ -1479,18 +1480,29 @@ void page_cache_async_readahead(struct address_space *mapping, page_cache_async_ra(&ractl, folio, req_count); } +/* + * Adjust readahead_control to ensure next folio comes from + * [_index, _index + _nr_pages) afterwards and reset _batch_count. + */ +static inline void __readahead_advance(struct readahead_control *rac) +{ + if (rac->_forward) + rac->_index += rac->_batch_count; + + rac->_nr_pages -= rac->_batch_count; + rac->_batch_count = 0; +} + static inline struct folio *__readahead_folio(struct readahead_control *ractl) { struct folio *folio; BUG_ON(ractl->_batch_count > ractl->_nr_pages); - ractl->_nr_pages -= ractl->_batch_count; - ractl->_index += ractl->_batch_count; + __readahead_advance(ractl); + ractl->_forward = true; - if (!ractl->_nr_pages) { - ractl->_batch_count = 0; + if (!ractl->_nr_pages) return NULL; - } folio = xa_load(&ractl->mapping->i_pages, ractl->_index); VM_BUG_ON_FOLIO(!folio_test_locked(folio), folio); @@ -1516,6 +1528,39 @@ static inline struct folio *readahead_folio(struct readahead_control *ractl) return folio; } +/** + * readahead_folio_last - Get the next folio to read, from the tail. + * @ractl: The current readahead request. + * + * Like readahead_folio(), but walks the range back-to-front. The folio is + * returned locked with its refcount dropped; the caller unlocks it once I/O + * completes. Compound folios are returned once, at their head index. + * + * Context: The folio is locked. + * Return: A pointer to the next folio, or %NULL when done. + */ +static inline struct folio *readahead_folio_last(struct readahead_control *ractl) +{ + struct folio *folio; + + /* Drop the previously returned batch from the remaining range. */ + __readahead_advance(ractl); + ractl->_forward = false; + + if (!ractl->_nr_pages) + return NULL; + + /* xa_load() follows sibling entries, so a tail index returns the head */ + folio = xa_load(&ractl->mapping->i_pages, + ractl->_index + ractl->_nr_pages - 1); + VM_WARN_ON_ONCE_FOLIO(!folio_test_locked(folio), folio); + + ractl->_batch_count = folio_nr_pages(folio); + + folio_put(folio); + return folio; +} + static inline unsigned int __readahead_batch(struct readahead_control *rac, struct page **array, unsigned int array_sz) { @@ -1524,9 +1569,8 @@ static inline unsigned int __readahead_batch(struct readahead_control *rac, struct folio *folio; BUG_ON(rac->_batch_count > rac->_nr_pages); - rac->_nr_pages -= rac->_batch_count; - rac->_index += rac->_batch_count; - rac->_batch_count = 0; + __readahead_advance(rac); + rac->_forward = true; xas_set(&xas, rac->_index); rcu_read_lock(); From d392570f29d323db096330993be9309c6e183f46 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:05 -0400 Subject: [PATCH 0619/1012] erofs: use folio_attach/detach_private() instead of direct assignment erofs_onlinefolio_init/split/end() use folio->private without setting PG_private or increasing folio refcount and it works. But after PG_private is replaced by checking folio->private in a future commit, it can break folio_expected_ref_count(), since the folio has private data without elevated refcount. Change them to use folio_attach/detach_private(). Folios during this process are locked as they are in the process of readahead, so no parallel migration/folio split can happen. Furthermore, because folio->private is used to store in-flight I/O counter and the counter reaches 0 when all I/O completes successfully without error or being dirty, ->private=0 causes folio_detach_private() to not drop the elevated folio refcount. Solve this issue by using bias=1 for the counter, so that ->private stays non NULL throughout every attach-to-detach process. Add a macro EROFS_ONLINEFOLIO_BIAS=1. While at it, fix the comment about ->private bit layout and add EROFS_ONLINEFOLIO_COUNT_MASK. It prepares for a future commit that removes PG_private. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-9-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reviewed-by: Gao Xiang Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Assisted-by: LLM Cc: Chao Yu Cc: Yue Hu Cc: Jeffle Xu Cc: Sandeep Dhavale Cc: Hongbo Li Cc: Chunhai Guo --- fs/erofs/data.c | 16 ++++++++++------ 1 file changed, 10 insertions(+), 6 deletions(-) diff --git a/fs/erofs/data.c b/fs/erofs/data.c index be63b89f086229..8b150aebf0af8c 100644 --- a/fs/erofs/data.c +++ b/fs/erofs/data.c @@ -239,19 +239,23 @@ int erofs_map_dev(struct super_block *sb, struct erofs_map_dev *map) /* * bit 30: I/O error occurred on this folio * bit 29: CPU has dirty data in D-cache (needs aliasing handling); - * bit 0 - 29: remaining parts to complete this folio + * bit 0 - 28: remaining parts to complete this folio, biased by 1 so that + * ->private stays non-NULL while the folio is attached */ #define EROFS_ONLINEFOLIO_EIO 30 #define EROFS_ONLINEFOLIO_DIRTY 29 +#define EROFS_ONLINEFOLIO_COUNT_MASK (BIT(EROFS_ONLINEFOLIO_DIRTY) - 1) +#define EROFS_ONLINEFOLIO_BIAS 1 void erofs_onlinefolio_init(struct folio *folio) { union { atomic_t o; void *v; - } u = { .o = ATOMIC_INIT(1) }; + } u = { .o = ATOMIC_INIT(1 + EROFS_ONLINEFOLIO_BIAS) }; - folio->private = u.v; /* valid only if file-backed folio is locked */ + /* valid only if file-backed folio is locked */ + folio_attach_private(folio, u.v); } void erofs_onlinefolio_split(struct folio *folio) @@ -265,14 +269,14 @@ void erofs_onlinefolio_end(struct folio *folio, int err, bool dirty) do { orig = atomic_read((atomic_t *)&folio->private); - DBG_BUGON(orig <= 0); + DBG_BUGON((orig & EROFS_ONLINEFOLIO_COUNT_MASK) <= EROFS_ONLINEFOLIO_BIAS); v = dirty << EROFS_ONLINEFOLIO_DIRTY; v |= (orig - 1) | (!!err << EROFS_ONLINEFOLIO_EIO); } while (atomic_cmpxchg((atomic_t *)&folio->private, orig, v) != orig); - if (v & (BIT(EROFS_ONLINEFOLIO_DIRTY) - 1)) + if ((v & EROFS_ONLINEFOLIO_COUNT_MASK) != EROFS_ONLINEFOLIO_BIAS) return; - folio->private = 0; + folio_detach_private(folio); if (v & BIT(EROFS_ONLINEFOLIO_DIRTY)) flush_dcache_folio(folio); folio_end_read(folio, !(v & BIT(EROFS_ONLINEFOLIO_EIO))); From a615dcf16b3ffe4a4b3d0a2365de6cd967ac3731 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:06 -0400 Subject: [PATCH 0620/1012] mm/page-flags: check page/folio->private instead of PG_private After the changes of the prior commits, page/folio->private != NULL is now equivalent to checking PG_private. Stop checking PG_private on pages and folios and use page/folio->private instead, except swapcache and hugetlb folios, because the former uses a field (swp_entry_t swap) overlapping with ->private and the latter sets its flags in ->private. Exclude swapcache and hugetlb when the code is meant to check PG_private only. PG_swapcache and folio->swap.val cannot be set/clear as a whole, so excluding swapcache with folio_test_swapcache() is not reliable. Instead, use folio_test_swapbacked(), since PG_swapbacked is stable when a folio is added to/removed from swapcache. Add a helper, folio_has_attached_private(), for this check. folio_test_private() and PagePrivate() now read folio/page->private plainly instead of an atomic read of PG_private bit, so KCSAN complains about possible data races. Annotate them with data_race(). folio_expected_ref_count() can be called without the folio lock, so annotate folio->mapping with data_race() while at it. folio_set/clear_private() and Set/ClearPagePrivate() become no-ops. PG_private is no longer checked at page free time. They will be removed in an upcoming commit. Remove KPF_PRIVATE since PG_private is no longer used. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-10-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Steven Rostedt Cc: Masami Hiramatsu Cc: Lorenzo Stoakes Cc: "Matthew Wilcox (Oracle)" Cc: Jan Kara Cc: Johannes Weiner Cc: "Liam R. Howlett" Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Mathieu Desnoyers Cc: Baolin Wang Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Matthew Brost Cc: Joshua Hahn Cc: Rakie Kim Cc: Byungchul Park Cc: Gregory Price Cc: Ying Huang Cc: Alistair Popple Cc: Qi Zheng Cc: Shakeel Butt Cc: Kairui Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu --- fs/proc/page.c | 1 - include/linux/kernel-page-flags.h | 1 - include/linux/mm.h | 19 +++++++---- include/linux/page-flags.h | 57 +++++++++++++++++++++++++++---- include/trace/events/pagemap.h | 2 +- mm/huge_memory.c | 2 +- mm/migrate.c | 2 +- mm/page-writeback.c | 2 +- mm/vmscan.c | 2 +- tools/mm/page-types.c | 2 -- 10 files changed, 68 insertions(+), 22 deletions(-) diff --git a/fs/proc/page.c b/fs/proc/page.c index 260772b20bd992..f90e1030825e94 100644 --- a/fs/proc/page.c +++ b/fs/proc/page.c @@ -232,7 +232,6 @@ u64 stable_page_flags(const struct page *page) u |= kpf_copy_bit(k, KPF_RESERVED, PG_reserved); u |= kpf_copy_bit(k, KPF_OWNER_2, PG_owner_2); - u |= kpf_copy_bit(k, KPF_PRIVATE, PG_private); u |= kpf_copy_bit(k, KPF_PRIVATE_2, PG_private_2); u |= kpf_copy_bit(k, KPF_OWNER_PRIVATE, PG_owner_priv_1); u |= kpf_copy_bit(k, KPF_ARCH, PG_arch_1); diff --git a/include/linux/kernel-page-flags.h b/include/linux/kernel-page-flags.h index 196778a087c4df..fe5ab6e50bd70e 100644 --- a/include/linux/kernel-page-flags.h +++ b/include/linux/kernel-page-flags.h @@ -11,7 +11,6 @@ #define KPF_RESERVED 32 #define KPF_MLOCKED 33 #define KPF_OWNER_2 34 -#define KPF_PRIVATE 35 #define KPF_PRIVATE_2 36 #define KPF_OWNER_PRIVATE 37 #define KPF_ARCH 38 diff --git a/include/linux/mm.h b/include/linux/mm.h index 30a3365bca8271..6c8df7715eb261 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -3005,9 +3005,9 @@ static inline bool folio_maybe_mapped_shared(struct folio *folio) * @folio: the folio * * Calculate the expected folio refcount, taking references from the pagecache, - * swapcache, PG_private and page table mappings into account. Useful in - * combination with folio_ref_count() to detect unexpected references (e.g., - * GUP or other temporary references). + * swapcache, private data (folio->private != NULL) and page table mappings into + * account. Useful in combination with folio_ref_count() to detect unexpected + * references (e.g., GUP or other temporary references). * * Does currently not consider references from the LRU cache. If the folio * was isolated from the LRU (which is the case during migration or split), @@ -3045,10 +3045,15 @@ static inline int folio_expected_ref_count(const struct folio *folio) ref_count += folio_test_swapcache(folio) << order; if (!folio_test_anon(folio)) { - /* One reference per page from the pagecache. */ - ref_count += !!folio->mapping << order; - /* One reference from PG_private. */ - ref_count += folio_test_private(folio); + /* + * One reference per page from the pagecache. + * Use data_race() since folio might not be locked. + */ + ref_count += !!data_race(folio->mapping) << order; + /* + * One reference from filesystem private data. + */ + ref_count += folio_has_attached_private(folio); } /* One reference per page table mapping. */ diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index 7080a6a1a79e72..6d839f50bdcb7e 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -575,9 +575,31 @@ FOLIO_FLAG(swapbacked, FOLIO_HEAD_PAGE) /* * Private page markings that may be used by the filesystem that owns the page * for its own purposes. - * - PG_private and PG_private_2 cause release_folio() and co to be invoked + * - folio->private and PG_private_2 cause release_folio() and co to be invoked */ -PAGEFLAG(Private, private, PF_ANY) + +static __always_inline bool folio_test_private(const struct folio *folio) +{ + /* + * data_race() is added for readers without holding the folio lock. + * Only the NULL/non-NULL answer is used and both are valid while + * private is being attached or detached, so the race is benign. + */ + return data_race(folio->private); +} + +static __always_inline int PagePrivate(const struct page *page) +{ + /* See folio_test_private() for data_race() use */ + return !!data_race(page->private); +} + +/* no-ops during transition */ +static __always_inline void folio_set_private(struct folio *folio) { } +static __always_inline void folio_clear_private(struct folio *folio) { } +static __always_inline void SetPagePrivate(struct page *page) { } +static __always_inline void ClearPagePrivate(struct page *page) { } + FOLIO_FLAG(private_2, FOLIO_HEAD_PAGE) /* owner_2 can be set on tail pages for anon memory */ @@ -1169,7 +1191,7 @@ static __always_inline void __ClearPageAnonExclusive(struct page *page) */ #define PAGE_FLAGS_CHECK_AT_FREE \ (1UL << PG_lru | 1UL << PG_locked | \ - 1UL << PG_private | 1UL << PG_private_2 | \ + 1UL << PG_private_2 | \ 1UL << PG_writeback | 1UL << PG_reserved | \ 1UL << PG_active | \ 1UL << PG_unevictable | __PG_MLOCKED | LRU_GEN_MASK) @@ -1193,8 +1215,31 @@ static __always_inline void __ClearPageAnonExclusive(struct page *page) (0xffUL /* order */ | 1UL << PG_has_hwpoisoned | \ 1UL << PG_large_rmappable | 1UL << PG_partially_mapped) -#define PAGE_FLAGS_PRIVATE \ - (1UL << PG_private | 1UL << PG_private_2) +/** + * folio_has_attached_private - check if the folio has private data attached + * @folio: The folio to check. + * + * Use this in code that may encounter swapcache or hugetlb folios but only + * wants to detect attached private data. + * + * Return: true if the folio has private data attached. + */ +static inline bool folio_has_attached_private(const struct folio *folio) +{ + /* + * Swapcache stores swp_entry_t in folio->swap, a union with + * folio->private, and hugetlb stores its own flags in folio->private; + * both are excluded. + * + * NOTE: For swapcache, folio->swap.val PG_swapcache are not set as + * a whole, so folio_test_swapcache() is not reliable to exclude + * swapcache. Use folio_test_swapbacked() instead, since it remains set + * when a folio is added to/removed from swapcache. + */ + + return folio_test_private(folio) && !folio_test_swapbacked(folio) && + !folio_test_hugetlb(folio); +} /** * folio_has_private - Determine if folio has private stuff * @folio: The folio to be checked @@ -1204,7 +1249,7 @@ static __always_inline void __ClearPageAnonExclusive(struct page *page) */ static inline int folio_has_private(const struct folio *folio) { - return !!(folio->flags.f & PAGE_FLAGS_PRIVATE); + return folio_has_attached_private(folio) || folio_test_private_2(folio); } #undef PF_ANY diff --git a/include/trace/events/pagemap.h b/include/trace/events/pagemap.h index 36c3a90f0accad..5d47b774633a4e 100644 --- a/include/trace/events/pagemap.h +++ b/include/trace/events/pagemap.h @@ -22,7 +22,7 @@ (folio_test_swapcache(folio) ? PAGEMAP_SWAPCACHE : 0) | \ (folio_test_swapbacked(folio) ? PAGEMAP_SWAPBACKED : 0) | \ (folio_test_mappedtodisk(folio) ? PAGEMAP_MAPPEDDISK : 0) | \ - (folio_test_private(folio) ? PAGEMAP_BUFFERS : 0) \ + (folio_has_attached_private(folio) ? PAGEMAP_BUFFERS : 0) \ ) TRACE_EVENT(mm_lru_insertion, diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 194188c292af3d..c822e831554b14 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4845,7 +4845,7 @@ static int split_huge_pages_pid(int pid, unsigned long vaddr_start, * will try to drop it before split and then check if the folio * can be split or not. So skip the check here. */ - if (!folio_test_private(folio) && + if (!folio_has_attached_private(folio) && folio_expected_ref_count(folio) != folio_ref_count(folio)) goto next; diff --git a/mm/migrate.c b/mm/migrate.c index a369d0c95c3860..b7b92925a28c30 100644 --- a/mm/migrate.c +++ b/mm/migrate.c @@ -1327,7 +1327,7 @@ static int migrate_folio_unmap(new_folio_t get_new_folio, * free the metadata, so the page can be freed. */ if (!src->mapping) { - if (folio_test_private(src)) { + if (folio_has_attached_private(src)) { try_to_free_buffers(src); goto out; } diff --git a/mm/page-writeback.c b/mm/page-writeback.c index eeab25d6ce3647..499a35473e4f31 100644 --- a/mm/page-writeback.c +++ b/mm/page-writeback.c @@ -2705,7 +2705,7 @@ bool filemap_dirty_folio(struct address_space *mapping, struct folio *folio) if (folio_test_set_dirty(folio)) return false; - __folio_mark_dirty(folio, mapping, !folio_test_private(folio)); + __folio_mark_dirty(folio, mapping, !folio_has_attached_private(folio)); if (mapping->host) { /* !PageAnon && !swapper_space */ diff --git a/mm/vmscan.c b/mm/vmscan.c index 80041e2b8049c7..dd6261c862794f 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -1029,7 +1029,7 @@ static void folio_check_dirty_writeback(struct folio *folio, *writeback = folio_test_writeback(folio); /* Verify dirty/writeback state if the filesystem supports it */ - if (!folio_test_private(folio)) + if (!folio_has_attached_private(folio)) return; mapping = folio_mapping(folio); diff --git a/tools/mm/page-types.c b/tools/mm/page-types.c index 7fc5a8be5997fb..47e4781c5fc38a 100644 --- a/tools/mm/page-types.c +++ b/tools/mm/page-types.c @@ -73,7 +73,6 @@ #define KPF_RESERVED 32 #define KPF_MLOCKED 33 #define KPF_OWNER_2 34 -#define KPF_PRIVATE 35 #define KPF_PRIVATE_2 36 #define KPF_OWNER_PRIVATE 37 #define KPF_ARCH 38 @@ -131,7 +130,6 @@ static const char * const page_flag_names[] = { [KPF_RESERVED] = "r:reserved", [KPF_MLOCKED] = "m:mlocked", [KPF_OWNER_2] = "d:owner_2", - [KPF_PRIVATE] = "P:private", [KPF_PRIVATE_2] = "p:private_2", [KPF_OWNER_PRIVATE] = "O:owner_private", [KPF_ARCH] = "h:arch", From 0367c08e684a1c42b7c4f1fff4885aa6f2ed34ff Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:07 -0400 Subject: [PATCH 0621/1012] treewide: remove folio_set/clear_private() usage They are no-ops now. Remove them. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-11-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Trond Myklebust Cc: Anna Schumaker Cc: "Matthew Wilcox (Oracle)" Cc: Jan Kara Cc: Matthew Brost Cc: Joshua Hahn Cc: Rakie Kim Cc: Byungchul Park Cc: Gregory Price Cc: Ying Huang Cc: Alistair Popple --- fs/nfs/write.c | 2 -- include/linux/pagemap.h | 4 +--- mm/migrate.c | 1 - 3 files changed, 1 insertion(+), 6 deletions(-) diff --git a/fs/nfs/write.c b/fs/nfs/write.c index 623e7ef1f73d57..b6967b5286691c 100644 --- a/fs/nfs/write.c +++ b/fs/nfs/write.c @@ -717,7 +717,6 @@ static void nfs_inode_add_request(struct nfs_page *req) nfs_lock_request(req); spin_lock(&mapping->i_private_lock); set_bit(PG_MAPPED, &req->wb_flags); - folio_set_private(folio); folio->private = req; spin_unlock(&mapping->i_private_lock); atomic_long_inc(&nfsi->nrequests); @@ -745,7 +744,6 @@ static void nfs_inode_remove_request(struct nfs_page *req) spin_lock(&mapping->i_private_lock); folio->private = NULL; - folio_clear_private(folio); clear_bit(PG_MAPPED, &req->wb_head->wb_flags); spin_unlock(&mapping->i_private_lock); diff --git a/include/linux/pagemap.h b/include/linux/pagemap.h index 1e3462357aaa45..bcbb0afe1a6816 100644 --- a/include/linux/pagemap.h +++ b/include/linux/pagemap.h @@ -594,7 +594,6 @@ static inline void folio_attach_private(struct folio *folio, void *data) { folio_get(folio); folio->private = data; - folio_set_private(folio); } /** @@ -629,9 +628,8 @@ static inline void *folio_detach_private(struct folio *folio) { void *data = folio_get_private(folio); - if (!folio_test_private(folio)) + if (!data) return NULL; - folio_clear_private(folio); folio->private = NULL; folio_put(folio); diff --git a/mm/migrate.c b/mm/migrate.c index b7b92925a28c30..7e3a81f0697442 100644 --- a/mm/migrate.c +++ b/mm/migrate.c @@ -835,7 +835,6 @@ void folio_migrate_flags(struct folio *newfolio, struct folio *folio) */ if (folio_test_swapcache(folio)) folio_clear_swapcache(folio); - folio_clear_private(folio); /* page->private contains hugetlb specific flags */ if (!folio_test_hugetlb(folio)) From ae35d77d616b2a3db63ed1e113a41c3e8c2d0706 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:08 -0400 Subject: [PATCH 0622/1012] ceph: replace PagePrivate() with page_private() PagePrivate() is going to be removed along with PG_private and its implementation is the same as page_private(). Link: https://lore.kernel.org/20260920-remove-pg_private-v5-12-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Ilya Dryomov Cc: Alex Markuze Cc: Viacheslav Dubeyko --- fs/ceph/addr.c | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/fs/ceph/addr.c b/fs/ceph/addr.c index e598b2d424ec16..0c00e9636b516f 100644 --- a/fs/ceph/addr.c +++ b/fs/ceph/addr.c @@ -70,9 +70,7 @@ static int ceph_netfs_check_write_begin(struct file *file, loff_t pos, unsigned static inline struct ceph_snap_context *page_snap_context(struct page *page) { - if (PagePrivate(page)) - return (void *)page->private; - return NULL; + return (void *)page_private(page); } /* @@ -124,8 +122,8 @@ static bool ceph_dirty_folio(struct address_space *mapping, struct folio *folio) spin_unlock(&ci->i_ceph_lock); /* - * Reference snap context in folio->private. Also set - * PagePrivate so that we get invalidate_folio callback. + * Reference snap context in folio->private. Setting folio->private is + * what gets us the invalidate_folio callback. */ VM_WARN_ON_FOLIO(folio->private, folio); folio_attach_private(folio, snapc); From 5b31ca56f0af4d67d659de62bd13f77cae9b6aba Mon Sep 17 00:00:00 2001 From: "Matthew Wilcox (Oracle)" Date: Sun, 20 Sep 2026 22:28:09 -0400 Subject: [PATCH 0623/1012] md: use folio_alloc_buffers() Remove the last user of alloc_page_buffers(). Use folio_alloc_buffers() instead, since alloc_page_buffers() is a wrap over it. Although the pages used in md-bitmap are not folios, as they are not mapped into userspace nor enter the page cache, but they still have buffer heads attached. Cleaning up the code to not use buffer heads is future work. [ziy@nvidia.com: reword commit message] Link: https://lore.kernel.org/20260920-remove-pg_private-v5-13-bb68b6a21869@nvidia.com Signed-off-by: Matthew Wilcox (Oracle) Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) --- drivers/md/md-bitmap.c | 5 +++-- fs/buffer.c | 8 -------- include/linux/buffer_head.h | 1 - 3 files changed, 3 insertions(+), 11 deletions(-) diff --git a/drivers/md/md-bitmap.c b/drivers/md/md-bitmap.c index b8325cb09a371f..5f1637f974c155 100644 --- a/drivers/md/md-bitmap.c +++ b/drivers/md/md-bitmap.c @@ -560,6 +560,7 @@ static int read_file_page(struct file *file, unsigned long index, { int ret = 0; struct inode *inode = file_inode(file); + struct folio *folio = page_folio(page); struct buffer_head *bh; sector_t block, blk_cur; unsigned long blocksize = i_blocksize(inode); @@ -567,12 +568,12 @@ static int read_file_page(struct file *file, unsigned long index, pr_debug("read bitmap file (%dB @ %llu)\n", (int)PAGE_SIZE, (unsigned long long)index << PAGE_SHIFT); - bh = alloc_page_buffers(page, blocksize); + bh = folio_alloc_buffers(folio, blocksize, GFP_NOFS | __GFP_ACCOUNT); if (!bh) { ret = -ENOMEM; goto out; } - attach_page_private(page, bh); + folio_attach_private(folio, bh); blk_cur = index << (PAGE_SHIFT - inode->i_blkbits); while (bh) { block = blk_cur; diff --git a/fs/buffer.c b/fs/buffer.c index ed966fa73b1ba2..020af5dbe2d05f 100644 --- a/fs/buffer.c +++ b/fs/buffer.c @@ -773,14 +773,6 @@ struct buffer_head *folio_alloc_buffers(struct folio *folio, unsigned long size, } EXPORT_SYMBOL_GPL(folio_alloc_buffers); -struct buffer_head *alloc_page_buffers(struct page *page, unsigned long size) -{ - gfp_t gfp = GFP_NOFS | __GFP_ACCOUNT; - - return folio_alloc_buffers(page_folio(page), size, gfp); -} -EXPORT_SYMBOL_GPL(alloc_page_buffers); - static inline void link_dev_buffers(struct folio *folio, struct buffer_head *head) { diff --git a/include/linux/buffer_head.h b/include/linux/buffer_head.h index fd2c7115c05427..6ce2db05c60f33 100644 --- a/include/linux/buffer_head.h +++ b/include/linux/buffer_head.h @@ -197,7 +197,6 @@ void folio_set_bh(struct buffer_head *bh, struct folio *folio, unsigned long offset); struct buffer_head *folio_alloc_buffers(struct folio *folio, unsigned long size, gfp_t gfp); -struct buffer_head *alloc_page_buffers(struct page *page, unsigned long size); struct buffer_head *create_empty_buffers(struct folio *folio, unsigned long blocksize, unsigned long b_state); void end_buffer_read_sync(struct buffer_head *bh, int uptodate); From 854311d845dd0f796f4d2fd8f2efe7cf71d84db0 Mon Sep 17 00:00:00 2001 From: "Matthew Wilcox (Oracle)" Date: Sun, 20 Sep 2026 22:28:10 -0400 Subject: [PATCH 0624/1012] md: use folio APIs in free_page() Convert the page to a folio. This removes some of the last uses of a few page APIs (detach_page_private(), page_buffers()). Replaces two calls to compound_head() with one. Remove the early return in free_buffers() to avoid memory leak and uninitialized file bitmap pages.[1][2] [ziy@nvidia.com: drop early return] Link: https://lore.kernel.org/20260920-remove-pg_private-v5-14-bb68b6a21869@nvidia.com Link: https://sashiko.dev/#/patchset/20260913-remove-pg_private-v4-0-848550f7574e%40nvidia.com?part=2 [1] Link: https://lore.kernel.org/all/aqgfA0QFi98gXYG2@casper.infradead.org/ [2] Signed-off-by: Matthew Wilcox (Oracle) Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) --- drivers/md/md-bitmap.c | 10 +++------- 1 file changed, 3 insertions(+), 7 deletions(-) diff --git a/drivers/md/md-bitmap.c b/drivers/md/md-bitmap.c index 5f1637f974c155..02a126968ad404 100644 --- a/drivers/md/md-bitmap.c +++ b/drivers/md/md-bitmap.c @@ -533,19 +533,15 @@ static void write_file_page(struct bitmap *bitmap, struct page *page, int wait) static void free_buffers(struct page *page) { - struct buffer_head *bh; - - if (!PagePrivate(page)) - return; + struct folio *folio = page_folio(page); + struct buffer_head *bh = folio_detach_private(folio); - bh = page_buffers(page); while (bh) { struct buffer_head *next = bh->b_this_page; free_buffer_head(bh); bh = next; } - detach_page_private(page); - put_page(page); + folio_put(folio); } /* read a page from a file. From e3840d522f5e669d90fa09bbe2b01c61740cd8c1 Mon Sep 17 00:00:00 2001 From: "Matthew Wilcox (Oracle)" Date: Sun, 20 Sep 2026 22:28:11 -0400 Subject: [PATCH 0625/1012] md: remove the last use of page_buffers() Convert the page to a folio and use folio_buffers() instead. This does introduce one extra call to compound_head(), but will simplify a later conversion of md-bitmap to use folios instead of pages. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-15-bb68b6a21869@nvidia.com Signed-off-by: Matthew Wilcox (Oracle) Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) --- drivers/md/md-bitmap.c | 3 ++- include/linux/buffer_head.h | 7 +------ 2 files changed, 3 insertions(+), 7 deletions(-) diff --git a/drivers/md/md-bitmap.c b/drivers/md/md-bitmap.c index 02a126968ad404..8adf6ddca3d605 100644 --- a/drivers/md/md-bitmap.c +++ b/drivers/md/md-bitmap.c @@ -516,7 +516,8 @@ static void end_bitmap_write(struct bio *bio) static void write_file_page(struct bitmap *bitmap, struct page *page, int wait) { - struct buffer_head *bh = page_buffers(page); + struct folio *folio = page_folio(page); + struct buffer_head *bh = folio_buffers(folio); while (bh && bh->b_blocknr) { atomic_inc(&bitmap->pending_writes); diff --git a/include/linux/buffer_head.h b/include/linux/buffer_head.h index 6ce2db05c60f33..4b0b7188472b2d 100644 --- a/include/linux/buffer_head.h +++ b/include/linux/buffer_head.h @@ -175,12 +175,7 @@ static inline unsigned long bh_offset(const struct buffer_head *bh) return (unsigned long)(bh)->b_data & (page_size(bh->b_page) - 1); } -/* If we *know* page->private refers to buffer_heads */ -#define page_buffers(page) \ - ({ \ - BUG_ON(!PagePrivate(page)); \ - ((struct buffer_head *)page_private(page)); \ - }) +/* If we *know* folio->private refers to buffer_heads */ #define folio_buffers(folio) folio_get_private(folio) void buffer_check_dirty_writeback(struct folio *folio, From 1216e42c072fb70b842fce32232bbe694c3e481c Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:12 -0400 Subject: [PATCH 0626/1012] treewide: remove PagePrivate() and PG_private from comments and docs PG_private and PagePrivate() are no longer used. Adjust related comments and documentations to refer to page/folio->private instead. hugetlbfs_reserv.rst is outdated and left unchanged. It should be rewritten. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-16-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Ilya Dryomov Cc: Alex Markuze Cc: Viacheslav Dubeyko Cc: Trond Myklebust Cc: Anna Schumaker Cc: Richard Weinberger Cc: Zhihao Cheng Cc: Lorenzo Stoakes Cc: "Liam R. Howlett" Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko --- Documentation/admin-guide/kdump/vmcoreinfo.rst | 2 +- Documentation/filesystems/vfs.rst | 6 +++--- fs/nfs/file.c | 4 ++-- fs/ubifs/file.c | 8 ++++---- include/linux/mm.h | 15 ++++++++------- include/linux/mm_types.h | 4 ++-- 6 files changed, 20 insertions(+), 19 deletions(-) diff --git a/Documentation/admin-guide/kdump/vmcoreinfo.rst b/Documentation/admin-guide/kdump/vmcoreinfo.rst index 7663c610fe9014..5f1df6d080508a 100644 --- a/Documentation/admin-guide/kdump/vmcoreinfo.rst +++ b/Documentation/admin-guide/kdump/vmcoreinfo.rst @@ -325,7 +325,7 @@ NR_FREE_PAGES On linux-2.6.21 or later, the number of free pages is in vm_stat[NR_FREE_PAGES]. Used to get the number of free pages. -PG_lru|PG_private|PG_swapcache|PG_swapbacked|PG_hwpoison|PG_head_mask +PG_lru|PG_swapcache|PG_swapbacked|PG_hwpoison|PG_head_mask -------------------------------------------------------------------------- Page attributes. These flags are used to filter various unnecessary for diff --git a/Documentation/filesystems/vfs.rst b/Documentation/filesystems/vfs.rst index d3a93eec3945f8..dec7816303c6a9 100644 --- a/Documentation/filesystems/vfs.rst +++ b/Documentation/filesystems/vfs.rst @@ -649,8 +649,8 @@ Writeback. The first can be used independently to the others. The VM can try to release clean pages in order to reuse them. To do this it can call -->release_folio on clean folios with the private -flag set. Clean pages without PagePrivate and with no external references +->release_folio on clean folios with folio->private set. Clean pages +without folio->private set and with no external references will be released without notice being given to the address_space. To achieve this functionality, pages need to be placed on an LRU with @@ -674,7 +674,7 @@ filemap_fdatawait_range, to wait for all writeback to complete. An address_space handler may attach extra information to a page, typically using the 'private' field in the 'struct page'. If such -information is attached, the PG_Private flag should be set. This will +information is attached, non-NULL 'private' field will cause various VM routines to make extra calls into the address_space handler to deal with that data. diff --git a/fs/nfs/file.c b/fs/nfs/file.c index e1bdd10b35f10d..38f830a6467c9b 100644 --- a/fs/nfs/file.c +++ b/fs/nfs/file.c @@ -484,7 +484,7 @@ static int nfs_write_end(const struct kiocb *iocb, * Partially or wholly invalidate a page * - Release the private state associated with a page if undergoing complete * page invalidation - * - Called if either PG_private or PG_fscache is set on the page + * - Called if either folio->private or PG_fscache is set on the page * - Caller holds page lock */ static void nfs_invalidate_folio(struct folio *folio, size_t offset, @@ -555,7 +555,7 @@ static void nfs_check_dirty_writeback(struct folio *folio, * Attempt to clear the private state associated with a page when an error * occurs that requires the cached contents of an inode to be written back or * destroyed - * - Called if either PG_private or fscache is set on the page + * - Called if either page->private or fscache is set on the page * - Caller holds page lock * - Return 0 if successful, -error otherwise */ diff --git a/fs/ubifs/file.c b/fs/ubifs/file.c index e73c28b12f97fd..aa0298ce451eff 100644 --- a/fs/ubifs/file.c +++ b/fs/ubifs/file.c @@ -12,14 +12,14 @@ * This file implements VFS file and inode operations for regular files, device * nodes and symlinks as well as address space operations. * - * UBIFS uses 2 page flags: @PG_private and @PG_checked. @PG_private is set if + * UBIFS uses folio->private and page flag @PG_checked. folio->private is set if * the page is dirty and is used for optimization purposes - dirty pages are - * not budgeted so the flag shows that 'ubifs_write_end()' should not release + * not budgeted so it shows that 'ubifs_write_end()' should not release * the budget for this page. The @PG_checked flag is set if full budgeting is * required for the page e.g., when it corresponds to a file hole or it is * beyond the file size. The budgeting is done in 'ubifs_write_begin()', because * it is OK to fail in this function, and the budget is released in - * 'ubifs_write_end()'. So the @PG_private and @PG_checked flags carry + * 'ubifs_write_end()'. So the folio->private and the @PG_checked flag carry * information about how the page was budgeted, to make it possible to release * the budget properly. * @@ -1509,7 +1509,7 @@ static vm_fault_t ubifs_vm_page_mkwrite(struct vm_fault *vmf) * * At the moment we do not know whether the folio is dirty or not, so we * assume that it is not and budget for a new folio. We could look at - * the @PG_private flag and figure this out, but we may race with write + * folio->private and figure this out, but we may race with write * back and the folio state may change by the time we lock it, so this * would need additional care. We do not bother with this at the * moment, although it might be good idea to do. Instead, we allocate diff --git a/include/linux/mm.h b/include/linux/mm.h index 6c8df7715eb261..0a2a7fc4a442fe 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -2049,20 +2049,21 @@ vm_fault_t finish_fault(struct vm_fault *vmf); * * A pagecache page contains an opaque `private' member, which belongs to the * page's address_space. Usually, this is the address of a circular list of - * the page's disk buffers. PG_private must be set to tell the VM to call - * into the filesystem to release these pages. + * the page's disk buffers. It tells the VM to call into the filesystem to + * release these pages. * * A folio may belong to an inode's memory mapping. In this case, * folio->mapping points to the inode, and folio->index is the file * offset of the folio, in units of PAGE_SIZE. * - * If pagecache pages are not associated with an inode, they are said to be - * anonymous pages. These may become associated with the swapcache, and in that - * case PG_swapcache is set, and page->private is an offset into the swapcache. + * If pagecache folios are not associated with an inode, they are said to be + * anonymous folios. These may become associated with the swapcache, and in that + * case PG_swapcache is set, and folio->private is an offset into the swapcache. * * In either case (swapcache or inode backed), the pagecache itself holds one - * reference to the page. Setting PG_private should also increment the - * refcount. The each user mapping also has a reference to the page. + * reference to the folio. Attaching filesystem private data via + * folio_attach_private() also increments the refcount. Each user mapping also + * has a reference to the folio. * * The pagecache pages are stored in a per-mapping radix tree, which is * rooted at mapping->i_pages, and indexed by offset. diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 2a3988178adfdc..0720a4e98286b2 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -108,7 +108,7 @@ struct page { }; /** * @private: Mapping-private opaque data. - * Usually used for buffer_heads if PagePrivate. + * Usually used for buffer_heads. * Used for swp_entry_t if swapcache flag set. * Indicates order in the buddy system if PageBuddy * or on pcp_llist. @@ -675,7 +675,7 @@ static inline void ptdesc_pmd_pts_init(struct ptdesc *ptdesc) #define STRUCT_PAGE_MAX_SHIFT (order_base_2(sizeof(struct page))) /* - * page_private can be used on tail pages. However, PagePrivate is only + * page_private can be used on tail pages. However, it is only * checked by the VM on the head page. So page_private on the tail pages * should be used for data that's ancillary to the head page (eg attaching * buffer heads to tail pages after attaching buffer heads to the head page) From a4f42419308c40a331811566b89eafc228c11219 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Sun, 20 Sep 2026 22:28:13 -0400 Subject: [PATCH 0627/1012] mm/page-flags: remove PG_private folio->private != NULL indicates a folio carries private data, replacing PG_private. All PG_private users are converted. Remove PG_private and reserve the space as PG_folio for future use. Unused PG_private functions are removed too. Link: https://lore.kernel.org/20260920-remove-pg_private-v5-17-bb68b6a21869@nvidia.com Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Baoquan He Cc: Mike Rapoport Cc: Pasha Tatashin Cc: Pratyush Yadav Cc: Jonathan Corbet Cc: "Matthew Wilcox (Oracle)" Cc: Jan Kara Cc: Steven Rostedt Cc: Masami Hiramatsu Cc: Dave Young Cc: Shuah Khan Cc: Lorenzo Stoakes Cc: "Liam R. Howlett" Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Mathieu Desnoyers --- include/linux/page-flags.h | 18 +----------------- include/trace/events/mmflags.h | 2 +- kernel/vmcore_info.c | 1 - 3 files changed, 2 insertions(+), 19 deletions(-) diff --git a/include/linux/page-flags.h b/include/linux/page-flags.h index 6d839f50bdcb7e..b0ddc652e76ccb 100644 --- a/include/linux/page-flags.h +++ b/include/linux/page-flags.h @@ -44,10 +44,6 @@ * Consequently, PG_reserved for a page mapped into user space can indicate * the zero page, the vDSO, MMIO pages or device memory. * - * The PG_private bitflag is set on pagecache pages if they contain filesystem - * specific data (which is normally at page->private). It can be used by - * private allocations for its own usage. - * * During initiation of disk I/O, PG_locked is set. This bit is set before I/O * and cleared when writeback _starts_ or when read _completes_. PG_writeback * is set before writeback starts and cleared when it finishes. @@ -105,7 +101,7 @@ enum pageflags { PG_owner_2, /* Owner use. If pagecache, fs may use */ PG_arch_1, PG_reserved, - PG_private, /* If pagecache, has fs-private data */ + PG_folio, /* Do not use: reserved for folio identification */ PG_private_2, /* If pagecache, has fs aux data */ PG_reclaim, /* To be reclaimed asap */ PG_swapbacked, /* Page is backed by RAM/swap */ @@ -588,18 +584,6 @@ static __always_inline bool folio_test_private(const struct folio *folio) return data_race(folio->private); } -static __always_inline int PagePrivate(const struct page *page) -{ - /* See folio_test_private() for data_race() use */ - return !!data_race(page->private); -} - -/* no-ops during transition */ -static __always_inline void folio_set_private(struct folio *folio) { } -static __always_inline void folio_clear_private(struct folio *folio) { } -static __always_inline void SetPagePrivate(struct page *page) { } -static __always_inline void ClearPagePrivate(struct page *page) { } - FOLIO_FLAG(private_2, FOLIO_HEAD_PAGE) /* owner_2 can be set on tail pages for anon memory */ diff --git a/include/trace/events/mmflags.h b/include/trace/events/mmflags.h index ef9aa388b84f7d..3c153b3ad84507 100644 --- a/include/trace/events/mmflags.h +++ b/include/trace/events/mmflags.h @@ -144,7 +144,7 @@ TRACE_DEFINE_ENUM(___GFP_LAST_BIT); DEF_PAGEFLAG_NAME(owner_2), \ DEF_PAGEFLAG_NAME(arch_1), \ DEF_PAGEFLAG_NAME(reserved), \ - DEF_PAGEFLAG_NAME(private), \ + DEF_PAGEFLAG_NAME(folio), \ DEF_PAGEFLAG_NAME(private_2), \ DEF_PAGEFLAG_NAME(writeback), \ DEF_PAGEFLAG_NAME(head), \ diff --git a/kernel/vmcore_info.c b/kernel/vmcore_info.c index 8614430ca212ae..5a417f8a922abd 100644 --- a/kernel/vmcore_info.c +++ b/kernel/vmcore_info.c @@ -216,7 +216,6 @@ static int __init crash_save_vmcoreinfo_init(void) VMCOREINFO_LENGTH(free_area.free_list, MIGRATE_TYPES); VMCOREINFO_NUMBER(NR_FREE_PAGES); VMCOREINFO_NUMBER(PG_lru); - VMCOREINFO_NUMBER(PG_private); VMCOREINFO_NUMBER(PG_swapcache); VMCOREINFO_NUMBER(PG_swapbacked); #define PAGE_SLAB_MAPCOUNT_VALUE (PGTY_slab << 24) From 82a68ec4248223ff53766a386ddba104b368dbfd Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:32 +0200 Subject: [PATCH 0628/1012] mm/swap: fix off-by-one in swap cache replace sanity check Patch series "mm/huge_memory: clean up and decouple the anon and file split helpers", v6. The folio split path handles anon, page cache and swap cache folios in one routine. That mixing is what makes the swap cache split restrictions hard to lift and to review. We now support uniform split to order-0 only, and no mappingless swap cache folios. And it has left a fair number of dead or redundant checks behind. This series prepares for lifting those restrictions by cleaning up the code first: split the routine into an anon and a file helper, and keep all swap cache handling in the anon helper. The file helper never sees a swap cache folio, folio_check_splittable() rejects them up front. Apart from two bug fixes (patch 1 and 2) and a slight adjustment of anon splitting (patch 12), this is a pure cleanup. Testing: The in-tree split_huge_page_test selftest (uniform, non-uniform and in-folio-offset splits of anon and pagecache folios) passes 62/62 over 600 runs on the patched kernel. ftrace function_graph tracing filtered on __folio_split() was used to compare per-call durations between the base and the patched kernel on the same x86-64 box (interleaved runs across alternating reboots. 135 split calls per run, 600 test runs): Before: 68.52 us, stddev: 1.58 After: 67.38 us, stddev: 1.33 The patched kernel is slightly faster. The stack usage and object size change as the config and compiler change, but in general the stack usage is reduced and object size is basically unchanged. This patch (of 17): The DEBUG_VM sanity check in __swap_cache_replace_folio() iterates the old folio's range with "while (ci_off++ < ci_end)", so the loop body runs on the already-incremented offset: the first entry is skipped and one entry past the range is read. For a folio split that entry belongs to the first after-split folio and was just repointed by the replacement loop above, so the check would warn spuriously whenever sub-folio orders differ from the head folio's. Currently we don't support non-uniform swapcache split, but this still needs a fix to clean it up and prepare for non-uniform swap cache split. Use the same do-while pattern as the replacement loop. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-0-ba1b4ba72c6f@tencent.com Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-1-ba1b4ba72c6f@tencent.com Fixes: 8578e0c00dcf ("mm, swap: use the swap table for the swap cache and switch API") Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Acked-by: Zi Yan Reviewed-by: Barry Song Acked-by: David Hildenbrand (Arm) Reviewed-by: Yeoreum Yun Acked-by: Kiryl Shutsemau (Meta) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/swap_state.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/swap_state.c b/mm/swap_state.c index 625c185a1ca4d5..cef44aadee6158 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -396,8 +396,9 @@ void __swap_cache_replace_folio(struct swap_cluster_info *ci, folio_order(old) != folio_order(new)) { ci_off = swp_cluster_offset(old->swap); ci_end = ci_off + folio_nr_pages(old); - while (ci_off++ < ci_end) + do { WARN_ON_ONCE(swp_tb_to_folio(__swap_table_get(ci, ci_off)) != old); + } while (++ci_off < ci_end); } } From 346054d5c5f501472bde6d5d898ac8c21db43313 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:33 +0200 Subject: [PATCH 0629/1012] mm/huge_memory: fix rejection of swap cache folios with a mapping A folio in the swap cache cannot be split if it has a mapping (shmem). The split code does a defensive check for this in __folio_freeze_and_split_unmapped, after the folio ref has been frozen and the NR_SHMEM_THPS/NR_FILE_THPS counters have been decremented. It rejects the split and returns -EINVAL without unfreezing the folio or restoring the counters. That error path is buggy, if it is ever taken. It leaves the folio frozen and stuck, skews the counters, and fires the VM_WARN_ON_ONCE_FOLIO for a state that is actually legitimate. Check for this case up front in folio_check_splittable and return -EBUSY before any state is modified, so the split routine always backs out cleanly. Also fix a bracket style issue that checkpatch.pl keeps complaining about. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-2-ba1b4ba72c6f@tencent.com Fixes: 00527733d0dc ("mm/huge_memory: add two new (not yet used) functions for folio_split()") Fixes: 714b056c8321 ("mm/huge_memory: convert VM_BUG* to VM_WARN* in __folio_split") Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Barry Song Acked-by: David Hildenbrand (Arm) Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 26 ++++++++++++++++---------- 1 file changed, 16 insertions(+), 10 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index c822e831554b14..7c55bc5e8da840 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3933,6 +3933,9 @@ static int __split_unmapped_folio(struct folio *folio, int new_order, int folio_check_splittable(struct folio *folio, unsigned int new_order, enum split_type split_type) { + const bool is_anon = folio_test_anon(folio); + const bool is_swapcache = folio_test_swapcache(folio); + VM_WARN_ON_FOLIO(!folio_test_locked(folio), folio); /* * Folios that just got truncated cannot get split. Signal to the @@ -3941,11 +3944,11 @@ int folio_check_splittable(struct folio *folio, unsigned int new_order, * TODO: this will also currently refuse folios without a mapping in the * swapcache (shmem or to-be-anon folios). */ - if (!folio->mapping && !folio_test_anon(folio)) + if (!folio->mapping && !is_anon) return -EBUSY; /* order-1 is not supported for anonymous THP. */ - if (folio_test_anon(folio) && new_order == 1) + if (is_anon && new_order == 1) return -EINVAL; /* @@ -3956,7 +3959,7 @@ int folio_check_splittable(struct folio *folio, unsigned int new_order, * swapcache folio split. Only uniform split to order-0 can be used * here. */ - if ((split_type == SPLIT_TYPE_NON_UNIFORM || new_order) && folio_test_swapcache(folio)) + if ((split_type == SPLIT_TYPE_NON_UNIFORM || new_order) && is_swapcache) return -EINVAL; if (is_huge_zero_folio(folio)) @@ -3965,6 +3968,15 @@ int folio_check_splittable(struct folio *folio, unsigned int new_order, if (folio_test_writeback(folio)) return -EBUSY; + /* + * A non-anon swapcache folio that still has a mapping can only be a + * shmem folio under SWAP IO, it's removed from either swap cache or + * shmem mapping afterward. There is little benefit in splitting them + * hence reject it here up front before touching anything. + */ + if (!is_anon && is_swapcache && folio->mapping) + return -EBUSY; + return 0; } @@ -4037,14 +4049,8 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n } } - if (folio_test_swapcache(folio)) { - if (mapping) { - VM_WARN_ON_ONCE_FOLIO(mapping, folio); - return -EINVAL; - } - + if (folio_test_swapcache(folio)) ci = swap_cluster_get_and_lock(folio); - } /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ if (do_lru) From e5a1b7aaee3ab2bd7f86e2bee9e4935ece1eb2e6 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:34 +0200 Subject: [PATCH 0630/1012] mm/huge_memory: invert folio_ref_freeze() check to reduce indentation Invert the folio_ref_freeze() success check in __folio_freeze_and_split_unmapped() to return early on failure, which removes one level of indentation from the entire success path. This is a pure refactoring with no functional change. It prepares the function to be split into separate helpers for anonymous and file-backed folios in a later patch. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-3-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Barry Song Acked-by: David Hildenbrand (Arm) Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 178 +++++++++++++++++++++++------------------------ 1 file changed, 88 insertions(+), 90 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 7c55bc5e8da840..9b2908271b6989 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4014,121 +4014,119 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n pgoff_t end, int *nr_shmem_dropped) { struct folio *end_folio = folio_next(folio); + struct swap_cluster_info *ci = NULL; struct folio *new_folio, *next; + struct lruvec *lruvec; int ret = 0; VM_WARN_ON_ONCE(!mapping && end); - if (folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { - struct swap_cluster_info *ci = NULL; - struct lruvec *lruvec; + if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) + return -EAGAIN; - /* Take off the deferred split queue while frozen and memcg set */ - folio_unqueue_deferred_split(folio); + /* Take off the deferred split queue while frozen and memcg set */ + folio_unqueue_deferred_split(folio); - /* - * deferred_split_scan() takes the folio off the queue before it - * splits it, so the unqueue above finds an empty list and - * leaves PG_partially_mapped set. - * Clear it here: the flag does not survive the split. - */ - folio_reset_partially_mapped(folio); + /* + * deferred_split_scan() takes the folio off the queue before it + * splits it, so the unqueue above finds an empty list and + * leaves PG_partially_mapped set. + * Clear it here: the flag does not survive the split. + */ + folio_reset_partially_mapped(folio); - if (mapping) { - int nr = folio_nr_pages(folio); - - if (folio_test_pmd_mappable(folio) && - new_order < HPAGE_PMD_ORDER) { - if (folio_test_swapbacked(folio)) { - lruvec_stat_mod_folio(folio, - NR_SHMEM_THPS, -nr); - } else { - lruvec_stat_mod_folio(folio, - NR_FILE_THPS, -nr); - } + if (mapping) { + int nr = folio_nr_pages(folio); + + if (folio_test_pmd_mappable(folio) && + new_order < HPAGE_PMD_ORDER) { + if (folio_test_swapbacked(folio)) { + lruvec_stat_mod_folio(folio, + NR_SHMEM_THPS, -nr); + } else { + lruvec_stat_mod_folio(folio, + NR_FILE_THPS, -nr); } } + } - if (folio_test_swapcache(folio)) - ci = swap_cluster_get_and_lock(folio); - - /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ - if (do_lru) - lruvec = folio_lruvec_lock(folio); + if (folio_test_swapcache(folio)) + ci = swap_cluster_get_and_lock(folio); - ret = __split_unmapped_folio(folio, new_order, split_at, xas, - mapping, split_type); + /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ + if (do_lru) + lruvec = folio_lruvec_lock(folio); - /* - * Unfreeze after-split folios and put them back to the right - * list. @folio should be kept frozon until page cache - * entries are updated with all the other after-split folios - * to prevent others seeing stale page cache entries. - * As a result, new_folio starts from the next folio of - * @folio. - */ - for (new_folio = folio_next(folio); new_folio != end_folio; - new_folio = next) { - unsigned long nr_pages = folio_nr_pages(new_folio); + ret = __split_unmapped_folio(folio, new_order, split_at, xas, + mapping, split_type); - next = folio_next(new_folio); + /* + * Unfreeze after-split folios and put them back to the right + * list. @folio should be kept frozon until page cache + * entries are updated with all the other after-split folios + * to prevent others seeing stale page cache entries. + * As a result, new_folio starts from the next folio of + * @folio. + */ + for (new_folio = folio_next(folio); new_folio != end_folio; + new_folio = next) { + unsigned long nr_pages = folio_nr_pages(new_folio); - zone_device_private_split_cb(folio, new_folio); + next = folio_next(new_folio); - folio_ref_unfreeze(new_folio, - folio_cache_ref_count(new_folio) + 1); + zone_device_private_split_cb(folio, new_folio); - if (do_lru) - lru_add_split_folio(folio, new_folio, lruvec, list); + folio_ref_unfreeze(new_folio, + folio_cache_ref_count(new_folio) + 1); - /* - * Anonymous folio with swap cache. - * NOTE: shmem in swap cache is not supported yet. - */ - if (ci) { - __swap_cache_replace_folio(ci, folio, new_folio); - continue; - } + if (do_lru) + lru_add_split_folio(folio, new_folio, lruvec, list); - /* Anonymous folio without swap cache */ - if (!mapping) - continue; + /* + * Anonymous folio with swap cache. + * NOTE: shmem in swap cache is not supported yet. + */ + if (ci) { + __swap_cache_replace_folio(ci, folio, new_folio); + continue; + } - /* Add the new folio to the page cache. */ - if (new_folio->index < end) { - __xa_store(&mapping->i_pages, new_folio->index, - new_folio, 0); - continue; - } + /* Anonymous folio without swap cache */ + if (!mapping) + continue; - VM_WARN_ON_ONCE(!nr_shmem_dropped); - /* Drop folio beyond EOF: ->index >= end */ - if (shmem_mapping(mapping) && nr_shmem_dropped) - *nr_shmem_dropped += nr_pages; - else if (folio_test_clear_dirty(new_folio)) - folio_account_cleaned( - new_folio, inode_to_wb(mapping->host)); - __filemap_remove_folio(new_folio, NULL); - folio_put_refs(new_folio, nr_pages); + /* Add the new folio to the page cache. */ + if (new_folio->index < end) { + __xa_store(&mapping->i_pages, new_folio->index, + new_folio, 0); + continue; } - zone_device_private_split_cb(folio, NULL); - /* - * Unfreeze @folio only after all page cache entries, which - * used to point to it, have been updated with new folios. - * Otherwise, a parallel folio_try_get() can grab @folio - * and its caller can see stale page cache entries. - */ - folio_ref_unfreeze(folio, folio_cache_ref_count(folio) + 1); + VM_WARN_ON_ONCE(!nr_shmem_dropped); + /* Drop folio beyond EOF: ->index >= end */ + if (shmem_mapping(mapping) && nr_shmem_dropped) + *nr_shmem_dropped += nr_pages; + else if (folio_test_clear_dirty(new_folio)) + folio_account_cleaned( + new_folio, inode_to_wb(mapping->host)); + __filemap_remove_folio(new_folio, NULL); + folio_put_refs(new_folio, nr_pages); + } - if (do_lru) - lruvec_unlock(lruvec); + zone_device_private_split_cb(folio, NULL); + /* + * Unfreeze @folio only after all page cache entries, which + * used to point to it, have been updated with new folios. + * Otherwise, a parallel folio_try_get() can grab @folio + * and its caller can see stale page cache entries. + */ + folio_ref_unfreeze(folio, folio_cache_ref_count(folio) + 1); - if (ci) - swap_cluster_unlock(ci); - } else { - return -EAGAIN; - } + if (do_lru) + lruvec_unlock(lruvec); + + if (ci) + swap_cluster_unlock(ci); return ret; } From 893525c3f2cacd00c2d385fd1746f715f4267f5c Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:35 +0200 Subject: [PATCH 0631/1012] mm/huge_memory: split the routine for splitting anon and file folio No functional change intended. Before adding more logic, split __folio_freeze_and_split_unmapped() into an anon and a file variant so each path can evolve independently. The two paths shared little beyond the folio freeze call, the LRU locking, and the unfreeze skeleton, but differed in all other per-folio bookkeeping and routines. While splitting, some cleanups become easy to apply, and helped drop a few now-redundant checks. The zone_device_private_split_cb() calls are only kept in the anon variant, as device private folios can only back anonymous memory, and add a VM_WARN_ON_ONCE_FOLIO() at the entry of the file variant. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-4-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 124 +++++++++++++++++++++++++++++------------------ 1 file changed, 78 insertions(+), 46 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 9b2908271b6989..a9057449d3a458 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4007,11 +4007,9 @@ static void folio_reset_partially_mapped(struct folio *folio) MTHP_STAT_NR_ANON_PARTIALLY_MAPPED, -1); } -static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int new_order, - struct page *split_at, struct xa_state *xas, - struct address_space *mapping, bool do_lru, - struct list_head *list, enum split_type split_type, - pgoff_t end, int *nr_shmem_dropped) +static int __folio_freeze_split_anon(struct folio *folio, + unsigned int new_order, struct page *split_at, bool do_lru, + struct list_head *list, enum split_type split_type) { struct folio *end_folio = folio_next(folio); struct swap_cluster_info *ci = NULL; @@ -4019,8 +4017,6 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n struct lruvec *lruvec; int ret = 0; - VM_WARN_ON_ONCE(!mapping && end); - if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) return -EAGAIN; @@ -4035,24 +4031,75 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n */ folio_reset_partially_mapped(folio); - if (mapping) { + if (folio_test_swapcache(folio)) + ci = swap_cluster_get_and_lock(folio); + + if (do_lru) + lruvec = folio_lruvec_lock(folio); + + ret = __split_unmapped_folio(folio, new_order, split_at, NULL, + NULL, split_type); + + /* + * Unfreeze the after-split folios and put them back to the right + * place. Keep the head @folio frozen until the end: sub entries + * in swap cache must be updated first, so a concurrent + * swap_cache_get_folio() cannot return the head folio for a sub + * entry (folio_try_get() will fail on the head @folio until unfreeze). + */ + for (new_folio = folio_next(folio); new_folio != end_folio; + new_folio = next) { + next = folio_next(new_folio); + zone_device_private_split_cb(folio, new_folio); + folio_ref_unfreeze(new_folio, + folio_cache_ref_count(new_folio) + 1); + if (do_lru) + lru_add_split_folio(folio, new_folio, lruvec, list); + if (ci) + __swap_cache_replace_folio(ci, folio, new_folio); + } + + zone_device_private_split_cb(folio, NULL); + folio_ref_unfreeze(folio, folio_cache_ref_count(folio) + 1); + + if (do_lru) + lruvec_unlock(lruvec); + if (ci) + swap_cluster_unlock(ci); + + return ret; +} + +static int __folio_freeze_split_file(struct folio *folio, + unsigned int new_order, struct page *split_at, + struct xa_state *xas, struct address_space *mapping, + bool do_lru, struct list_head *list, + enum split_type split_type, pgoff_t end, int *nr_shmem_dropped) +{ + struct folio *end_folio = folio_next(folio); + struct folio *new_folio, *next; + struct lruvec *lruvec; + int ret; + + /* Currently device private folios can only back anonymous memory. */ + VM_WARN_ON_ONCE_FOLIO(folio_is_device_private(folio), folio); + + if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) + return -EAGAIN; + + if (folio_test_pmd_mappable(folio) && + new_order < HPAGE_PMD_ORDER) { int nr = folio_nr_pages(folio); - if (folio_test_pmd_mappable(folio) && - new_order < HPAGE_PMD_ORDER) { - if (folio_test_swapbacked(folio)) { - lruvec_stat_mod_folio(folio, - NR_SHMEM_THPS, -nr); - } else { - lruvec_stat_mod_folio(folio, - NR_FILE_THPS, -nr); - } + if (folio_test_swapbacked(folio)) { + lruvec_stat_mod_folio(folio, + NR_SHMEM_THPS, -nr); + } else { + lruvec_stat_mod_folio(folio, + NR_FILE_THPS, -nr); } } - if (folio_test_swapcache(folio)) - ci = swap_cluster_get_and_lock(folio); - /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ if (do_lru) lruvec = folio_lruvec_lock(folio); @@ -4062,7 +4109,7 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n /* * Unfreeze after-split folios and put them back to the right - * list. @folio should be kept frozon until page cache + * list. @folio should be kept frozen until page cache * entries are updated with all the other after-split folios * to prevent others seeing stale page cache entries. * As a result, new_folio starts from the next folio of @@ -4072,29 +4119,15 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n new_folio = next) { unsigned long nr_pages = folio_nr_pages(new_folio); + /* compute next before the folio can be freed below */ next = folio_next(new_folio); - zone_device_private_split_cb(folio, new_folio); - folio_ref_unfreeze(new_folio, folio_cache_ref_count(new_folio) + 1); if (do_lru) lru_add_split_folio(folio, new_folio, lruvec, list); - /* - * Anonymous folio with swap cache. - * NOTE: shmem in swap cache is not supported yet. - */ - if (ci) { - __swap_cache_replace_folio(ci, folio, new_folio); - continue; - } - - /* Anonymous folio without swap cache */ - if (!mapping) - continue; - /* Add the new folio to the page cache. */ if (new_folio->index < end) { __xa_store(&mapping->i_pages, new_folio->index, @@ -4113,7 +4146,6 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n folio_put_refs(new_folio, nr_pages); } - zone_device_private_split_cb(folio, NULL); /* * Unfreeze @folio only after all page cache entries, which * used to point to it, have been updated with new folios. @@ -4125,9 +4157,6 @@ static int __folio_freeze_and_split_unmapped(struct folio *folio, unsigned int n if (do_lru) lruvec_unlock(lruvec); - if (ci) - swap_cluster_unlock(ci); - return ret; } @@ -4269,7 +4298,10 @@ static int __folio_split(struct folio *folio, unsigned int new_order, /* block interrupt reentry in xa_lock and spinlock */ local_irq_disable(); - if (mapping) { + if (is_anon) { + ret = __folio_freeze_split_anon(folio, new_order, split_at, + true, list, split_type); + } else { /* * Check if the folio is present in page cache. * We assume all tail are present too, if folio is there. @@ -4280,10 +4312,11 @@ static int __folio_split(struct folio *folio, unsigned int new_order, ret = -EAGAIN; goto fail; } + ret = __folio_freeze_split_file(folio, new_order, split_at, &xas, mapping, + true, list, split_type, end, + &nr_shmem_dropped); } - ret = __folio_freeze_and_split_unmapped(folio, new_order, split_at, &xas, mapping, - true, list, split_type, end, &nr_shmem_dropped); fail: if (mapping) xas_unlock(&xas); @@ -4383,9 +4416,8 @@ int folio_split_unmapped(struct folio *folio, unsigned int new_order) return -EAGAIN; local_irq_disable(); - ret = __folio_freeze_and_split_unmapped(folio, new_order, &folio->page, NULL, - NULL, false, NULL, SPLIT_TYPE_UNIFORM, - 0, NULL); + ret = __folio_freeze_split_anon(folio, new_order, &folio->page, + false, NULL, SPLIT_TYPE_UNIFORM); local_irq_enable(); return ret; } From f285eea002b5dc468089071bbc226b3cbd73ac27 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:36 +0200 Subject: [PATCH 0632/1012] mm/huge_memory: rename __split_unmapped_folio() to __split_frozen_folio() The helper splits a folio whose refcount is frozen: the frozen refcount is the state it relies on, while unmapping is arranged by the caller beforehand. The old name caused confusion and people may try to call the helper on non-frozen folios. Also add a VM_WARN_ON_ONCE_FOLIO(folio_mapped(folio)) to self document that frozen implies unmapped. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-5-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Suggested-by: Zi Yan Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 23 +++++++++++++---------- 1 file changed, 13 insertions(+), 10 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index a9057449d3a458..b66b855df6f832 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3810,8 +3810,8 @@ static void __split_folio_to_order(struct folio *folio, int old_order, } /** - * __split_unmapped_folio() - splits an unmapped @folio to lower order folios in - * two ways: uniform split or non-uniform split. + * __split_frozen_folio() - splits a frozen @folio to lower order folios + * in two ways: uniform split or non-uniform split. * @folio: the to-be-split folio * @new_order: the smallest order of the after split folios (since buddy * allocator like split generates folios with orders from @folio's @@ -3850,7 +3850,7 @@ static void __split_folio_to_order(struct folio *folio, int old_order, * Return: 0 - successful, <0 - failed (if -ENOMEM is returned, @folio might be * split but not to @new_order, the caller needs to check) */ -static int __split_unmapped_folio(struct folio *folio, int new_order, +static int __split_frozen_folio(struct folio *folio, int new_order, struct page *split_at, struct xa_state *xas, struct address_space *mapping, enum split_type split_type) { @@ -3860,6 +3860,9 @@ static int __split_unmapped_folio(struct folio *folio, int new_order, struct folio *old_folio = folio; int split_order; + /* Frozen implies unmapped, callers unmap before splitting. */ + VM_WARN_ON_ONCE_FOLIO(folio_mapped(folio), folio); + /* * split to new_order one order at a time. For uniform split, * folio is split to new_order directly. @@ -4037,8 +4040,8 @@ static int __folio_freeze_split_anon(struct folio *folio, if (do_lru) lruvec = folio_lruvec_lock(folio); - ret = __split_unmapped_folio(folio, new_order, split_at, NULL, - NULL, split_type); + ret = __split_frozen_folio(folio, new_order, split_at, NULL, + NULL, split_type); /* * Unfreeze the after-split folios and put them back to the right @@ -4104,8 +4107,8 @@ static int __folio_freeze_split_file(struct folio *folio, if (do_lru) lruvec = folio_lruvec_lock(folio); - ret = __split_unmapped_folio(folio, new_order, split_at, xas, - mapping, split_type); + ret = __split_frozen_folio(folio, new_order, split_at, xas, + mapping, split_type); /* * Unfreeze after-split folios and put them back to the right @@ -4169,9 +4172,9 @@ static int __folio_freeze_split_file(struct folio *folio, * @list: after-split folios will be put on it if non NULL * @split_type: perform uniform split or not (non-uniform split) * - * It calls __split_unmapped_folio() to perform uniform and non-uniform split. + * It calls __split_frozen_folio() to perform uniform and non-uniform split. * It is in charge of checking whether the split is supported or not and - * preparing @folio for __split_unmapped_folio(). + * preparing @folio for __split_frozen_folio(). * * After splitting, the after-split folio containing @lock_at remains locked * and others are unlocked: @@ -4274,7 +4277,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, i_mmap_lock_read(mapping); /* - *__split_unmapped_folio() may need to trim off pages beyond + * __split_frozen_folio() may need to trim off pages beyond * EOF: but on 32-bit, i_size_read() takes an irq-unsafe * seqlock, which cannot be nested inside the page tree lock. * So note end now: i_size itself may be changed at any moment, From 40bf0cd69964568f2d975c9d393bd85db3427797 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:37 +0200 Subject: [PATCH 0633/1012] mm/huge_memory: consolidate irq and locking for folio split Let each split helper handle its own locking instead of relying on the caller, so both helpers manage their own irq and locking state. This lets __folio_split() drop its local irq handling and fail label, preparing for further cleanup. The file path now uses xas_lock_irq() instead of local_irq_disable() with xas_lock(). The two are equivalent on non-RT, and TRANSPARENT_HUGEPAGE cannot be enabled on RT anyway. This conversion also buys consistency: every other place in mm/ that freezes a folio while it is still reachable through the page cache already takes the lock this way. This was actually the last plain xas_lock() on mapping->i_pages left in mm. If we are going to support RT, spinning on frozen folio refs could be a problem, but it already exists in many places and should be fixed generically. The anon helper keeps a single local_irq_disable() as before, because it has to cover several plain spinlocks at once. The dropped xas_reset() was a no-op as the xa_state is not walked before the xas_load() under the lock. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-6-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Kiryl Shutsemau (Meta) Reviewed-by: Yeoreum Yun Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 58 ++++++++++++++++++++++-------------------------- 1 file changed, 27 insertions(+), 31 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index b66b855df6f832..5460cf63a84fbb 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4020,8 +4020,12 @@ static int __folio_freeze_split_anon(struct folio *folio, struct lruvec *lruvec; int ret = 0; - if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) + local_irq_disable(); + + if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { + local_irq_enable(); return -EAGAIN; + } /* Take off the deferred split queue while frozen and memcg set */ folio_unqueue_deferred_split(folio); @@ -4069,6 +4073,7 @@ static int __folio_freeze_split_anon(struct folio *folio, lruvec_unlock(lruvec); if (ci) swap_cluster_unlock(ci); + local_irq_enable(); return ret; } @@ -4087,8 +4092,21 @@ static int __folio_freeze_split_file(struct folio *folio, /* Currently device private folios can only back anonymous memory. */ VM_WARN_ON_ONCE_FOLIO(folio_is_device_private(folio), folio); - if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) - return -EAGAIN; + xas_lock_irq(xas); + + /* + * Check if the folio is present in page cache. + * We assume all tail are present too, if folio is there. + */ + if (xas_load(xas) != folio) { + ret = -EAGAIN; + goto fail; + } + + if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { + ret = -EAGAIN; + goto fail; + } if (folio_test_pmd_mappable(folio) && new_order < HPAGE_PMD_ORDER) { @@ -4160,6 +4178,8 @@ static int __folio_freeze_split_file(struct folio *folio, if (do_lru) lruvec_unlock(lruvec); +fail: + xas_unlock_irq(xas); return ret; } @@ -4299,32 +4319,13 @@ static int __folio_split(struct folio *folio, unsigned int new_order, unmap_folio(folio); - /* block interrupt reentry in xa_lock and spinlock */ - local_irq_disable(); - if (is_anon) { + if (is_anon) ret = __folio_freeze_split_anon(folio, new_order, split_at, true, list, split_type); - } else { - /* - * Check if the folio is present in page cache. - * We assume all tail are present too, if folio is there. - */ - xas_lock(&xas); - xas_reset(&xas); - if (xas_load(&xas) != folio) { - ret = -EAGAIN; - goto fail; - } + else ret = __folio_freeze_split_file(folio, new_order, split_at, &xas, mapping, true, list, split_type, end, &nr_shmem_dropped); - } - -fail: - if (mapping) - xas_unlock(&xas); - - local_irq_enable(); if (nr_shmem_dropped) shmem_uncharge(mapping->host, nr_shmem_dropped); @@ -4408,8 +4409,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, */ int folio_split_unmapped(struct folio *folio, unsigned int new_order) { - int ret = 0; - VM_WARN_ON_ONCE_FOLIO(folio_mapped(folio), folio); VM_WARN_ON_ONCE_FOLIO(!folio_test_locked(folio), folio); VM_WARN_ON_ONCE_FOLIO(!folio_test_large(folio), folio); @@ -4418,11 +4417,8 @@ int folio_split_unmapped(struct folio *folio, unsigned int new_order) if (folio_expected_ref_count(folio) != folio_ref_count(folio) - 1) return -EAGAIN; - local_irq_disable(); - ret = __folio_freeze_split_anon(folio, new_order, &folio->page, - false, NULL, SPLIT_TYPE_UNIFORM); - local_irq_enable(); - return ret; + return __folio_freeze_split_anon(folio, new_order, &folio->page, + false, NULL, SPLIT_TYPE_UNIFORM); } /* From c628c138d48f63b014a7a3c91e7370f0b225b04e Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:38 +0200 Subject: [PATCH 0634/1012] mm/huge_memory: move EOF trimming into the file split helper Instead of receiving @end and @nr_shmem_dropped from the caller, the file split helper now computes the EOF boundary and trims pages beyond it itself, as this is only needed for file split. This drops the redundant parameter passing and sanity check. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-7-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Acked-by: David Hildenbrand (Arm) Reviewed-by: Kiryl Shutsemau (Meta) Reviewed-by: Yeoreum Yun Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 41 +++++++++++++++++++---------------------- 1 file changed, 19 insertions(+), 22 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 5460cf63a84fbb..8c621131ee0662 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4082,16 +4082,29 @@ static int __folio_freeze_split_file(struct folio *folio, unsigned int new_order, struct page *split_at, struct xa_state *xas, struct address_space *mapping, bool do_lru, struct list_head *list, - enum split_type split_type, pgoff_t end, int *nr_shmem_dropped) + enum split_type split_type) { struct folio *end_folio = folio_next(folio); struct folio *new_folio, *next; + int nr_shmem_dropped = 0; struct lruvec *lruvec; + pgoff_t end; int ret; /* Currently device private folios can only back anonymous memory. */ VM_WARN_ON_ONCE_FOLIO(folio_is_device_private(folio), folio); + /* + * The loop below may need to trim off pages beyond + * EOF: but on 32-bit, i_size_read() takes an irq-unsafe + * seqlock, which cannot be nested inside the page tree lock. + * So note end now: i_size itself may be changed at any moment, + * but folio lock is good enough to serialize the trimming. + */ + end = DIV_ROUND_UP(i_size_read(mapping->host), PAGE_SIZE); + if (shmem_mapping(mapping)) + end = shmem_fallocend(mapping->host, end); + xas_lock_irq(xas); /* @@ -4156,10 +4169,9 @@ static int __folio_freeze_split_file(struct folio *folio, continue; } - VM_WARN_ON_ONCE(!nr_shmem_dropped); /* Drop folio beyond EOF: ->index >= end */ - if (shmem_mapping(mapping) && nr_shmem_dropped) - *nr_shmem_dropped += nr_pages; + if (shmem_mapping(mapping)) + nr_shmem_dropped += nr_pages; else if (folio_test_clear_dirty(new_folio)) folio_account_cleaned( new_folio, inode_to_wb(mapping->host)); @@ -4180,6 +4192,8 @@ static int __folio_freeze_split_file(struct folio *folio, fail: xas_unlock_irq(xas); + if (nr_shmem_dropped) + shmem_uncharge(mapping->host, nr_shmem_dropped); return ret; } @@ -4216,9 +4230,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, struct anon_vma *anon_vma = NULL; int old_order = folio_order(folio); struct folio *new_folio, *next; - int nr_shmem_dropped = 0; enum ttu_flags ttu_flags = 0; - pgoff_t end = 0; int ret; VM_WARN_ON_ONCE_FOLIO(!folio_test_locked(folio), folio); @@ -4295,17 +4307,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, anon_vma = NULL; i_mmap_lock_read(mapping); - - /* - * __split_frozen_folio() may need to trim off pages beyond - * EOF: but on 32-bit, i_size_read() takes an irq-unsafe - * seqlock, which cannot be nested inside the page tree lock. - * So note end now: i_size itself may be changed at any moment, - * but folio lock is good enough to serialize the trimming. - */ - end = DIV_ROUND_UP(i_size_read(mapping->host), PAGE_SIZE); - if (shmem_mapping(mapping)) - end = shmem_fallocend(mapping->host, end); } /* @@ -4324,11 +4325,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, true, list, split_type); else ret = __folio_freeze_split_file(folio, new_order, split_at, &xas, mapping, - true, list, split_type, end, - &nr_shmem_dropped); - - if (nr_shmem_dropped) - shmem_uncharge(mapping->host, nr_shmem_dropped); + true, list, split_type); if (!ret && is_anon && !folio_is_device_private(folio)) ttu_flags = TTU_USE_SHARED_ZEROPAGE; From ba7687273dc447eded6a3450959616234b1d1ed8 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:39 +0200 Subject: [PATCH 0635/1012] mm/huge_memory: move unmap and remap into the split helpers To prepare for further cleanup, move the unmap/remap handling from __folio_split() into the split helpers. Only anon folios need to be remapped, so remap_page() is now only called for anon splits and the anon check in remap_page() is redundant and can be removed. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-8-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 32 ++++++++++++++++++-------------- 1 file changed, 18 insertions(+), 14 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 8c621131ee0662..8b041460d977f3 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3640,9 +3640,6 @@ static void remap_page(struct folio *folio, unsigned long nr, int flags) { int i = 0; - /* If unmap_folio() uses try_to_migrate() on file, remove this check */ - if (!folio_test_anon(folio)) - return; for (;;) { remove_migration_ptes(folio, folio, TTU_RMAP_LOCKED | flags); i += folio_nr_pages(folio); @@ -4016,15 +4013,23 @@ static int __folio_freeze_split_anon(struct folio *folio, { struct folio *end_folio = folio_next(folio); struct swap_cluster_info *ci = NULL; + const int old_order = folio_order(folio); struct folio *new_folio, *next; + enum ttu_flags ttu_flags = 0; struct lruvec *lruvec; + bool need_remap = false; int ret = 0; + if (folio_mapped(folio)) { + need_remap = true; + unmap_folio(folio); + } + local_irq_disable(); if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { - local_irq_enable(); - return -EAGAIN; + ret = -EAGAIN; + goto out_no_split; } /* Take off the deferred split queue while frozen and memcg set */ @@ -4073,7 +4078,13 @@ static int __folio_freeze_split_anon(struct folio *folio, lruvec_unlock(lruvec); if (ci) swap_cluster_unlock(ci); +out_no_split: local_irq_enable(); + if (need_remap) { + if (!ret && !folio_is_device_private(folio)) + ttu_flags = TTU_USE_SHARED_ZEROPAGE; + remap_page(folio, 1 << old_order, ttu_flags); + } return ret; } @@ -4105,6 +4116,8 @@ static int __folio_freeze_split_file(struct folio *folio, if (shmem_mapping(mapping)) end = shmem_fallocend(mapping->host, end); + unmap_folio(folio); + xas_lock_irq(xas); /* @@ -4189,7 +4202,6 @@ static int __folio_freeze_split_file(struct folio *folio, if (do_lru) lruvec_unlock(lruvec); - fail: xas_unlock_irq(xas); if (nr_shmem_dropped) @@ -4230,7 +4242,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, struct anon_vma *anon_vma = NULL; int old_order = folio_order(folio); struct folio *new_folio, *next; - enum ttu_flags ttu_flags = 0; int ret; VM_WARN_ON_ONCE_FOLIO(!folio_test_locked(folio), folio); @@ -4318,8 +4329,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, goto out_unlock; } - unmap_folio(folio); - if (is_anon) ret = __folio_freeze_split_anon(folio, new_order, split_at, true, list, split_type); @@ -4327,11 +4336,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, ret = __folio_freeze_split_file(folio, new_order, split_at, &xas, mapping, true, list, split_type); - if (!ret && is_anon && !folio_is_device_private(folio)) - ttu_flags = TTU_USE_SHARED_ZEROPAGE; - - remap_page(folio, 1 << old_order, ttu_flags); - /* * Drop the mapping while the inode is still pinned. @folio stays * locked and present in the page cache until the loop below, so From 5b22f5160d82fc4ef1a3ce94c3e7ab9e37fc059b Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:40 +0200 Subject: [PATCH 0636/1012] mm/huge_memory: rename remap_page() to remap_anon_folio() remap_page() now only has one caller, __folio_freeze_split_anon(), and is only ever called for anon folios: unmap_folio() currently leaves file folios unmapped after the split, so they need no remapping. Rename it to remap_anon_folio() to make that explicit, and add a VM_WARN_ON_FOLIO() documenting it. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-9-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Kiryl Shutsemau (Meta) Reviewed-by: Yeoreum Yun Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 8b041460d977f3..09de66b6b76e07 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3551,7 +3551,6 @@ static void unmap_folio(struct folio *folio) /* * Anon pages need migration entries to preserve them, but file * pages can simply be left unmapped, then faulted back on demand. - * If that is ever changed (perhaps for mlock), update remap_page(). */ if (folio_test_anon(folio)) try_to_migrate(folio, ttu_flags); @@ -3636,10 +3635,17 @@ bool unmap_huge_pmd_locked(struct vm_area_struct *vma, unsigned long addr, return __discard_anon_folio_pmd_locked(vma, addr, pmdp, folio); } -static void remap_page(struct folio *folio, unsigned long nr, int flags) +static void remap_anon_folio(struct folio *folio, unsigned long nr, int flags) { int i = 0; + /* + * unmap_folio() installs migration entries only for anon folios, + * so currently only anon folios need to be remapped. File folios + * stay unmapped after the split and are faulted back on demand. + */ + VM_WARN_ON_FOLIO(!folio_test_anon(folio), folio); + for (;;) { remove_migration_ptes(folio, folio, TTU_RMAP_LOCKED | flags); i += folio_nr_pages(folio); @@ -3723,7 +3729,7 @@ static void __split_folio_to_order(struct folio *folio, int old_order, * * Note that for mapped sub-pages of an anonymous THP, * PG_anon_exclusive has been cleared in unmap_folio() and is stored in - * the migration entry instead from where remap_page() will restore it. + * the migration entry instead from where remap_anon_folio() will restore it. * We can still have PG_anon_exclusive set on effectively unmapped and * unreferenced sub-pages of an anonymous THP: we can simply drop * PG_anon_exclusive (-> PG_mappedtodisk) for these here. @@ -4083,7 +4089,7 @@ static int __folio_freeze_split_anon(struct folio *folio, if (need_remap) { if (!ret && !folio_is_device_private(folio)) ttu_flags = TTU_USE_SHARED_ZEROPAGE; - remap_page(folio, 1 << old_order, ttu_flags); + remap_anon_folio(folio, 1 << old_order, ttu_flags); } return ret; From c9df995bec8c34b1feb220fd209b7e4fe1864478 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:41 +0200 Subject: [PATCH 0637/1012] mm/huge_memory: move the racy refcount check into unmap_folio() The check only exists to avoid the expensive PMD-splitting unmap of a folio that cannot be split anyway. Move it from __folio_split() and folio_split_unmapped() into one check in unmap_folio(), right before the PMD split. All split helpers get the same early check without repeating it. unmap_folio() now returns -EAGAIN if the check fails and the split helpers propagate the error. folio_split_unmapped() drops its own copy of the check: it works on already unmapped folios and the definitive folio_ref_freeze() in __folio_freeze_split_anon() still catches unexpected references. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-10-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Kiryl Shutsemau (Meta) Reviewed-by: Yeoreum Yun Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 34 ++++++++++++++++++---------------- 1 file changed, 18 insertions(+), 16 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 09de66b6b76e07..ad92c2b554a044 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3538,13 +3538,22 @@ void vma_adjust_trans_huge(struct vm_area_struct *vma, split_huge_pmd_if_needed(next, end); } -static void unmap_folio(struct folio *folio) +/* + * A return value of 0 does not mean that unmapping succeeded. It might + * still have failed, but remap_anon_folio() must be called afterwards, + * for anon folios. + */ +static int unmap_folio(struct folio *folio) { enum ttu_flags ttu_flags = TTU_RMAP_LOCKED | TTU_SYNC | TTU_BATCH_FLUSH; VM_BUG_ON_FOLIO(!folio_test_large(folio), folio); + /* Racy check if we can split the page, before we split PMDs */ + if (folio_expected_ref_count(folio) != folio_ref_count(folio) - 1) + return -EAGAIN; + if (folio_test_pmd_mappable(folio)) ttu_flags |= TTU_SPLIT_HUGE_PMD; @@ -3558,6 +3567,8 @@ static void unmap_folio(struct folio *folio) try_to_unmap(folio, ttu_flags | TTU_IGNORE_MLOCK); try_to_unmap_flush(); + + return 0; } static bool __discard_anon_folio_pmd_locked(struct vm_area_struct *vma, @@ -4028,7 +4039,9 @@ static int __folio_freeze_split_anon(struct folio *folio, if (folio_mapped(folio)) { need_remap = true; - unmap_folio(folio); + ret = unmap_folio(folio); + if (ret) + return ret; } local_irq_disable(); @@ -4122,7 +4135,9 @@ static int __folio_freeze_split_file(struct folio *folio, if (shmem_mapping(mapping)) end = shmem_fallocend(mapping->host, end); - unmap_folio(folio); + ret = unmap_folio(folio); + if (ret) + return ret; xas_lock_irq(xas); @@ -4326,15 +4341,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, i_mmap_lock_read(mapping); } - /* - * Racy check if we can split the page, before unmap_folio() will - * split PMDs - */ - if (folio_expected_ref_count(folio) != folio_ref_count(folio) - 1) { - ret = -EAGAIN; - goto out_unlock; - } - if (is_anon) ret = __folio_freeze_split_anon(folio, new_order, split_at, true, list, split_type); @@ -4373,7 +4379,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, free_folio_and_swap_cache(new_folio); } -out_unlock: if (anon_vma) { anon_vma_unlock_write(anon_vma); put_anon_vma(anon_vma); @@ -4421,9 +4426,6 @@ int folio_split_unmapped(struct folio *folio, unsigned int new_order) VM_WARN_ON_ONCE_FOLIO(!folio_test_large(folio), folio); VM_WARN_ON_ONCE_FOLIO(!folio_test_anon(folio), folio); - if (folio_expected_ref_count(folio) != folio_ref_count(folio) - 1) - return -EAGAIN; - return __folio_freeze_split_anon(folio, new_order, &folio->page, false, NULL, SPLIT_TYPE_UNIFORM); } From 4bc751b06b3f6991e1420a0562e194deb63c7b41 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:42 +0200 Subject: [PATCH 0638/1012] mm/huge_memory: move filemap management into the file split helper Only file split needs the filemap and xarray handling and related variables. Move them out of __folio_split() into the file helper so the helper is self-contained, and simplify the parameters. No functional change. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-11-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Kiryl Shutsemau (Meta) Reviewed-by: Yeoreum Yun Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 102 ++++++++++++++++++++--------------------------- 1 file changed, 44 insertions(+), 58 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index ad92c2b554a044..6457192747aa6e 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4110,16 +4110,42 @@ static int __folio_freeze_split_anon(struct folio *folio, static int __folio_freeze_split_file(struct folio *folio, unsigned int new_order, struct page *split_at, - struct xa_state *xas, struct address_space *mapping, bool do_lru, struct list_head *list, enum split_type split_type) { + struct address_space *mapping = folio->mapping; + XA_STATE(xas, &mapping->i_pages, folio->index); struct folio *end_folio = folio_next(folio); struct folio *new_folio, *next; int nr_shmem_dropped = 0; + unsigned int min_order; struct lruvec *lruvec; pgoff_t end; - int ret; + gfp_t gfp; + int ret = 0; + + min_order = mapping_min_folio_order(mapping); + if (new_order < min_order) + return -EINVAL; + + gfp = current_gfp_context(mapping_gfp_mask(mapping) & GFP_RECLAIM_MASK); + if (!filemap_release_folio(folio, gfp)) + return -EBUSY; + + mapping_set_update(&xas, mapping); + + if (split_type == SPLIT_TYPE_UNIFORM) { + const int old_order = folio_order(folio); + + xas_set_order(&xas, folio->index, new_order); + xas_split_alloc(&xas, folio, old_order, gfp); + if (xas_error(&xas)) { + ret = xas_error(&xas); + goto fail_free; + } + } + + i_mmap_lock_read(mapping); /* Currently device private folios can only back anonymous memory. */ VM_WARN_ON_ONCE_FOLIO(folio_is_device_private(folio), folio); @@ -4137,15 +4163,15 @@ static int __folio_freeze_split_file(struct folio *folio, ret = unmap_folio(folio); if (ret) - return ret; + goto fail_mmap_unlock; - xas_lock_irq(xas); + xas_lock_irq(&xas); /* * Check if the folio is present in page cache. * We assume all tail are present too, if folio is there. */ - if (xas_load(xas) != folio) { + if (xas_load(&xas) != folio) { ret = -EAGAIN; goto fail; } @@ -4172,7 +4198,7 @@ static int __folio_freeze_split_file(struct folio *folio, if (do_lru) lruvec = folio_lruvec_lock(folio); - ret = __split_frozen_folio(folio, new_order, split_at, xas, + ret = __split_frozen_folio(folio, new_order, split_at, &xas, mapping, split_type); /* @@ -4224,9 +4250,19 @@ static int __folio_freeze_split_file(struct folio *folio, if (do_lru) lruvec_unlock(lruvec); fail: - xas_unlock_irq(xas); + xas_unlock_irq(&xas); +fail_mmap_unlock: if (nr_shmem_dropped) shmem_uncharge(mapping->host, nr_shmem_dropped); + /* + * Drop the mapping while the inode is still pinned. @folio stays + * locked and present in the page cache, so eviction cannot free + * the inode yet, nothing past this point may touch the inode or + * the mapping. + */ + i_mmap_unlock_read(mapping); +fail_free: + xas_destroy(&xas); return ret; } @@ -4255,11 +4291,9 @@ static int __folio_split(struct folio *folio, unsigned int new_order, struct page *split_at, struct page *lock_at, struct list_head *list, enum split_type split_type) { - XA_STATE(xas, &folio->mapping->i_pages, folio->index); struct folio *end_folio = folio_next(folio); bool is_anon = folio_test_anon(folio); struct mem_cgroup *memcg, *old_memcg; - struct address_space *mapping = NULL; struct anon_vma *anon_vma = NULL; int old_order = folio_order(folio); struct folio *new_folio, *next; @@ -4306,60 +4340,15 @@ static int __folio_split(struct folio *folio, unsigned int new_order, goto out; } anon_vma_lock_write(anon_vma); - mapping = NULL; - } else { - unsigned int min_order; - gfp_t gfp; - - mapping = folio->mapping; - min_order = mapping_min_folio_order(mapping); - if (new_order < min_order) { - ret = -EINVAL; - goto out; - } - - gfp = current_gfp_context(mapping_gfp_mask(mapping) & - GFP_RECLAIM_MASK); - - if (!filemap_release_folio(folio, gfp)) { - ret = -EBUSY; - goto out; - } - - mapping_set_update(&xas, mapping); - - if (split_type == SPLIT_TYPE_UNIFORM) { - xas_set_order(&xas, folio->index, new_order); - xas_split_alloc(&xas, folio, old_order, gfp); - if (xas_error(&xas)) { - ret = xas_error(&xas); - goto out; - } - } - - anon_vma = NULL; - i_mmap_lock_read(mapping); } if (is_anon) ret = __folio_freeze_split_anon(folio, new_order, split_at, true, list, split_type); else - ret = __folio_freeze_split_file(folio, new_order, split_at, &xas, mapping, + ret = __folio_freeze_split_file(folio, new_order, split_at, true, list, split_type); - /* - * Drop the mapping while the inode is still pinned. @folio stays - * locked and present in the page cache until the loop below, so - * eviction cannot free the inode yet; @lock_at is not enough, it may - * be a tail beyond EOF that the split already dropped from the page - * cache. Nothing past this point may touch the inode or the mapping. - */ - if (mapping) { - i_mmap_unlock_read(mapping); - mapping = NULL; - } - /* * Unlock all after-split folios except the one containing * @lock_at page. If @folio is not split, it will be kept locked. @@ -4383,14 +4372,11 @@ static int __folio_split(struct folio *folio, unsigned int new_order, anon_vma_unlock_write(anon_vma); put_anon_vma(anon_vma); } - if (mapping) - i_mmap_unlock_read(mapping); out: /* restore to caller's old_memcg */ set_active_memcg(old_memcg); mem_cgroup_put(memcg); out_no_memcg: - xas_destroy(&xas); if (is_pmd_order(old_order)) count_vm_event(!ret ? THP_SPLIT_PAGE : THP_SPLIT_PAGE_FAILED); count_mthp_stat(old_order, !ret ? MTHP_STAT_SPLIT : MTHP_STAT_SPLIT_FAILED); From 4969ec1a1e53702cc1047e82761cc2489cacf3c6 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:43 +0200 Subject: [PATCH 0639/1012] mm/huge_memory: move anon_vma handling into the anon split helper Only anon split needs the anon_vma, and it only needs it to unmap and remap. Move the folio_get_anon_vma()/anon_vma_lock_write() pair out of __folio_split() into the anon helper next to the folio_mapped() check that already gates unmap_folio(). This makes the anon_vma conditional on folio_mapped(), which is a behaviour change but should be fine. folio_get_anon_vma() returns NULL whenever !folio_mapped(), so an anon folio with folio_mapcount() == 0 used to get -EBUSY from split_huge_page() and is now split instead. A realistic case is a THP that has been fully swapped out and is still in the swap cache: swap PTEs do not contribute mapcount, so it is !folio_mapped() but still alive. That should be safe and right to have because: - folio_ref_freeze() below still rejects a folio that picked up any reference, a mapping or a GUP pin, in the meantime. - A parallel split is excluded by the folio lock. The anon_vma write lock was added to serialize split in commit 062f1af2170a ("mm: thp: acquire the anon_vma rwsem for write during split"), when split_huge_page() did not hold the folio lock throughout. commit e9b61f19858a ("thp: reintroduce split_huge_page()") later made the folio lock a caller requirement and added the folio_ref_freeze() scheme, so that has been covered ever since. - Unmapped path is already exercised by folio_split_unmapped(), and the swap cache split already runs well for a partially swapped-out mapped THP. - A !folio_mapped() folio cannot become mapped meanwhile: mapping it requires the folio lock. For mapped folios the anon_vma write lock is now released before __folio_split() unlocks the after-split sub-folios, where previously it was held across that loop; that window is harmless as the sub-folios stay folio-locked and referenced until it. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-12-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Kiryl Shutsemau Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 57 +++++++++++++++++++++++++----------------------- 1 file changed, 30 insertions(+), 27 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 6457192747aa6e..ec955488d280f9 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4032,16 +4032,37 @@ static int __folio_freeze_split_anon(struct folio *folio, struct swap_cluster_info *ci = NULL; const int old_order = folio_order(folio); struct folio *new_folio, *next; + struct anon_vma *anon_vma = NULL; enum ttu_flags ttu_flags = 0; struct lruvec *lruvec; - bool need_remap = false; int ret = 0; + /* + * Unmap/remap needs the anon_vma, so we first take a reference on + * it to prevent it from disappearing, and lock it for write here, + * letting unmap_folio() walk the rmap with TTU_RMAP_LOCKED. + * + * folio_mapped() is not stable here, but it can only change in + * one direction while the folio is locked. The mapcount can drop + * to zero at any time, zap_pte_range() takes no folio lock. It + * cannot go up: swapin, migration and uffd move all lock the folio + * before mapping it, and fork only copies PTEs that already exist. + * + * So if we see the folio mapped, the worst case is an empty rmap + * walk. If we see it unmapped, it stays unmapped and needs neither + * the reference nor the lock. Anything else needs a reference + * first and folio_ref_freeze() below catches it. + * + * Note that entirely swapped-out THPs are unmapped but can be split. + */ if (folio_mapped(folio)) { - need_remap = true; + anon_vma = folio_get_anon_vma(folio); + if (!anon_vma) + return -EBUSY; + anon_vma_lock_write(anon_vma); ret = unmap_folio(folio); if (ret) - return ret; + goto out_unlock; } local_irq_disable(); @@ -4099,11 +4120,16 @@ static int __folio_freeze_split_anon(struct folio *folio, swap_cluster_unlock(ci); out_no_split: local_irq_enable(); - if (need_remap) { + if (anon_vma) { if (!ret && !folio_is_device_private(folio)) ttu_flags = TTU_USE_SHARED_ZEROPAGE; remap_anon_folio(folio, 1 << old_order, ttu_flags); } +out_unlock: + if (anon_vma) { + anon_vma_unlock_write(anon_vma); + put_anon_vma(anon_vma); + } return ret; } @@ -4294,7 +4320,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, struct folio *end_folio = folio_next(folio); bool is_anon = folio_test_anon(folio); struct mem_cgroup *memcg, *old_memcg; - struct anon_vma *anon_vma = NULL; int old_order = folio_order(folio); struct folio *new_folio, *next; int ret; @@ -4325,23 +4350,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, memcg = get_mem_cgroup_from_folio(folio); old_memcg = set_active_memcg(memcg); - if (is_anon) { - /* - * The caller does not necessarily hold an mmap_lock that would - * prevent the anon_vma disappearing so we first we take a - * reference to it and then lock the anon_vma for write. This - * is similar to folio_lock_anon_vma_read except the write lock - * is taken to serialise against parallel split or collapse - * operations. - */ - anon_vma = folio_get_anon_vma(folio); - if (!anon_vma) { - ret = -EBUSY; - goto out; - } - anon_vma_lock_write(anon_vma); - } - if (is_anon) ret = __folio_freeze_split_anon(folio, new_order, split_at, true, list, split_type); @@ -4368,11 +4376,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, free_folio_and_swap_cache(new_folio); } - if (anon_vma) { - anon_vma_unlock_write(anon_vma); - put_anon_vma(anon_vma); - } -out: /* restore to caller's old_memcg */ set_active_memcg(old_memcg); mem_cgroup_put(memcg); From 859e9eda19348ddba3d51a565e23f83101c03e79 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:44 +0200 Subject: [PATCH 0640/1012] mm/huge_memory: move memcg switch into the file split helper The xarray node allocations in __folio_freeze_split_file() need to be charged to the folio's memcg, so move the memcg switch from __folio_split() into the helper. The anon split helper and the after-split folio freeing perform no chargeable allocations, so no memcg handling is left in __folio_split(). Rename its out_no_memcg label to out. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-13-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Acked-by: Zi Yan Acked-by: David Hildenbrand (Arm) Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 36 +++++++++++++++++++----------------- 1 file changed, 19 insertions(+), 17 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index ec955488d280f9..46d00bf6e43796 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4142,6 +4142,7 @@ static int __folio_freeze_split_file(struct folio *folio, struct address_space *mapping = folio->mapping; XA_STATE(xas, &mapping->i_pages, folio->index); struct folio *end_folio = folio_next(folio); + struct mem_cgroup *memcg, *old_memcg; struct folio *new_folio, *next; int nr_shmem_dropped = 0; unsigned int min_order; @@ -4154,9 +4155,18 @@ static int __folio_freeze_split_file(struct folio *folio, if (new_order < min_order) return -EINVAL; + /* + * Switch to folio's memcg as xarray node allocation can happen and + * needs to charge to it. + */ + memcg = get_mem_cgroup_from_folio(folio); + old_memcg = set_active_memcg(memcg); + gfp = current_gfp_context(mapping_gfp_mask(mapping) & GFP_RECLAIM_MASK); - if (!filemap_release_folio(folio, gfp)) - return -EBUSY; + if (!filemap_release_folio(folio, gfp)) { + ret = -EBUSY; + goto fail_free; + } mapping_set_update(&xas, mapping); @@ -4288,6 +4298,9 @@ static int __folio_freeze_split_file(struct folio *folio, */ i_mmap_unlock_read(mapping); fail_free: + /* Restore the previously active memcg */ + set_active_memcg(old_memcg); + mem_cgroup_put(memcg); xas_destroy(&xas); return ret; } @@ -4319,7 +4332,6 @@ static int __folio_split(struct folio *folio, unsigned int new_order, { struct folio *end_folio = folio_next(folio); bool is_anon = folio_test_anon(folio); - struct mem_cgroup *memcg, *old_memcg; int old_order = folio_order(folio); struct folio *new_folio, *next; int ret; @@ -4329,27 +4341,20 @@ static int __folio_split(struct folio *folio, unsigned int new_order, if (folio != page_folio(split_at) || folio != page_folio(lock_at)) { ret = -EINVAL; - goto out_no_memcg; + goto out; } if (new_order >= old_order) { ret = -EINVAL; - goto out_no_memcg; + goto out; } ret = folio_check_splittable(folio, new_order, split_type); if (ret) { VM_WARN_ONCE(ret == -EINVAL, "Tried to split an unsplittable folio"); - goto out_no_memcg; + goto out; } - /* - * switch to folio's memcg as xarray node allocation can happen and - * needs to charge to it. - */ - memcg = get_mem_cgroup_from_folio(folio); - old_memcg = set_active_memcg(memcg); - if (is_anon) ret = __folio_freeze_split_anon(folio, new_order, split_at, true, list, split_type); @@ -4376,10 +4381,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, free_folio_and_swap_cache(new_folio); } - /* restore to caller's old_memcg */ - set_active_memcg(old_memcg); - mem_cgroup_put(memcg); -out_no_memcg: +out: if (is_pmd_order(old_order)) count_vm_event(!ret ? THP_SPLIT_PAGE : THP_SPLIT_PAGE_FAILED); count_mthp_stat(old_order, !ret ? MTHP_STAT_SPLIT : MTHP_STAT_SPLIT_FAILED); From f2e6ed975bb70100f6c668748b1569d7e1ed15f4 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:45 +0200 Subject: [PATCH 0641/1012] mm/huge_memory: drop the unused do_lru argument of the file split helper The only caller of __folio_freeze_split_file() always passes do_lru as true, so the argument and the branches gated on it are dead code. Drop it. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-14-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Acked-by: David Hildenbrand (Arm) Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 16 +++++----------- 1 file changed, 5 insertions(+), 11 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 46d00bf6e43796..4bcd57540eea13 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4136,8 +4136,7 @@ static int __folio_freeze_split_anon(struct folio *folio, static int __folio_freeze_split_file(struct folio *folio, unsigned int new_order, struct page *split_at, - bool do_lru, struct list_head *list, - enum split_type split_type) + struct list_head *list, enum split_type split_type) { struct address_space *mapping = folio->mapping; XA_STATE(xas, &mapping->i_pages, folio->index); @@ -4231,9 +4230,7 @@ static int __folio_freeze_split_file(struct folio *folio, } /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ - if (do_lru) - lruvec = folio_lruvec_lock(folio); - + lruvec = folio_lruvec_lock(folio); ret = __split_frozen_folio(folio, new_order, split_at, &xas, mapping, split_type); @@ -4255,8 +4252,7 @@ static int __folio_freeze_split_file(struct folio *folio, folio_ref_unfreeze(new_folio, folio_cache_ref_count(new_folio) + 1); - if (do_lru) - lru_add_split_folio(folio, new_folio, lruvec, list); + lru_add_split_folio(folio, new_folio, lruvec, list); /* Add the new folio to the page cache. */ if (new_folio->index < end) { @@ -4282,9 +4278,7 @@ static int __folio_freeze_split_file(struct folio *folio, * and its caller can see stale page cache entries. */ folio_ref_unfreeze(folio, folio_cache_ref_count(folio) + 1); - - if (do_lru) - lruvec_unlock(lruvec); + lruvec_unlock(lruvec); fail: xas_unlock_irq(&xas); fail_mmap_unlock: @@ -4360,7 +4354,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, true, list, split_type); else ret = __folio_freeze_split_file(folio, new_order, split_at, - true, list, split_type); + list, split_type); /* * Unlock all after-split folios except the one containing From 163eff1dc39717960e675edb8a6522895fae4fdf Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:46 +0200 Subject: [PATCH 0642/1012] mm/huge_memory: clean up after-split folio freeing in __folio_split Replace free_folio_and_swap_cache() with an explicit folio_free_swap() and folio_put() in the after-split loop. free_folio_and_swap_cache() must trylock it again and re-check folio_mapped() before freeing the swap cache entries. If the trylock loses a race, the entries are left behind even though the folio reference is dropped. The sub folios are still locked here, so just directly call folio_free_swap() under the lock if it's unmapped, then unlock and drop the reference. This makes the swap cache freeing deterministic and the reference drop explicit. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-15-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 4bcd57540eea13..2614a4676d7ba6 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4325,7 +4325,8 @@ static int __folio_split(struct folio *folio, unsigned int new_order, struct list_head *list, enum split_type split_type) { struct folio *end_folio = folio_next(folio); - bool is_anon = folio_test_anon(folio); + const bool is_anon = folio_test_anon(folio); + const bool is_swapcache = folio_test_swapcache(folio); int old_order = folio_order(folio); struct folio *new_folio, *next; int ret; @@ -4365,14 +4366,16 @@ static int __folio_split(struct folio *folio, unsigned int new_order, if (new_folio == page_folio(lock_at)) continue; - folio_unlock(new_folio); /* * Subpages whose mapping has been zapped may be freed * earlier, but freeing them requires taking the - * lru_lock, so we defer put_page() on tail pages until + * lru_lock, so we defer folio_put() on tail pages until * after the split completes. */ - free_folio_and_swap_cache(new_folio); + if (is_swapcache && !folio_mapped(new_folio)) + folio_free_swap(new_folio); + folio_unlock(new_folio); + folio_put(new_folio); } out: @@ -4399,7 +4402,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, * isolated from LRU (if applicable) * * Upon return, the folio is not remapped, split folios are not added to LRU, - * free_folio_and_swap_cache() is not called, and new folios remain locked. + * folio_free_swap() is not called, and new folios remain locked. * * Return: 0 on success, -EAGAIN if the folio cannot be split (e.g., due to * insufficient reference count or extra pins). From 0d7859d847edf25378fff1660d99a05edd98e534 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:47 +0200 Subject: [PATCH 0643/1012] mm/huge_memory: count only swap cache refs in anon folio split Only __folio_freeze_split_anon() sees anon folios and swap cache folios now. The file split helper only handles page cache folios, which hold exactly folio_nr_pages() references. Rename folio_cache_ref_count() to folio_swapcache_ref_count() and drop the anon check so the helper counts what its name says. The file split helper now uses folio_nr_pages() directly. No feature change. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-16-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 35 +++++++++++++++-------------------- 1 file changed, 15 insertions(+), 20 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 2614a4676d7ba6..349c75b832cd9e 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3997,10 +3997,10 @@ int folio_check_splittable(struct folio *folio, unsigned int new_order, return 0; } -/* Number of folio references from the pagecache or the swapcache. */ -static unsigned int folio_cache_ref_count(const struct folio *folio) +/* Number of folio references from the swapcache. */ +static unsigned int folio_swapcache_ref_count(const struct folio *folio) { - if (folio_test_anon(folio) && !folio_test_swapcache(folio)) + if (!folio_test_swapcache(folio)) return 0; return folio_nr_pages(folio); } @@ -4067,7 +4067,7 @@ static int __folio_freeze_split_anon(struct folio *folio, local_irq_disable(); - if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { + if (!folio_ref_freeze(folio, folio_swapcache_ref_count(folio) + 1)) { ret = -EAGAIN; goto out_no_split; } @@ -4104,7 +4104,7 @@ static int __folio_freeze_split_anon(struct folio *folio, next = folio_next(new_folio); zone_device_private_split_cb(folio, new_folio); folio_ref_unfreeze(new_folio, - folio_cache_ref_count(new_folio) + 1); + folio_swapcache_ref_count(new_folio) + 1); if (do_lru) lru_add_split_folio(folio, new_folio, lruvec, list); if (ci) @@ -4112,7 +4112,7 @@ static int __folio_freeze_split_anon(struct folio *folio, } zone_device_private_split_cb(folio, NULL); - folio_ref_unfreeze(folio, folio_cache_ref_count(folio) + 1); + folio_ref_unfreeze(folio, folio_swapcache_ref_count(folio) + 1); if (do_lru) lruvec_unlock(lruvec); @@ -4138,6 +4138,7 @@ static int __folio_freeze_split_file(struct folio *folio, unsigned int new_order, struct page *split_at, struct list_head *list, enum split_type split_type) { + const long old_nr_pages = folio_nr_pages(folio); struct address_space *mapping = folio->mapping; XA_STATE(xas, &mapping->i_pages, folio->index); struct folio *end_folio = folio_next(folio); @@ -4211,22 +4212,16 @@ static int __folio_freeze_split_file(struct folio *folio, goto fail; } - if (!folio_ref_freeze(folio, folio_cache_ref_count(folio) + 1)) { + if (!folio_ref_freeze(folio, old_nr_pages + 1)) { ret = -EAGAIN; goto fail; } - if (folio_test_pmd_mappable(folio) && - new_order < HPAGE_PMD_ORDER) { - int nr = folio_nr_pages(folio); - - if (folio_test_swapbacked(folio)) { - lruvec_stat_mod_folio(folio, - NR_SHMEM_THPS, -nr); - } else { - lruvec_stat_mod_folio(folio, - NR_FILE_THPS, -nr); - } + if (folio_test_pmd_mappable(folio) && new_order < HPAGE_PMD_ORDER) { + if (folio_test_swapbacked(folio)) + lruvec_stat_mod_folio(folio, NR_SHMEM_THPS, -old_nr_pages); + else + lruvec_stat_mod_folio(folio, NR_FILE_THPS, -old_nr_pages); } /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ @@ -4250,7 +4245,7 @@ static int __folio_freeze_split_file(struct folio *folio, next = folio_next(new_folio); folio_ref_unfreeze(new_folio, - folio_cache_ref_count(new_folio) + 1); + folio_nr_pages(new_folio) + 1); lru_add_split_folio(folio, new_folio, lruvec, list); @@ -4277,7 +4272,7 @@ static int __folio_freeze_split_file(struct folio *folio, * Otherwise, a parallel folio_try_get() can grab @folio * and its caller can see stale page cache entries. */ - folio_ref_unfreeze(folio, folio_cache_ref_count(folio) + 1); + folio_ref_unfreeze(folio, folio_nr_pages(folio) + 1); lruvec_unlock(lruvec); fail: xas_unlock_irq(&xas); From 272f4f04036a4a43cbff535e6a5bad22c4e00c71 Mon Sep 17 00:00:00 2001 From: Kairui Song Date: Wed, 23 Sep 2026 23:32:48 +0200 Subject: [PATCH 0644/1012] mm/huge_memory: drop the redundant mapping argument of __split_frozen_folio The mapping parameter only served as a non-NULL check to detect whether page cache entries need updating. The xa_state pointer conveys exactly the same information: the anon split helper passes NULL and the file split helper passes &xas, which is non-NULL iff the folio is in the page cache. Use the xas pointer instead and drop the parameter, along with its kerneldoc entry. Link: https://lore.kernel.org/20260923-swap-thp-cleanup-v6-17-ba1b4ba72c6f@tencent.com Signed-off-by: Kairui Song Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Yeoreum Yun Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Lorenzo Stoakes Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Chris Li Cc: Kemeng Shi Cc: Nhat Pham Cc: Baoquan He Cc: Barry Song Cc: Youngjun Park Cc: Shivam Kalra Cc: Kairui Song --- mm/huge_memory.c | 11 ++++------- 1 file changed, 4 insertions(+), 7 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 349c75b832cd9e..8f8bf60a22649a 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3833,7 +3833,6 @@ static void __split_folio_to_order(struct folio *folio, int old_order, * @split_at: in buddy allocator like split, the folio containing @split_at * will be split until its order becomes @new_order. * @xas: xa_state pointing to folio->mapping->i_pages and locked by caller - * @mapping: @folio->mapping * @split_type: if the split is uniform or not (buddy allocator like split) * * @@ -3866,7 +3865,7 @@ static void __split_folio_to_order(struct folio *folio, int old_order, */ static int __split_frozen_folio(struct folio *folio, int new_order, struct page *split_at, struct xa_state *xas, - struct address_space *mapping, enum split_type split_type) + enum split_type split_type) { const bool is_anon = folio_test_anon(folio); int old_order = folio_order(folio); @@ -3890,7 +3889,7 @@ static int __split_frozen_folio(struct folio *folio, int new_order, if (is_anon && split_order == 1) continue; - if (mapping) { + if (xas) { /* * uniform split has xas_split_alloc() called before * irq is disabled to allocate enough memory, whereas @@ -4089,8 +4088,7 @@ static int __folio_freeze_split_anon(struct folio *folio, if (do_lru) lruvec = folio_lruvec_lock(folio); - ret = __split_frozen_folio(folio, new_order, split_at, NULL, - NULL, split_type); + ret = __split_frozen_folio(folio, new_order, split_at, NULL, split_type); /* * Unfreeze the after-split folios and put them back to the right @@ -4226,8 +4224,7 @@ static int __folio_freeze_split_file(struct folio *folio, /* lock lru list/PageCompound, ref frozen by page_ref_freeze */ lruvec = folio_lruvec_lock(folio); - ret = __split_frozen_folio(folio, new_order, split_at, &xas, - mapping, split_type); + ret = __split_frozen_folio(folio, new_order, split_at, &xas, split_type); /* * Unfreeze after-split folios and put them back to the right From c0f1504f7f53c4c636d3bc0876e0308f02e2b3c6 Mon Sep 17 00:00:00 2001 From: Guillaume Morin Date: Mon, 14 Sep 2026 17:36:21 +0200 Subject: [PATCH 0645/1012] selftests/mm: hugetlb_madv_vs_map: add underflow test Add a test that checks for underflows when a parent unmaps the page first. Also check that when the child exits the reserve count is correct. Link: https://lore.kernel.org/all/alEJkwn5VlTTH_ZX@bender.morinfr.org/ Link: https://lore.kernel.org/aqgUdbtumaO8RiIb@bender.morinfr.org Signed-off-by: Guillaume Morin Signed-off-by: Andrew Morton Reviewed-by: Breno Leitao Reviewed-by: Mike Rapoport --- .../testing/selftests/mm/hugepage_settings.c | 9 ++ .../testing/selftests/mm/hugepage_settings.h | 1 + .../selftests/mm/hugetlb_madv_vs_map.c | 141 +++++++++++++++--- 3 files changed, 130 insertions(+), 21 deletions(-) diff --git a/tools/testing/selftests/mm/hugepage_settings.c b/tools/testing/selftests/mm/hugepage_settings.c index d7917dce3abac7..584054736ce99f 100644 --- a/tools/testing/selftests/mm/hugepage_settings.c +++ b/tools/testing/selftests/mm/hugepage_settings.c @@ -449,6 +449,15 @@ unsigned long hugetlb_free_pages(unsigned long size) return read_num(path); } +unsigned long hugetlb_nr_resv_pages(unsigned long size) +{ + char path[PATH_MAX]; + + hugetlb_sysfs_path(path, sizeof(path), size, "resv_hugepages"); + + return read_num(path); +} + static bool __hugetlb_setup(unsigned long size, unsigned long nr) { unsigned long free = hugetlb_free_pages(size); diff --git a/tools/testing/selftests/mm/hugepage_settings.h b/tools/testing/selftests/mm/hugepage_settings.h index 726c73c43c05ba..548e9d288d1d16 100644 --- a/tools/testing/selftests/mm/hugepage_settings.h +++ b/tools/testing/selftests/mm/hugepage_settings.h @@ -98,6 +98,7 @@ unsigned long default_huge_page_size(void); unsigned long hugetlb_nr_pages(unsigned long size); void hugetlb_set_nr_pages(unsigned long size, unsigned long nr); unsigned long hugetlb_free_pages(unsigned long size); +unsigned long hugetlb_nr_resv_pages(unsigned long size); static inline void hugetlb_save_settings(void) { diff --git a/tools/testing/selftests/mm/hugetlb_madv_vs_map.c b/tools/testing/selftests/mm/hugetlb_madv_vs_map.c index f94549efcc6ff3..0f15eff1da0403 100644 --- a/tools/testing/selftests/mm/hugetlb_madv_vs_map.c +++ b/tools/testing/selftests/mm/hugetlb_madv_vs_map.c @@ -3,18 +3,6 @@ * A test case that must run on a system with one and only one huge page available. * # echo 1 > /sys/kernel/mm/hugepages/hugepages-2048kB/nr_hugepages * - * During setup, the test allocates the only available page, and starts three threads: - * - thread1: - * * madvise(MADV_DONTNEED) on the allocated huge page - * - thread 2: - * * Write to the allocated huge page - * - thread 3: - * * Try to allocated an extra huge page (which must not available) - * - * The test fails if thread3 is able to allocate a page. - * - * Touching the first page after thread3's allocation will raise a SIGBUS - * * Author: Breno Leitao */ #include @@ -22,6 +10,7 @@ #include #include #include +#include #include #include "vm_util.h" @@ -74,7 +63,21 @@ void *map_extra(void *unused) return NULL; } -int main(void) +/* During setup in main, the only available page was allocated. This test then + * starts three threads: + * + * - thread1: + * * madvise(MADV_DONTNEED) on the allocated huge page + * - thread 2: + * * Write to the allocated huge page + * - thread 3: + * * Try to allocated an extra huge page (which must not available) + * + * The test fails if thread3 is able to allocate a page. + * + * Touching the first page after thread3's allocation will raise a SIGBUS + */ +void test_madv_vs_map(void) { pthread_t thread1, thread2, thread3; void *ret; @@ -85,13 +88,6 @@ int main(void) */ int max = 10; - ksft_print_header(); - ksft_set_plan(1); - - if (!hugetlb_setup_default_exact(1)) - ksft_exit_skip("This test needs one and only one page to execute. Got %lu\n", - hugetlb_free_default_pages()); - mmap_size = default_huge_page_size(); while (max--) { @@ -100,7 +96,7 @@ int main(void) -1, 0); if ((unsigned long)huge_ptr == -1) - ksft_exit_fail_msg("Failed to allocate huge page\n"); + ksft_exit_fail_perror("Failed to allocate huge page"); pthread_create(&thread1, NULL, madv, NULL); pthread_create(&thread2, NULL, touch, NULL); @@ -120,5 +116,108 @@ int main(void) } ksft_test_result_pass("No unexpected huge page allocations\n"); +} + +/* We create a child process, then unmap the page in the parent while the child + * waits and verify that there is no underflow of the reserved count. We also + * verify that after the child exits, the reserved count is properly restored. + */ +void test_underflow(void) +{ + pid_t pid; + int pipe_fds[2]; + unsigned long nr_reserved = 0; + + huge_ptr = mmap(NULL, mmap_size, PROT_READ | PROT_WRITE, + MAP_PRIVATE | MAP_ANONYMOUS | MAP_HUGETLB, -1, 0); + + if ((unsigned long)huge_ptr == -1) + ksft_exit_fail_perror("Failed to allocate huge page"); + + nr_reserved = hugetlb_nr_resv_pages(default_huge_page_size()); + if (nr_reserved != 1) + ksft_exit_fail_msg("Unexpected number of reserved pages: %lu, expected 1\n", + nr_reserved); + + /* Force the fault to ensure the reservation is consumed */ + *huge_ptr = 0; + nr_reserved = hugetlb_nr_resv_pages(default_huge_page_size()); + if (nr_reserved != 0) + ksft_exit_fail_msg("Unexpected number of reserved pages: %lu, expected 0\n", + nr_reserved); + + if (pipe(pipe_fds) != 0) + ksft_exit_fail_perror("pipe failed"); + + pid = fork(); + if (pid < 0) + ksft_exit_fail_perror("fork failed"); + + if (pid == 0) { + /* Child: Simply wait for the parent */ + char b; + + close(pipe_fds[1]); + if (read(pipe_fds[0], &b, 1) < 0) + ksft_perror("child read failed"); + /* Let the parent do the cleanup */ + _exit(0); + } + + /* Parent */ + close(pipe_fds[0]); + + /* First unmap, this will close the vma */ + if (munmap(huge_ptr, mmap_size) != 0) { + ksft_perror("munmap failed"); + goto err_cleanup; + } + + nr_reserved = hugetlb_nr_resv_pages(default_huge_page_size()); + if (nr_reserved == 0) { + ksft_test_result_pass("Underflow not present!\n"); + } else { + ksft_test_result_fail("Unexpected HugePages_Rsvd=%ld after munmap, should be 0\n", + nr_reserved); + goto err_cleanup; + } + /* Make the child exit, this should restore HugePages_Rsvd to 0 */ + if (write(pipe_fds[1], &nr_reserved, 1) < 0) { + /* If write failed, the child is likely already gone */ + ksft_exit_fail_perror("write failed"); + } + close(pipe_fds[1]); + if (waitpid(pid, NULL, 0) <= 0) + ksft_exit_fail_msg("waitpid failed\n"); + + nr_reserved = hugetlb_nr_resv_pages(default_huge_page_size()); + if (nr_reserved == 0) + ksft_test_result_pass("After the child dies, HugePages_Rsvd is properly set to 0\n"); + else + ksft_exit_fail_msg("Unexpected HugePages_Rsvd=%ld after the child termination munmap, should be 0\n", nr_reserved); + + return; + +err_cleanup: + if (write(pipe_fds[1], &nr_reserved, 1) < 0) + ksft_exit_fail_perror("write failed"); + if (waitpid(pid, NULL, 0) <= 0) + ksft_exit_fail_perror("waitpid failed"); + + ksft_exit_fail(); +} + +int main(void) +{ + ksft_print_header(); + ksft_set_plan(3); + + if (!hugetlb_setup_default_exact(1)) + ksft_exit_skip("This test needs one and only one page to execute. Got %lu\n", + hugetlb_free_default_pages()); + + test_madv_vs_map(); + test_underflow(); + ksft_finished(); } From b38a7946e752973f6166c1e5a3e11e67e4afcea9 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:10 +0100 Subject: [PATCH 0646/1012] mm/vma: fix mmap_prepare file handling, remove file_doesnt_need_get Patch series "mm: make VMA flag semantics explicit, eliminate VM_SPECIAL", v3. The VM_SPECIAL / VMA_SPECIAL_FLAGS mask conflates several unrelated properties: * Is this kernel-owned, whether MMIO, kernel-allocated pages, or ordinary pages a driver maps itself? * Can it be expanded or merged? * Is this a 'weird' case like mlock where migration might race and we 'have' to set invalid flags to notify? * Is it another 'weird' case where we just want to stop GUP from touching it? Driver writers have often been confused about this, and who can blame them? It also interacts badly with the eternal edgecase known as hugetlb - which sets VMA_DONTEXPAND_BIT but doesn't also want to be treated like a 'special' flag. Another issue is that we cannot make sensible assumptions about flag use. It's not possible to assume VMA_IO_BIT means iommu because drivers abuse it and mlock abuses it. Special is also an overloaded term in mm. VDSO and VVAR mappings are also called 'special' but they're special in a... special way. Sometimes things are called special that are a subset of VMA_SPECIAL_FLAGS (VMA_PFNMAP_BIT and VMA_MIXEDMAP_BIT for instance when it comes to zapping or vm_normal_folio()). There's a specific kind of special for THP too, which considers PFN map, mixed map 'special' but DAX not. It's all rather a mess. This series brings some order to things by both limiting what drivers can do with VMA flags and switching to using predicates that describe behaviour, not arbitrary flags. It establishes the invariant that only kernel-owned mappings may set VMA_IO_BIT or clear VMA_MAYWRITE_BIT in an mmap hook, enforcing this by validating VMA state after every mmap and mmap_prepare hook. It updates usbmon and sg to mmap_prepare in order to do so, adding a new mmap action for mapping discontiguous kernel pages, and has hfi1 and the ALSA PCM status page map their pages eagerly instead. It also establishes the invariant that VMA_MIXEDMAP_BIT be set when mapping kernel memory, something that is usually the case but happens not to be for some users - specifically defio, cmt_speech, uprobes and the bpf arena, all of which are updated to do the right thing. It replaces VM_SPECIAL and arbitrary flag tests with predicates that say what is actually being tested: vma_is_kernel_owned() Does a driver or kernel code manage a VMA's life cycle? vma_is_fixed_mapping() Is the VMA not permitted to be expanded or merged? vma_is_persistent() Do bytes written to the VMA stay written, and bytes read stay the same unless userland changes them? vma_can_merge() Can the VMA be merged with a compatible neighbour? vma_can_gup() Can GUP obtain pages from the VMA, i.e. is it neither a PFN map nor memory-mapped I/O? Remaining raw VMA_IO_BIT, VMA_PFNMAP_BIT and VMA_MIXEDMAP_BIT tests scattered across mm are also converted to predicates where it makes sense to do so. And also the opportunity is taken to eliminate THP's vma_is_special_huge() which was an existing source of confusion. This patch (of 39): The map->file_doesnt_need_get flag is confusing and the existing implementation has holes. Drivers are permitted to change the owning file of a mapping. If they do so, they are required to take a reference on that file. The mmap() operation which ultimately invokes __mmap_region() is guaranteed to drop the refcount for the original file the mapping was made under, but this is not true for the replaced file. This has been addressed so far by tracking map->file_doesnt_need_get, which is rather poorly named and unfortunately fails to correctly track whether or not an additional put were needed in a number of cases. Make life easier by removing this flag, and instead drop the reference for both mmap_prepare and the deprecated mmap callback in a new function put_map(). Track whether this needs to be done by aligning mmap_state with vm_area_desc and store the original file in the map->file field, keeping the updated file in map->vm_file. In order to have the same behaviour for both types of hooks, only drop the reference __mmap_new_file_vma() itself took in its error path, deferring the replaced file's reference to put_map(). To make this work correctly, map->vm_file has to be updated before any error handling, so update __mmap_new_file_vma() and call_mmap_prepare() to set this field first. Also when mmap_prepare() changes the file and is then merged, the reference count also must be decremented, so update the logic to call put_map() in this case too. Also update __compat_vma_mmap() to manually perform this step for stacked file systems using the compatibility layer, and update compat_set_vma_from_desc() to replace vma_set_file() with a correct refcount/file update. No in-tree driver is impacted by the incorrect implementation of this currently (no driver that does this is mergeable for one), so this does not need to be a fix. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-0-4583d8a23bca@kernel.org Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-1-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis (Meta) Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- mm/internal.h | 1 + mm/util.c | 5 +++- mm/vma.c | 81 +++++++++++++++++++++++++++++++-------------------- mm/vma.h | 6 ++-- 4 files changed, 58 insertions(+), 35 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 05179c4b2090ef..323fa1514b92a3 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -7,6 +7,7 @@ #ifndef __MM_INTERNAL_H #define __MM_INTERNAL_H +#include #include #include #include diff --git a/mm/util.c b/mm/util.c index bf0513d1d3d086..016932780925e1 100644 --- a/mm/util.c +++ b/mm/util.c @@ -1228,8 +1228,11 @@ int __compat_vma_mmap(struct vm_area_desc *desc, /* Perform any preparatory tasks for mmap action. */ err = mmap_action_prepare(desc); - if (err) + if (err) { + if (desc->vm_file != vma->vm_file) + fput(desc->vm_file); return err; + } /* Update the VMA from the descriptor. */ compat_set_vma_from_desc(vma, desc); /* Complete any specified mmap actions. */ diff --git a/mm/vma.c b/mm/vma.c index c24ee55b7ffff5..da51c470ceaa72 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -24,7 +24,8 @@ struct mmap_state { vm_flags_t vm_flags; vma_flags_t vma_flags; }; - struct file *file; + struct file *file; /* mmap()-specified file. */ + struct file *vm_file; /* May be updated by mmap_prepare. */ pgprot_t page_prot; /* User-defined fields, perhaps updated by .mmap_prepare(). */ @@ -43,8 +44,6 @@ struct mmap_state { /* Determine if we can check KSM flags early in mmap() logic. */ bool check_ksm_early :1; - /* If .mmap_prepare changed the file, we don't need to pin. */ - bool file_doesnt_need_get :1; }; #define MMAP_STATE(name, mm_, vmi_, addr_, len_, pgoff_, anon_pgoff_, vma_flags_, file_) \ @@ -58,6 +57,7 @@ struct mmap_state { .pglen = PHYS_PFN(len_), \ .vma_flags = vma_flags_, \ .file = file_, \ + .vm_file = file_, \ .page_prot = vma_flags_to_page_prot(vma_flags_), \ } @@ -70,7 +70,7 @@ struct mmap_state { .vma_flags = (map_)->vma_flags, \ .pgoff = (map_)->pgoff, \ .anon_pgoff = (map_)->anon_pgoff, \ - .file = (map_)->file, \ + .file = (map_)->vm_file, \ .prev = (map_)->prev, \ .middle = vma_, \ .next = (vma_) ? NULL : (map_)->next, \ @@ -2462,7 +2462,7 @@ void mm_drop_all_locks(struct mm_struct *mm) */ static bool accountable_mapping(struct mmap_state *map) { - const struct file *file = map->file; + const struct file *file = map->vm_file; /* * hugetlb has its own accounting separate from the core VM @@ -2511,7 +2511,7 @@ static void vms_abort_munmap_vmas(struct vma_munmap_struct *vms, static void update_ksm_flags(struct mmap_state *map) { - map->vma_flags = ksm_vma_flags(map->mm, map->file, map->vma_flags); + map->vma_flags = ksm_vma_flags(map->mm, map->vm_file, map->vma_flags); } static void set_desc_from_map(struct vm_area_desc *desc, @@ -2521,7 +2521,7 @@ static void set_desc_from_map(struct vm_area_desc *desc, desc->end = map->end; desc->pgoff = map->pgoff; - desc->vm_file = map->file; + desc->vm_file = map->vm_file; desc->vma_flags = map->vma_flags; desc->page_prot = map->page_prot; } @@ -2601,6 +2601,10 @@ static int __mmap_setup(struct mmap_state *map, struct vm_area_desc *desc, return 0; } +static bool map_same_file(struct mmap_state *map) +{ + return map->vm_file == map->file; +} static int __mmap_new_file_vma(struct mmap_state *map, struct vm_area_struct *vma) @@ -2608,20 +2612,23 @@ static int __mmap_new_file_vma(struct mmap_state *map, struct vma_iterator *vmi = map->vmi; int error; - vma->vm_file = map->file; - if (!map->file_doesnt_need_get) - get_file(map->file); + vma->vm_file = map->vm_file; + if (map_same_file(map)) + get_file(map->vm_file); - if (!map->file->f_op->mmap) + if (!map->vm_file->f_op->mmap) return 0; error = mmap_file(vma->vm_file, vma); + map->vm_file = vma->vm_file; + if (error) { UNMAP_STATE(unmap, vmi, vma, vma->vm_start, vma->vm_end, map->prev, map->next); - fput(vma->vm_file); - vma->vm_file = NULL; + if (map_same_file(map)) + fput(map->vm_file); + vma->vm_file = NULL; vma_iter_set(vmi, vma->vm_end); /* Undo any partial mapping done by a device driver. */ unmap_region(&unmap); @@ -2638,7 +2645,6 @@ static int __mmap_new_file_vma(struct mmap_state *map, !vma_flags_test(&map->vma_flags, VMA_MAYWRITE_BIT) && vma_test(vma, VMA_MAYWRITE_BIT)); - map->file = vma->vm_file; map->vma_flags = vma->flags; return 0; @@ -2646,7 +2652,7 @@ static int __mmap_new_file_vma(struct mmap_state *map, static void map_set_anon(struct mmap_state *map) { - map->file = NULL; + map->vm_file = NULL; map->vm_ops = NULL; map->pgoff = map->addr >> PAGE_SHIFT; } @@ -2658,7 +2664,7 @@ static bool map_is_private(const struct mmap_state *map) static bool map_is_anon(const struct mmap_state *map) { - return map_is_private(map) && !map->file; + return map_is_private(map) && !map->vm_file; } /* @@ -2703,7 +2709,7 @@ static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, } /* Invoke callbacks. */ - if (map->file) + if (map->vm_file) error = __mmap_new_file_vma(map, vma); else if (!is_anon) error = shmem_zero_setup(vma); @@ -2812,10 +2818,14 @@ static int call_mmap_prepare(struct mmap_state *map, int err; /* Invoke the hook. */ - err = vfs_mmap_prepare(map->file, desc); + err = vfs_mmap_prepare(map->vm_file, desc); if (err) return err; + /* Update first so file refcount tracked correctly. */ + if (desc->vm_file != map->vm_file) + map->vm_file = desc->vm_file; + /* It's invalid for mmap_prepare hooks to clear vm_ops. */ if (!desc->vm_ops) return -EINVAL; @@ -2826,10 +2836,6 @@ static int call_mmap_prepare(struct mmap_state *map, /* Update fields permitted to be changed. */ map->pgoff = desc->pgoff; - if (desc->vm_file != map->file) { - map->file_doesnt_need_get = true; - map->file = desc->vm_file; - } map->vma_flags = desc->vma_flags; map->page_prot = desc->page_prot; /* User-defined fields. */ @@ -2841,7 +2847,7 @@ static int call_mmap_prepare(struct mmap_state *map, * anonymous mappings. Rather than allowing these mappings to be odd * outliers, simply make them truly anonymous. */ - if (map_is_private(map) && file_is_dev_zero(map->file)) + if (map_is_private(map) && file_is_dev_zero(map->vm_file)) map_set_anon(map); return 0; @@ -2860,7 +2866,7 @@ static void set_vma_user_defined_fields(struct vm_area_struct *vma, */ static bool can_set_ksm_flags_early(struct mmap_state *map) { - struct file *file = map->file; + struct file *file = map->vm_file; /* Anonymous mappings have no driver which can change them. */ if (!file) @@ -2883,6 +2889,20 @@ static bool can_set_ksm_flags_early(struct mmap_state *map) return false; } +static void put_map(struct mmap_state *map) +{ + /* + * An error occurred or the VMA was merged. + * + * If the file was changed by the driver (which is required to increment + * the replacement file's reference count), drop its reference count. + * + * On error, the caller always drops the original file regardless. + */ + if (map->vm_file && !map_same_file(map)) + fput(map->vm_file); +} + static unsigned long __mmap_region(struct file *file, unsigned long addr, unsigned long len, vma_flags_t vma_flags, unsigned long pgoff, struct list_head *uf) @@ -2937,7 +2957,10 @@ static unsigned long __mmap_region(struct file *file, unsigned long addr, __mmap_complete(&map, vma); - if (have_mmap_prepare && allocated_new) { + if (!allocated_new) { + /* Merged, so need to drop refcount. */ + put_map(&map); + } else if (have_mmap_prepare) { error = mmap_action_complete(vma, &desc.action, /*is_compat=*/false); if (error) @@ -2951,13 +2974,7 @@ static unsigned long __mmap_region(struct file *file, unsigned long addr, if (map.charged) vm_unacct_memory(map.charged); abort_munmap: - /* - * This indicates that .mmap_prepare has set a new file, differing from - * desc->vm_file. But since we're aborting the operation, only the - * original file will be cleaned up. Ensure we clean up both. - */ - if (map.file_doesnt_need_get) - fput(map.file); + put_map(&map); vms_abort_munmap_vmas(&map.vms, &map.mas_detach); return error; } diff --git a/mm/vma.h b/mm/vma.h index f856d9ace3a6a6..4665b40163fa67 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -394,8 +394,10 @@ static inline void compat_set_vma_from_desc(struct vm_area_struct *vma, /* Mutable fields. Populated with initial state. */ vma_set_pgoff(vma, desc->pgoff); - if (desc->vm_file != vma->vm_file) - vma_set_file(vma, desc->vm_file); + if (desc->vm_file != vma->vm_file) { + fput(vma->vm_file); + vma->vm_file = desc->vm_file; + } vma->flags = desc->vma_flags; vma->vm_page_prot = desc->page_prot; From a4120b8d149b53bfefbcabccbce9094784859c3a Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:12 +0100 Subject: [PATCH 0647/1012] mm/vma: introduce and use vma_[flags_]can_merge() Replace the open-coded VMA_SPECIAL_FLAGS check in the VMA merge logic with two new functions vma_flags_can_merge() and vma_can_merge() and update the merge logic to use the former. This abstracts the check and expresses it in terms of the desired behaviour rather than an arbitrary and confusing VMA flag. This also lays the groundwork for making further improvements in VMA flag usage. Also update the userland VMA tests to reflect the change. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-3-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Suren Baghdasaryan Reviewed-by: Zi Yan Reviewed-by: Gregory Price (Meta) Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 21 +++++++++++++++++++++ mm/vma.c | 19 +++++++++++-------- tools/testing/vma/include/dup.h | 5 +++++ 3 files changed, 37 insertions(+), 8 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 0a2a7fc4a442fe..b3fdc5e100c48a 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1613,6 +1613,27 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) return is_shared_maywrite(&vma->flags); } +/** + * vma_flags_can_merge() - Do the specified VMA flags permit the VMA to be + * merged with another? + * @flags: The VMA flags to test. + * Returns: true if the flags permit merging, false otherwise. + */ +static inline bool vma_flags_can_merge(const vma_flags_t *flags) +{ + return !vma_flags_test_any_mask(flags, VMA_SPECIAL_FLAGS); +} + +/** + * vma_can_merge() - Do @vma's flags permit it to be merged with another VMA? + * @vma: The VMA to test. + * Returns: true if the flags permit merging, otherwise false. + */ +static inline bool vma_can_merge(const struct vm_area_struct *vma) +{ + return vma_flags_can_merge(&vma->flags); +} + /** * vma_kernel_pagesize - Default page size granularity for this VMA. * @vma: The user mapping. diff --git a/mm/vma.c b/mm/vma.c index da51c470ceaa72..801155efaff29e 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -924,13 +924,14 @@ static __must_check struct vm_area_struct *vma_merge_existing_range( vmg->state = VMA_MERGE_NOMERGE; + if (!vma_flags_can_merge(&vmg->vma_flags)) + return NULL; /* - * If a special mapping or if the range being modified is neither at the - * furthermost left or right side of the VMA, then we have no chance of - * merging and should abort. + * If the range being modified is neither at the furthermost left or + * right side of the VMA, then we have no chance of merging and should + * abort. */ - if (vma_flags_test_any_mask(&vmg->vma_flags, VMA_SPECIAL_FLAGS) || - (!left_side && !right_side)) + if (!left_side && !right_side) return NULL; if (left_side) @@ -1152,9 +1153,11 @@ struct vm_area_struct *vma_merge_new_range(struct vma_merge_struct *vmg) vmg->state = VMA_MERGE_NOMERGE; - /* Special VMAs are unmergeable, also if no prev/next. */ - if (vma_flags_test_any_mask(&vmg->vma_flags, VMA_SPECIAL_FLAGS) || - (!prev && !next)) + if (!vma_flags_can_merge(&vmg->vma_flags)) + return NULL; + + /* VMAs with no prev/next are unmergeable. */ + if (!prev && !next) return NULL; can_merge_left = can_vma_merge_left(vmg); diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 16c09dac59d9b4..2fd422789717fe 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1647,3 +1647,8 @@ static inline bool file_is_dev_zero(const struct file *file) { return file && file->f_op == &zero_fops; } + +static inline bool vma_flags_can_merge(const vma_flags_t *flags) +{ + return !vma_flags_test_any_mask(flags, VMA_SPECIAL_FLAGS); +} From b1c63cd47a3196886c274077c5b89e3c6acdd0b0 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:13 +0100 Subject: [PATCH 0648/1012] mm: consistently validate VMA state after mmap[_prepare] hooks When the f_op->mmap_prepare or deprecated f_op->mmap hooks are invoked, the driver might have done something crazy that is not permitted by the kernel. Currently we check for three such cases in __mmap_new_file_vma(), but only if the legacy f_op->mmap hook is used: * Did sparc ADI result in invalid flags? * Did the driver alter vma->vm_start? * Did the driver make a file-backed mapping on a read-only file writable? Generalise these checks for both mmap_prepare and mmap and apply to all invocations of mmap_file(), the f_op->mmap and f_op->mmap_prepare handling in the core VMA code and the mmap_prepare compatibility layer. Also extend the vm_start check to vm_end also - drivers must not change the VMA range at all. We also WARN_ON_ONCE() on these conditions as they are things that should simply not occur in the kernel and it's important to call it out when it does. We invoke mmap_prepare_validate() after mmap_action_prepare(), as mmap actions often manipulate state in the descriptor thus providing the final state the VMA will be derived from. Also call mmap_validate_vma_flags() in insert_vm_struct() to ensure that special regions which are inserted (such as a VDSO or VVAR) also satisfy the sanity checks. This way every VMA established through an mmap hook, whether via mmap() or the compatibility layer, or inserted via insert_vm_struct(), has been validated. brk() VMAs never pass through a driver hook and so need no such check. While we're here, also fixup a couple disjoint blocks of #ifdef CONFIG_MMU. Finally, update the VMA userland tests to reflect the change. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-4-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Suren Baghdasaryan Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/internal.h | 51 ++++++++++------ mm/util.c | 19 ++++-- mm/vma.c | 100 +++++++++++++++++++++++++++----- mm/vma.h | 25 +++++++- tools/testing/vma/include/dup.h | 10 ++++ 5 files changed, 163 insertions(+), 42 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 323fa1514b92a3..88f310cc215078 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -213,6 +213,24 @@ static inline void *folio_raw_mapping(const struct folio *folio) return (void *)(mapping & ~FOLIO_MAPPING_FLAGS); } +/* + * If the VMA has a close hook then close it, and since closing it might leave + * it in an inconsistent state which makes the use of any hooks suspect, clear + * them down by installing dummy empty hooks. + */ +static inline void vma_close(struct vm_area_struct *vma) +{ + if (vma->vm_ops && vma->vm_ops->close) { + vma->vm_ops->close(vma); + + /* + * The mapping is in an inconsistent state, and no further hooks + * may be invoked upon it. + */ + vma->vm_ops = &vma_dummy_vm_ops; + } +} + /* * This is a file-backed mapping, and is about to be memory mapped - invoke its * mmap hook and safely handle error conditions. On error, VMA hooks will be @@ -225,8 +243,12 @@ static inline void *folio_raw_mapping(const struct folio *folio) */ static inline int mmap_file(struct file *file, struct vm_area_struct *vma) { - int err = vfs_mmap(file, vma); + const unsigned long prev_start = vma->vm_start; + const unsigned long prev_end = vma->vm_end; + const vma_flags_t prev_flags = vma->flags; + int err; + err = vfs_mmap(file, vma); /* * Either we tried to call the file hook for mmap() and an error arose * or a driver set vma->vm_ops = NULL intending there to be no VMA @@ -239,26 +261,17 @@ static inline int mmap_file(struct file *file, struct vm_area_struct *vma) */ if (unlikely(err || !vma->vm_ops)) vma->vm_ops = &vma_dummy_vm_ops; + if (unlikely(err)) + return err; - return err; -} - -/* - * If the VMA has a close hook then close it, and since closing it might leave - * it in an inconsistent state which makes the use of any hooks suspect, clear - * them down by installing dummy empty hooks. - */ -static inline void vma_close(struct vm_area_struct *vma) -{ - if (vma->vm_ops && vma->vm_ops->close) { - vma->vm_ops->close(vma); - - /* - * The mapping is in an inconsistent state, and no further hooks - * may be invoked upon it. - */ - vma->vm_ops = &vma_dummy_vm_ops; + err = mmap_hook_validate(prev_start, prev_end, &prev_flags, vma); + if (unlikely(err)) { + vma->vm_start = prev_start; + vma->vm_end = prev_end; + vma_close(vma); } + + return err; } /* unmap_vmas is in mm/memory.c */ diff --git a/mm/util.c b/mm/util.c index 016932780925e1..bdd5923eebc7be 100644 --- a/mm/util.c +++ b/mm/util.c @@ -1224,19 +1224,28 @@ EXPORT_SYMBOL(compat_set_desc_from_vma); int __compat_vma_mmap(struct vm_area_desc *desc, struct vm_area_struct *vma) { + struct vm_area_desc prev_desc; int err; + /* Derive state prior to mmap_prepare hook. */ + compat_set_desc_from_vma(&prev_desc, desc->file, vma); /* Perform any preparatory tasks for mmap action. */ err = mmap_action_prepare(desc); - if (err) { - if (desc->vm_file != vma->vm_file) - fput(desc->vm_file); - return err; - } + if (err) + goto err_put; + /* Check the caller did nothing crazy. */ + err = mmap_prepare_validate(&prev_desc, desc); + if (err) + goto err_put; /* Update the VMA from the descriptor. */ compat_set_vma_from_desc(vma, desc); /* Complete any specified mmap actions. */ return mmap_action_complete(vma, &desc->action, /*is_compat=*/true); + +err_put: + if (desc->vm_file != vma->vm_file) + fput(desc->vm_file); + return err; } EXPORT_SYMBOL(__compat_vma_mmap); diff --git a/mm/vma.c b/mm/vma.c index 801155efaff29e..7ee0308dd8be26 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2638,16 +2638,6 @@ static int __mmap_new_file_vma(struct mmap_state *map, return error; } - /* Drivers cannot alter the address of the VMA. */ - WARN_ON_ONCE(map->addr != vma->vm_start); - /* - * Drivers should not permit writability when previously it was - * disallowed. - */ - VM_WARN_ON_ONCE(!vma_flags_same_pair(&map->vma_flags, &vma->flags) && - !vma_flags_test(&map->vma_flags, VMA_MAYWRITE_BIT) && - vma_test(vma, VMA_MAYWRITE_BIT)); - map->vma_flags = vma->flags; return 0; @@ -2725,11 +2715,6 @@ static int __mmap_new_vma(struct mmap_state *map, struct vm_area_struct **vmap, vma->flags = map->vma_flags; } -#ifdef CONFIG_SPARC64 - /* TODO: Fix SPARC ADI! */ - WARN_ON_ONCE(!arch_validate_flags(map->vm_flags)); -#endif - /* Lock the VMA since it is modified after insertion into VMA tree */ vma_start_write(vma); vma_iter_store_new(vmi, vma); @@ -2792,6 +2777,80 @@ static void __mmap_complete(struct mmap_state *map, struct vm_area_struct *vma) vma_set_page_prot(vma); } +/* Check to ensure that the VMA flags of a newly mapped VMA are sane. */ +static int mmap_validate_vma_flags(const vma_flags_t *flags) +{ +#ifdef CONFIG_SPARC64 + const vm_flags_t legacy_flags = vma_flags_to_legacy(*flags); + + /* TODO: Fix SPARC ADI! */ + if (WARN_ON_ONCE(!arch_validate_flags(legacy_flags))) + return -EINVAL; +#endif + + return 0; +} + +/* Check to ensure a driver hasn't done something crazy. */ +static int mmap_validate(unsigned long prev_start, unsigned long prev_end, + unsigned long curr_start, unsigned long curr_end, + const vma_flags_t *prev_flags, + const vma_flags_t *curr_flags) +{ + bool was_maywrite, is_maywrite; + + /* Drivers cannot alter the range of the VMA. */ + if (WARN_ON_ONCE(prev_start != curr_start || prev_end != curr_end)) + return -EINVAL; + + was_maywrite = vma_flags_test(prev_flags, VMA_MAYWRITE_BIT); + is_maywrite = vma_flags_test(curr_flags, VMA_MAYWRITE_BIT); + + /* A driver may not make a previously unwritable mapping writable. */ + if (WARN_ON_ONCE(!was_maywrite && is_maywrite)) + return -EINVAL; + + return mmap_validate_vma_flags(curr_flags); +} + +/** + * mmap_prepare_validate() - Ensure the driver hasn't violated invariants in its + * f_op->mmap_prepare hook. + * @prev_desc: The VMA descriptor prior to the mmap_prepare hook being called. + * @desc: The VMA descriptor after the mmap_prepare hook has been called. + * + * Returns: 0 on success, otherwise an error. + */ +int mmap_prepare_validate(const struct vm_area_desc *prev_desc, + const struct vm_area_desc *desc) +{ + return mmap_validate(prev_desc->start, prev_desc->end, + desc->start, desc->end, + &prev_desc->vma_flags, &desc->vma_flags); +} + +/** + * mmap_hook_validate() - Ensure the driver hasn't violated invariants in + * its f_op->mmap hook. + * @prev_start: The start of the mapping prior to the mmap hook. + * @prev_end: The end of the mapping prior to the mmap hook. + * @prev_flags: The VMA flags set for the VMA prior to the mmap hook. + * @vma: The VMA after the hook has been applied. + * + * Returns: 0 on success, otherwise an error. + */ +int mmap_hook_validate(unsigned long prev_start, unsigned long prev_end, + const vma_flags_t *prev_flags, + const struct vm_area_struct *vma) +{ + const unsigned long start = vma->vm_start; + const unsigned long end = vma->vm_end; + const vma_flags_t *flags = &vma->flags; + + return mmap_validate(prev_start, prev_end, start, end, prev_flags, + flags); +} + static int call_action_prepare(struct mmap_state *map, struct vm_area_desc *desc) { @@ -2818,6 +2877,7 @@ static int call_action_prepare(struct mmap_state *map, static int call_mmap_prepare(struct mmap_state *map, struct vm_area_desc *desc) { + const struct vm_area_desc prev_desc = *desc; int err; /* Invoke the hook. */ @@ -2837,6 +2897,11 @@ static int call_mmap_prepare(struct mmap_state *map, if (err) return err; + /* Check the caller did nothing crazy. */ + err = mmap_prepare_validate(&prev_desc, desc); + if (err) + return err; + /* Update fields permitted to be changed. */ map->pgoff = desc->pgoff; map->vma_flags = desc->vma_flags; @@ -3472,10 +3537,15 @@ int __vm_munmap(unsigned long start, size_t len, bool unlock) int insert_vm_struct(struct mm_struct *mm, struct vm_area_struct *vma) { unsigned long charged = vma_pages(vma); + int err; if (find_vma_intersection(mm, vma->vm_start, vma->vm_end)) return -ENOMEM; + err = mmap_validate_vma_flags(&vma->flags); + if (err) + return err; + if (vma_test(vma, VMA_ACCOUNT_BIT) && security_vm_enough_memory_mm(mm, charged)) return -ENOMEM; diff --git a/mm/vma.h b/mm/vma.h index 4665b40163fa67..b9b99fa02a861d 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -787,14 +787,19 @@ struct vm_area_struct *vm_area_alloc(struct mm_struct *mm); struct vm_area_struct *vm_area_dup(struct vm_area_struct *orig); void vm_area_free(struct vm_area_struct *vma); -/* vma_exec.c */ #ifdef CONFIG_MMU +int mmap_prepare_validate(const struct vm_area_desc *prev_desc, + const struct vm_area_desc *desc); + +int mmap_hook_validate(unsigned long prev_start, unsigned long prev_end, + const vma_flags_t *prev_flags, + const struct vm_area_struct *vma); + +/* vma_exec.c */ int create_init_stack_vma(struct mm_struct *mm, struct vm_area_struct **vmap, unsigned long *top_mem_p); int relocate_vma_down(struct vm_area_struct *vma, unsigned long shift); -#endif -#ifdef CONFIG_MMU /* * Denies creating a writable executable mapping or gaining executable permissions. * @@ -843,6 +848,20 @@ static inline bool map_deny_write_exec(const vma_flags_t *old, return false; } +#else +static inline int mmap_prepare_validate(const struct vm_area_desc *prev_desc, + const struct vm_area_desc *desc) +{ + return 0; +} + +static inline int mmap_hook_validate(unsigned long prev_start, + unsigned long prev_end, + const vma_flags_t *prev_flags, + const struct vm_area_struct *vma) +{ + return 0; +} #endif struct vm_area_struct *__install_special_mapping(struct mm_struct *mm, diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 2fd422789717fe..2986ae6ca1e5ed 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1359,13 +1359,23 @@ static inline int vfs_mmap_prepare(struct file *file, struct vm_area_desc *desc) return file->f_op->mmap_prepare(desc); } +int mmap_prepare_validate(const struct vm_area_desc *prev_desc, + const struct vm_area_desc *desc); + static inline int __compat_vma_mmap(struct vm_area_desc *desc, struct vm_area_struct *vma) { + struct vm_area_desc prev_desc; int err; + /* Derive state prior to mmap_prepare hook. */ + compat_set_desc_from_vma(&prev_desc, desc->file, vma); /* Perform any preparatory tasks for mmap action. */ err = mmap_action_prepare(desc); + if (err) + return err; + /* Check the caller did nothing crazy. */ + err = mmap_prepare_validate(&prev_desc, desc); if (err) return err; /* Update the VMA from the descriptor. */ From 7764867fe7bdff1e3ce0e5f20614ffa3b3cd6b09 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:14 +0100 Subject: [PATCH 0649/1012] mm/vma: ensure mmap_prepare doesn't set actions on a mergeable vma When a user requests an mmap_action be performed in mmap_prepare, this involves populating the VMA range with data. However, if the VMA is mergeable, it might then mistakenly be merged with another VMA without having populated the range. Every mmap action currently available sets VMA flags such that the VMA cannot be merged. However, to ensure that no future mmap action falls foul of this, assert that this is the case upon mmap_prepare validation. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-5-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Suren Baghdasaryan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/vma.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/mm/vma.c b/mm/vma.c index 7ee0308dd8be26..b5f11b01fd9c1a 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2824,6 +2824,15 @@ static int mmap_validate(unsigned long prev_start, unsigned long prev_end, int mmap_prepare_validate(const struct vm_area_desc *prev_desc, const struct vm_area_desc *desc) { + /* + * It is not valid to execute mmap actions for VMAs which can be merged, + * as any such merge would leave portions of the mapping incorrectly + * unmapped. + */ + if (vma_flags_can_merge(&desc->vma_flags) && + WARN_ON_ONCE(desc->action.type != MMAP_NOTHING)) + return -EINVAL; + return mmap_validate(prev_desc->start, prev_desc->end, desc->start, desc->end, &prev_desc->vma_flags, &desc->vma_flags); From 6907ad38d7f7b6f8e2253d2dfd3c37a1606fc4fe Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:15 +0100 Subject: [PATCH 0650/1012] mm: make map_kernel_pages_[prepare,complete] internal and unexported There's no reason to export the symbols for these functions which are only called from internal mm logic, additionally there's no reason for them to be declared in mm.h. This patch therefore removes the exports and moves the declarations to mm/internal.h. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-6-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Suren Baghdasaryan Reviewed-by: Gregory Price (Meta) Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 3 --- mm/internal.h | 3 +++ mm/memory.c | 2 -- 3 files changed, 3 insertions(+), 5 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index b3fdc5e100c48a..ac99e77f0a2230 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4775,9 +4775,6 @@ int remap_pfn_range(struct vm_area_struct *vma, unsigned long addr, int vm_insert_page(struct vm_area_struct *, unsigned long addr, struct page *); int vm_insert_pages(struct vm_area_struct *vma, unsigned long addr, struct page **pages, unsigned long *num); -int map_kernel_pages_prepare(struct vm_area_desc *desc); -int map_kernel_pages_complete(struct vm_area_struct *vma, - struct mmap_action *action); int vm_map_pages(struct vm_area_struct *vma, struct page **pages, unsigned long num); int vm_map_pages_zero(struct vm_area_struct *vma, struct page **pages, diff --git a/mm/internal.h b/mm/internal.h index 88f310cc215078..e75b3fc8ce0041 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -1515,6 +1515,9 @@ int remap_pfn_range_prepare(struct vm_area_desc *desc); int remap_pfn_range_complete(struct vm_area_struct *vma, struct mmap_action *action); int simple_ioremap_prepare(struct vm_area_desc *desc); +int map_kernel_pages_prepare(struct vm_area_desc *desc); +int map_kernel_pages_complete(struct vm_area_struct *vma, + struct mmap_action *action); static inline int io_remap_pfn_range_prepare(struct vm_area_desc *desc) { diff --git a/mm/memory.c b/mm/memory.c index 926276d4192026..448342883e9daa 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2628,7 +2628,6 @@ int map_kernel_pages_prepare(struct vm_area_desc *desc) return 0; } -EXPORT_SYMBOL(map_kernel_pages_prepare); int map_kernel_pages_complete(struct vm_area_struct *vma, struct mmap_action *action) @@ -2640,7 +2639,6 @@ int map_kernel_pages_complete(struct vm_area_struct *vma, action->map_kernel.pages, &nr_pages, vma->vm_page_prot); } -EXPORT_SYMBOL(map_kernel_pages_complete); /** * vm_insert_page - insert single page into user vma From de57d7b9bb36efeb81b09287d3cbb6e5c011a49a Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:16 +0100 Subject: [PATCH 0651/1012] mm/vma: tidy up map kernel pages enum values MMAP_MAP_KERNEL_PAGES is a mouthful, discard the MAP_ as that's implied by MMAP. Also while we're here delete useless comments for mmap actions whose names clearly indicate what they are for. Also update the userland VMA tests to reflect this change. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-7-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Reviewed-by: Suren Baghdasaryan Reviewed-by: Gregory Price (Meta) Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 2 +- include/linux/mm_types.h | 8 ++++---- mm/util.c | 8 ++++---- tools/testing/vma/include/dup.h | 8 ++++---- 4 files changed, 13 insertions(+), 13 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index ac99e77f0a2230..f8c715557ec930 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4630,7 +4630,7 @@ static inline void mmap_action_map_kernel_pages(struct vm_area_desc *desc, { struct mmap_action *action = &desc->action; - action->type = MMAP_MAP_KERNEL_PAGES; + action->type = MMAP_KERNEL_PAGES; action->map_kernel.start = start; action->map_kernel.pages = pages; action->map_kernel.nr_pages = nr_pages; diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 0720a4e98286b2..9b6bbdc8e53f39 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -815,11 +815,11 @@ struct pfnmap_track_ctx { /* What action should be taken after an .mmap_prepare call is complete? */ enum mmap_action_type { - MMAP_NOTHING, /* Mapping is complete, no further action. */ - MMAP_REMAP_PFN, /* Remap PFN range. */ - MMAP_IO_REMAP_PFN, /* I/O remap PFN range. */ + MMAP_NOTHING, + MMAP_REMAP_PFN, + MMAP_IO_REMAP_PFN, MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ - MMAP_MAP_KERNEL_PAGES, /* Map kernel page range from array. */ + MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ }; /* diff --git a/mm/util.c b/mm/util.c index bdd5923eebc7be..438170490e7fd6 100644 --- a/mm/util.c +++ b/mm/util.c @@ -1467,7 +1467,7 @@ int mmap_action_prepare(struct vm_area_desc *desc) return io_remap_pfn_range_prepare(desc); case MMAP_SIMPLE_IO_REMAP: return simple_ioremap_prepare(desc); - case MMAP_MAP_KERNEL_PAGES: + case MMAP_KERNEL_PAGES: return map_kernel_pages_prepare(desc); } @@ -1498,7 +1498,7 @@ int mmap_action_complete(struct vm_area_struct *vma, case MMAP_REMAP_PFN: err = remap_pfn_range_complete(vma, action); break; - case MMAP_MAP_KERNEL_PAGES: + case MMAP_KERNEL_PAGES: err = map_kernel_pages_complete(vma, action); break; case MMAP_IO_REMAP_PFN: @@ -1521,7 +1521,7 @@ int mmap_action_prepare(struct vm_area_desc *desc) case MMAP_REMAP_PFN: case MMAP_IO_REMAP_PFN: case MMAP_SIMPLE_IO_REMAP: - case MMAP_MAP_KERNEL_PAGES: + case MMAP_KERNEL_PAGES: WARN_ON_ONCE(1); /* nommu cannot handle these. */ break; } @@ -1542,7 +1542,7 @@ int mmap_action_complete(struct vm_area_struct *vma, case MMAP_REMAP_PFN: case MMAP_IO_REMAP_PFN: case MMAP_SIMPLE_IO_REMAP: - case MMAP_MAP_KERNEL_PAGES: + case MMAP_KERNEL_PAGES: WARN_ON_ONCE(1); /* nommu cannot handle this. */ err = -EINVAL; diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 2986ae6ca1e5ed..1098655a5f4a3d 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -454,11 +454,11 @@ static __always_inline bool vma_flags_empty(const vma_flags_t *flags) /* What action should be taken after an .mmap_prepare call is complete? */ enum mmap_action_type { - MMAP_NOTHING, /* Mapping is complete, no further action. */ - MMAP_REMAP_PFN, /* Remap PFN range. */ - MMAP_IO_REMAP_PFN, /* I/O remap PFN range. */ + MMAP_NOTHING, + MMAP_REMAP_PFN, + MMAP_IO_REMAP_PFN, MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ - MMAP_MAP_KERNEL_PAGES, /* Map kernel page range from an array. */ + MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ }; /* From be16eeb04d3d660c28a767ab2de781a37d046315 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:17 +0100 Subject: [PATCH 0652/1012] mm: add mmap action for discontiguous kernel page mapping The existing kernel page mapping mmap actions allow for partial and full mapping of an array of struct page pointers. However some drivers require the mapping of discontiguous ranges. Permit this by providing discontig_kernel_page_ops which allows a driver to specify how the operation should begin and how batches of pages should be retrieved. It uses the minimum exposed interface to do so, providing address, page offset and both vm_private_data state and a local private state object. ops->init can establish state for the operation, and ops->get outputs the pages to map and their count. Should an error arise the core unmaps the VMA, invoking vm_ops->close, which is therefore where any state established by ops->init is released. Batches may not exceed the VMA, but may map less than its full range in case the driver wishes to allow the user to map an area larger than the available data. To use it, users invoke mmap_action_map_discontig_kernel_pages() with initial local private state and a set of operations. Users can then use one of the provided helper functions to perform an action: * discontig_kernel_map_abort() - Abort and leave the mapping as it has been accumulated so far. * discontig_kernel_map_page() - Map a single page, or a compound page given its head page. * discontig_kernel_map_page_range() - Maps a struct page ** array of a specified count. The userland VMA tests are updated accordingly. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-8-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 45 +++++++++++++ include/linux/mm_types.h | 44 ++++++++++++- mm/internal.h | 3 + mm/memory.c | 108 ++++++++++++++++++++++++++++++-- mm/util.c | 7 +++ tools/testing/vma/include/dup.h | 11 +++- 6 files changed, 209 insertions(+), 9 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index f8c715557ec930..5602a89156775c 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4654,10 +4654,55 @@ static inline void mmap_action_map_kernel_pages_full(struct vm_area_desc *desc, vma_desc_pages(desc)); } +static inline +void mmap_action_map_discontig_kernel_pages(struct vm_area_desc *desc, + void *init_private, const struct discontig_kernel_page_ops *ops) +{ + struct mmap_action *action = &desc->action; + + action->type = MMAP_DISCONTIG_KERNEL_PAGES; + action->map_kernel_discontig.init_private = init_private; + action->map_kernel_discontig.ops = ops; +} + int mmap_action_prepare(struct vm_area_desc *desc); int mmap_action_complete(struct vm_area_struct *vma, struct mmap_action *action, bool is_compat); +static inline void +discontig_kernel_map_abort(struct discontig_kernel_page_state *state) +{ + state->action = DISCONTIG_KERNEL_PAGE_ABORT; +} + +static inline void +discontig_kernel_map_page(struct discontig_kernel_page_state *state, + struct page *page) +{ + struct folio *folio = page_folio(page); + + if (folio_test_large(folio)) { + VM_WARN_ON_ONCE(page != folio_page(folio, 0)); + state->action = DISCONTIG_KERNEL_PAGE_MAP_COMPOUND_PAGE; + state->__folio = folio; + state->__nr_pages = min(state->nr_pages_remain, + folio_nr_pages(folio)); + } else { + state->action = DISCONTIG_KERNEL_PAGE_MAP_PAGE; + state->__page = page; + state->__nr_pages = 1; + } +} + +static inline void +discontig_kernel_map_page_range(struct discontig_kernel_page_state *state, + struct page **page_arr, unsigned long nr_pages) +{ + state->action = DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE; + state->__page_arr = page_arr; + state->__nr_pages = nr_pages; +} + /* Look up the first VMA which exactly match the interval vm_start ... vm_end */ static inline struct vm_area_struct *find_exact_vma(struct mm_struct *mm, unsigned long vm_start, unsigned long vm_end) diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 9b6bbdc8e53f39..95a768fac987a5 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -818,8 +818,44 @@ enum mmap_action_type { MMAP_NOTHING, MMAP_REMAP_PFN, MMAP_IO_REMAP_PFN, - MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ - MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ + MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ + MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ + MMAP_DISCONTIG_KERNEL_PAGES, /* Map kernel discontig page range. */ +}; + +enum discontig_kernel_page_action { + DISCONTIG_KERNEL_PAGE_ABORT, + DISCONTIG_KERNEL_PAGE_MAP_PAGE, + DISCONTIG_KERNEL_PAGE_MAP_COMPOUND_PAGE, + DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE, +}; + +struct discontig_kernel_page_state { + /* Map state. */ + const unsigned long start; /* Start address of VMA. */ + const unsigned long end; /* End address of VMA. */ + unsigned long addr; /* The current address to be mapped. */ + pgoff_t pgoff; /* The current pgoff to be mapped. */ + unsigned long nr_pages_mapped; /* The number of pages mapped. */ + unsigned long nr_pages_remain; /* The number of pages remaining. */ + + /* User-defined state. */ + void *vm_private_data; /* VMA private data. */ + void *private; /* Mapping private data. */ + + /* Users should not touch these, use discontig_kernel_map_*() helpers. */ + enum discontig_kernel_page_action action; + union { + struct page *__page; + struct folio *__folio; + struct page **__page_arr; + }; + unsigned long __nr_pages; +}; + +struct discontig_kernel_page_ops { + int (*init)(void *vm_private_data, void **private); + int (*get)(struct discontig_kernel_page_state *state); }; /* @@ -844,6 +880,10 @@ struct mmap_action { unsigned long nr_pages; pgoff_t pgoff; } map_kernel; + struct { + void *init_private; + const struct discontig_kernel_page_ops *ops; + } map_kernel_discontig; }; enum mmap_action_type type; diff --git a/mm/internal.h b/mm/internal.h index e75b3fc8ce0041..a1970ff52ad7b9 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -1518,6 +1518,9 @@ int simple_ioremap_prepare(struct vm_area_desc *desc); int map_kernel_pages_prepare(struct vm_area_desc *desc); int map_kernel_pages_complete(struct vm_area_struct *vma, struct mmap_action *action); +int map_discontig_kernel_pages_prepare(struct vm_area_desc *desc); +int map_discontig_kernel_pages_complete(struct vm_area_struct *vma, + struct mmap_action *action); static inline int io_remap_pfn_range_prepare(struct vm_area_desc *desc) { diff --git a/mm/memory.c b/mm/memory.c index 448342883e9daa..45b21bb04a18b9 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2609,17 +2609,23 @@ int vm_insert_pages(struct vm_area_struct *vma, unsigned long addr, } EXPORT_SYMBOL(vm_insert_pages); +static void __map_kernel_pages_prepare(struct vm_area_desc *desc) +{ + if (vma_desc_test(desc, VMA_MIXEDMAP_BIT)) + return; + + VM_WARN_ON_ONCE(mmap_read_trylock(desc->mm)); + VM_WARN_ON_ONCE(vma_desc_test(desc, VMA_PFNMAP_BIT)); + vma_desc_set_flags(desc, VMA_MIXEDMAP_BIT); +} + int map_kernel_pages_prepare(struct vm_area_desc *desc) { const struct mmap_action *action = &desc->action; const unsigned long addr = action->map_kernel.start; unsigned long nr_pages, end; - if (!vma_desc_test(desc, VMA_MIXEDMAP_BIT)) { - VM_WARN_ON_ONCE(mmap_read_trylock(desc->mm)); - VM_WARN_ON_ONCE(vma_desc_test(desc, VMA_PFNMAP_BIT)); - vma_desc_set_flags(desc, VMA_MIXEDMAP_BIT); - } + __map_kernel_pages_prepare(desc); nr_pages = action->map_kernel.nr_pages; end = addr + PAGE_SIZE * nr_pages; @@ -2640,6 +2646,98 @@ int map_kernel_pages_complete(struct vm_area_struct *vma, &nr_pages, vma->vm_page_prot); } +int map_discontig_kernel_pages_prepare(struct vm_area_desc *desc) +{ + const struct mmap_action *action = &desc->action; + const struct discontig_kernel_page_ops *ops = + action->map_kernel_discontig.ops; + + /* At minimum need to be able to get pages. */ + if (WARN_ON_ONCE(!ops || !ops->get)) + return -EINVAL; + + __map_kernel_pages_prepare(desc); + return 0; +} + +static int apply_discontig_action(struct vm_area_struct *vma, + struct discontig_kernel_page_state *state) +{ + unsigned long nr_pages = state->__nr_pages; + unsigned long addr = state->addr; + unsigned long i; + + if (state->action == DISCONTIG_KERNEL_PAGE_MAP_PAGE) + return insert_page(vma, addr, state->__page, + vma->vm_page_prot, /*mkwrite=*/false); + if (state->action == DISCONTIG_KERNEL_PAGE_MAP_PAGE_RANGE) + return insert_pages(vma, addr, state->__page_arr, + &nr_pages, vma->vm_page_prot); + + /* Compound folio - have to iterate through each page. */ + for (i = 0; i < nr_pages; i++, addr += PAGE_SIZE) { + struct page *page = folio_page(state->__folio, i); + int err; + + err = insert_page(vma, addr, page, vma->vm_page_prot, + /*mkwrite=*/false); + if (err) + return err; + } + return 0; +} + +int map_discontig_kernel_pages_complete(struct vm_area_struct *vma, + struct mmap_action *action) +{ + const struct discontig_kernel_page_ops *ops = + action->map_kernel_discontig.ops; + struct discontig_kernel_page_state state = { + .start = vma->vm_start, + .end = vma->vm_end, + .addr = vma->vm_start, + .pgoff = vma->vm_pgoff, + .nr_pages_mapped = 0, + .nr_pages_remain = vma_pages(vma), + .vm_private_data = vma->vm_private_data, + .private = action->map_kernel_discontig.init_private, + }; + int err = 0; + + if (ops->init) + err = ops->init(vma->vm_private_data, &state.private); + if (err) + return err; + + do { + unsigned long end, pgoff_end; + unsigned long nr_pages; + + /* Default to abort. */ + state.action = DISCONTIG_KERNEL_PAGE_ABORT; + err = ops->get(&state); + if (err || state.action == DISCONTIG_KERNEL_PAGE_ABORT) + return err; + nr_pages = state.__nr_pages; + + if (!nr_pages || nr_pages > state.nr_pages_remain) + return -EINVAL; + end = state.addr + PAGE_SIZE * nr_pages; + pgoff_end = state.pgoff + nr_pages; + + err = apply_discontig_action(vma, &state); + if (err) + return err; + + state.addr = end; + state.pgoff = pgoff_end; + state.nr_pages_mapped += nr_pages; + state.nr_pages_remain -= nr_pages; + } while (state.addr < vma->vm_end); + + return 0; +} + /** * vm_insert_page - insert single page into user vma * @vma: user vma to map to diff --git a/mm/util.c b/mm/util.c index 438170490e7fd6..c5ee52aede1e41 100644 --- a/mm/util.c +++ b/mm/util.c @@ -1469,6 +1469,8 @@ int mmap_action_prepare(struct vm_area_desc *desc) return simple_ioremap_prepare(desc); case MMAP_KERNEL_PAGES: return map_kernel_pages_prepare(desc); + case MMAP_DISCONTIG_KERNEL_PAGES: + return map_discontig_kernel_pages_prepare(desc); } WARN_ON_ONCE(1); @@ -1501,6 +1503,9 @@ int mmap_action_complete(struct vm_area_struct *vma, case MMAP_KERNEL_PAGES: err = map_kernel_pages_complete(vma, action); break; + case MMAP_DISCONTIG_KERNEL_PAGES: + err = map_discontig_kernel_pages_complete(vma, action); + break; case MMAP_IO_REMAP_PFN: case MMAP_SIMPLE_IO_REMAP: /* Should have been delegated. */ @@ -1522,6 +1527,7 @@ int mmap_action_prepare(struct vm_area_desc *desc) case MMAP_IO_REMAP_PFN: case MMAP_SIMPLE_IO_REMAP: case MMAP_KERNEL_PAGES: + case MMAP_DISCONTIG_KERNEL_PAGES: WARN_ON_ONCE(1); /* nommu cannot handle these. */ break; } @@ -1543,6 +1549,7 @@ int mmap_action_complete(struct vm_area_struct *vma, case MMAP_IO_REMAP_PFN: case MMAP_SIMPLE_IO_REMAP: case MMAP_KERNEL_PAGES: + case MMAP_DISCONTIG_KERNEL_PAGES: WARN_ON_ONCE(1); /* nommu cannot handle this. */ err = -EINVAL; diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 1098655a5f4a3d..1d5f6b3cbd21e8 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -457,14 +457,17 @@ enum mmap_action_type { MMAP_NOTHING, MMAP_REMAP_PFN, MMAP_IO_REMAP_PFN, - MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ - MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ + MMAP_SIMPLE_IO_REMAP, /* I/O remap with guardrails. */ + MMAP_KERNEL_PAGES, /* Map kernel page range from array. */ + MMAP_DISCONTIG_KERNEL_PAGES, /* Map kernel discontig page range. */ }; /* * Describes an action an mmap_prepare hook can instruct to be taken to complete * the mapping of a VMA. Specified in vm_area_desc. */ +struct discontig_kernel_page_ops; + struct mmap_action { union { struct { @@ -483,6 +486,10 @@ struct mmap_action { unsigned long nr_pages; pgoff_t pgoff; } map_kernel; + struct { + void *init_private; + const struct discontig_kernel_page_ops *ops; + } map_kernel_discontig; }; enum mmap_action_type type; From 4c2bf505060352907a5fd875966aed9d8099ff85 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:18 +0100 Subject: [PATCH 0653/1012] docs: filesystems: update mmap_prepare docs for discontig kernel pgs Describe the newly introduced discontiguous kernel page mapping mechanism, detailing how to use it sensibly and how the API looks. Explicitly detail the various discontiguous actions available and how to use them. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-9-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- Documentation/filesystems/mmap_prepare.rst | 81 ++++++++++++++++++++++ 1 file changed, 81 insertions(+) diff --git a/Documentation/filesystems/mmap_prepare.rst b/Documentation/filesystems/mmap_prepare.rst index 82c99c95ad854e..a476e1006bf126 100644 --- a/Documentation/filesystems/mmap_prepare.rst +++ b/Documentation/filesystems/mmap_prepare.rst @@ -164,5 +164,86 @@ pointer. These are: sufficient entries in the page array to cover the entire range of the described VMA. +* mmap_action_map_discontig_kernel_pages() - Maps a discontiguous range of + `struct page` pointers over the VMA. They must span from the start of the VMA, + but may terminate prior to the end (leaving the remainder unmapped). + **NOTE:** The ``action`` field should never normally be manipulated directly, rather you ought to use one of these helpers. + +Discontiguous Actions +===================== + +Some actions can be performed across discontiguous ranges. + +Map kernel pages +---------------- + +To map kernel pages discontiguously, you must provide hooks using ``struct +discontig_kernel_page_ops``: + +.. code-block:: C + + struct discontig_kernel_page_ops { + int (*init)(void *vm_private_data, void **private); + int (*get)(struct discontig_kernel_page_state *state); + }; + +The ``init`` hook is optional and allows state to be established before the +operation starts, for instance taking a reference count. Nothing is invoked +after the operation, so ``init`` must not leave locks held, and state that must +be released once the mapping goes away should be released in +``vm_ops->close``. + +The ``init`` hook, if provided, is invoked prior to the operation starting. It +may update what is pointed to by ``vm_private_data`` and/or ``private``. If an +error is returned, then the operation is aborted. The ``private`` field can be +reassigned. + +**NOTE:** The operation may sleep between invocations of ``get``, so locks +needed to stabilise state must be taken and released within each hook. + +The ``get`` handler is the key means through which the operation is +executed. The current state of the operation is provided through ``struct +discontig_kernel_page_state``: + +.. code-block:: C + + struct discontig_kernel_page_state { + /* Map state. */ + unsigned long start; /* Start address of VMA. */ + unsigned long end; /* End address of VMA. */ + unsigned long addr; /* The current address to be mapped. */ + pgoff_t pgoff; /* The current pgoff to be mapped. */ + unsigned long nr_pages_mapped; /* The number of pages mapped. */ + unsigned long nr_pages_remain; /* The number of pages remaining. */ + + /* User-defined state. */ + void *vm_private_data; /* VMA private data. */ + void *private; /* Mapping private data. */ + + /* Users should not touch these, use discontig_kernel_map_*() helpers. */ + ... internal fields ... + }; + +With ``private`` being an additional user-controllable state variable, +initialised via ``mmap_action_map_discontig_kernel_pages()``, and +``vm_private_data`` being equal to the ``desc->private_data`` field set in +the ``mmap_prepare()`` hook. + +In the ``get`` hook, the user must choose how to map kernel pages: + +* ``discontig_kernel_map_abort()`` - Call this to abort the operation, whatever + has been mapped so far will be retained, the rest of the mapping will SIGBUS + if accessed. +* ``discontig_kernel_map_page()`` - Maps a single page, correctly handling + compound pages (if the compound page is bigger than the remaining pages in the + VMA, then only those pages that fit will be mapped). For a compound page, the + head page must be passed. +* ``discontig_kernel_map_page_range()`` - Map an array of pages of a specified + size. Note that if the number of pages specified exceeds the VMA size then an + error will arise. + +If an error arises after ``init`` succeeded, the core unmaps the VMA, invoking +``vm_ops->close`` if set, which is therefore the place to release any state +that ``init`` established. From 335a9c6d4dc5366f5d18b440a8938ebbf8d8ef9b Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:19 +0100 Subject: [PATCH 0654/1012] drivers/usb/mon: update to use mmap_prepare + map kernel pages Replace the deprecated .mmap hook with its replacement .mmap_prepare. As part of this change, additionally take the approach of mapping pages upon mmap rather than providing a fault handler. The page span cannot be mutated when an mmap mapping is in place, so this is safe to do in advance (the MON_IOCT_RING_SIZE ioctl operation exits -EBUSY if it's attempted, gated by the rp->mmap_active reference count). Utilise the newly introduced mmap_action_map_discontig_kernel_pages() to do this, which allows for iteration over pages in mon_bin_discontig_get(). mon_bin_discontig_init() increments the rp->mmap_active reference count to stabilise page spans. Should an error arise the core unmaps the VMA and mon_bin_vma_close() drops the reference again. The vm_ops->close hook implemented in mon_bin_vma_close() will ensure correct reference count arithmetic upon unmap (with mon_bin_vma_open() accounting for splitting). The existing semantics are all retained, including not mapping past the range of available pages, with a SIGBUS being raised in a userland process that attempts to access past this point. Ultimately insert_page() is invoked to insert each page, which increments the reference count on each mapped page. This mimics what was being done previously, only we pre-map the entire range rather than doing so on demand. The existing fault handler did nothing that required demand paging, and was presumably implemented this way for historical reasons. One behavioural difference: pages are no longer faulted in on demand, so a page discarded with MADV_DONTNEED is not repopulated and a subsequent access raises SIGBUS, as with other pre-populated kernel mappings. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-10-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Greg Kroah-Hartman Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- drivers/usb/mon/mon_bin.c | 82 +++++++++++++++++++++++++-------------- 1 file changed, 53 insertions(+), 29 deletions(-) diff --git a/drivers/usb/mon/mon_bin.c b/drivers/usb/mon/mon_bin.c index 687f6a8981f34f..9d00b21a8153be 100644 --- a/drivers/usb/mon/mon_bin.c +++ b/drivers/usb/mon/mon_bin.c @@ -1219,6 +1219,15 @@ mon_bin_poll(struct file *file, struct poll_table_struct *wait) return mask; } +static void __mon_bin_vma_open(struct mon_reader_bin *rp) +{ + unsigned long flags; + + spin_lock_irqsave(&rp->b_lock, flags); + rp->mmap_active++; + spin_unlock_irqrestore(&rp->b_lock, flags); +} + /* * open and close: just keep track of how many times the device is * mapped, to use the proper memory allocation function. @@ -1226,64 +1235,79 @@ mon_bin_poll(struct file *file, struct poll_table_struct *wait) static void mon_bin_vma_open(struct vm_area_struct *vma) { struct mon_reader_bin *rp = vma->vm_private_data; - unsigned long flags; - spin_lock_irqsave(&rp->b_lock, flags); - rp->mmap_active++; - spin_unlock_irqrestore(&rp->b_lock, flags); + __mon_bin_vma_open(rp); } -static void mon_bin_vma_close(struct vm_area_struct *vma) +static void __mon_bin_vma_close(struct mon_reader_bin *rp) { unsigned long flags; - struct mon_reader_bin *rp = vma->vm_private_data; spin_lock_irqsave(&rp->b_lock, flags); rp->mmap_active--; spin_unlock_irqrestore(&rp->b_lock, flags); } -/* - * Map ring pages to user space. - */ -static vm_fault_t mon_bin_vma_fault(struct vm_fault *vmf) +static void mon_bin_vma_close(struct vm_area_struct *vma) { - struct mon_reader_bin *rp = vmf->vma->vm_private_data; + struct mon_reader_bin *rp = vma->vm_private_data; + + __mon_bin_vma_close(rp); +} + +static const struct vm_operations_struct mon_bin_vm_ops = { + .open = mon_bin_vma_open, + .close = mon_bin_vma_close, +}; + +static int mon_bin_discontig_init(void *vm_private_data, void **private) +{ + struct mon_reader_bin *rp = vm_private_data; + + /* Dropped by mon_bin_vma_close() on unmap, including on error. */ + __mon_bin_vma_open(rp); + return 0; +} + +static int mon_bin_discontig_get(struct discontig_kernel_page_state *state) +{ + struct mon_reader_bin *rp = state->vm_private_data; unsigned long offset, chunk_idx; - struct page *pageptr; unsigned long flags; spin_lock_irqsave(&rp->b_lock, flags); - offset = vmf->pgoff << PAGE_SHIFT; + + offset = state->pgoff << PAGE_SHIFT; if (offset >= rp->b_size) { spin_unlock_irqrestore(&rp->b_lock, flags); - return VM_FAULT_SIGBUS; + discontig_kernel_map_abort(state); + return 0; } chunk_idx = offset / CHUNK_SIZE; - pageptr = rp->b_vec[chunk_idx].pg; - get_page(pageptr); - vmf->page = pageptr; + discontig_kernel_map_page(state, rp->b_vec[chunk_idx].pg); + spin_unlock_irqrestore(&rp->b_lock, flags); return 0; } -static const struct vm_operations_struct mon_bin_vm_ops = { - .open = mon_bin_vma_open, - .close = mon_bin_vma_close, - .fault = mon_bin_vma_fault, +static const struct discontig_kernel_page_ops mon_discontig_ops = { + .init = mon_bin_discontig_init, + .get = mon_bin_discontig_get, }; -static int mon_bin_mmap(struct file *filp, struct vm_area_struct *vma) +static int mon_bin_mmap_prepare(struct vm_area_desc *desc) { - /* don't do anything here: "fault" will set up page table entries */ - vma->vm_ops = &mon_bin_vm_ops; + const struct file *filp = desc->file; - if (vma->vm_flags & VM_WRITE) + if (vma_desc_test(desc, VMA_WRITE_BIT)) return -EPERM; - vm_flags_mod(vma, VM_DONTEXPAND | VM_DONTDUMP, VM_MAYWRITE); - vma->vm_private_data = filp->private_data; - mon_bin_vma_open(vma); + desc->vm_ops = &mon_bin_vm_ops; + vma_desc_clear_flags(desc, VMA_MAYWRITE_BIT); + vma_desc_set_flags(desc, VMA_DONTEXPAND_BIT, VMA_DONTDUMP_BIT); + desc->private_data = filp->private_data; + + mmap_action_map_discontig_kernel_pages(desc, NULL, &mon_discontig_ops); return 0; } @@ -1298,7 +1322,7 @@ static const struct file_operations mon_fops_binary = { .compat_ioctl = mon_bin_compat_ioctl, #endif .release = mon_bin_release, - .mmap = mon_bin_mmap, + .mmap_prepare = mon_bin_mmap_prepare, }; static int mon_bin_wait_event(struct file *file, struct mon_reader_bin *rp) From 68dc61238688df49b69ddd946df7e782819fbc56 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:20 +0100 Subject: [PATCH 0655/1012] infiniband: update hfi1 to use remap_vmalloc_range() In cases which map chip memory from vmalloc()'d ranges, the hfi1 infiniband drivers currently installs a fault handler, and then smuggles the kernel virtual address of this range in vma->vm_pgoff. This is exposing KASLR-sensitive internal kernel state in the VMA, and is entirely unnecessary. Instead, use remap_vmalloc_range() to remap the VMA to the span, and eliminate the fault handler altogether. remap_vmalloc_range() checks that the VMA does not extend beyond the vmalloc area, and the driver already requires the VMA to exactly match the span of the memory being mapped, so this has no impact. The memory is all preallocated so not having a fault handler has no impact either, other than pre-mapping the ranges which is beneficial. We also remove the VM_IO flag as it's not appropriate here, and the VM_DONTEXPAND flag as remap_vmalloc_range() will set it (and also mark the range correctly as a mixed map). We also update the vmalloc paths to place the virtual kernel address in memvirt, rather than overloading the physical address memaddr. We predicate the vmalloc handling on the vmalloc flag before we check memvirt for the virtual address-derived PFN remap path, so this works fine. remap_vmalloc_range() requires that the vmalloc()'d areas were all allocated using vmalloc_user() - each of cq->comps, uctxt->subctxt_rcvegrbuf, uctxt->subctxt_rcvhdr_base, uctxt->subctxt_uregbase and dd->events were allocated this way, so that requirement is satisfied. We also remove VM_IO and VM_DONTEXPAND from the STATUS command, as these are both set on remap. Finally, we remove VM_DONTEXPAND from the PIO_BUFS, PIO_BUFS_SOP and UREGS commands, as these are also all set on remap. PIO_CRED retains it, as dma_mmap_coherent() may map via vm_insert_page() on the IOMMU-DMA path, which sets only VM_MIXEDMAP. The RCV_HDRQ, RCV_EGRBUF and RTAIL commands also map via dma_mmap_coherent() and never set VM_DONTEXPAND, so set it for them for the same reason. Note that we retain expected behaviour throughout - the vmalloc remapped ranges set VM_MIXEDMAP | VM_DONTDUMP | VM_DONTEXPAND for each range. VM_IO was never appropriate as the ranges are explicitly not MMIO, and the reference to the v3.7 VM_RESERVED semantics map on to VM_MIXEDMAP | VM_DONTDUMP | VM_DONTEXPAND correctly - no core dump, unmergeable, no normal vm page for purposes of reclaim/migration/etc. There is a change in behaviour in that pages mapped using remap_vmalloc_range() will now have normal GUP-able pages, however this should have no impact as there is no reason not to allow this. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-11-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- drivers/infiniband/hw/hfi1/file_ops.c | 84 +++++++++------------------ 1 file changed, 26 insertions(+), 58 deletions(-) diff --git a/drivers/infiniband/hw/hfi1/file_ops.c b/drivers/infiniband/hw/hfi1/file_ops.c index 1a36f995c4f6db..0fc2ba5f1f9113 100644 --- a/drivers/infiniband/hw/hfi1/file_ops.c +++ b/drivers/infiniband/hw/hfi1/file_ops.c @@ -70,7 +70,6 @@ static int set_ctxt_pkey(struct hfi1_ctxtdata *uctxt, unsigned long arg); static int ctxt_reset(struct hfi1_ctxtdata *uctxt); static int manage_rcvq(struct hfi1_ctxtdata *uctxt, u16 subctxt, unsigned long arg); -static vm_fault_t vma_fault(struct vm_fault *vmf); static long hfi1_file_ioctl(struct file *fp, unsigned int cmd, unsigned long arg); @@ -85,10 +84,6 @@ static const struct file_operations hfi1_file_ops = { .llseek = noop_llseek, }; -static const struct vm_operations_struct vm_ops = { - .fault = vma_fault, -}; - /* * Types of memories mapped into user processes' space */ @@ -304,13 +299,13 @@ static ssize_t hfi1_write_iter(struct kiocb *kiocb, struct iov_iter *from) return reqs; } -static inline void mmap_cdbg(u16 ctxt, u8 subctxt, u8 type, u8 mapio, u8 vmf, +static inline void mmap_cdbg(u16 ctxt, u8 subctxt, u8 type, u8 mapio, u8 is_vmalloc, u64 memaddr, void *memvirt, dma_addr_t memdma, ssize_t memlen, struct vm_area_struct *vma) { hfi1_cdbg(PROC, - "%u:%u type:%u io/vf/dma:%d/%d/%d, addr:0x%llx, len:%lu(%lu), flags:0x%lx", - ctxt, subctxt, type, mapio, vmf, !!memdma, + "%u:%u type:%u io/vmalloc/dma:%d/%d/%d, addr:0x%llx, len:%lu(%lu), flags:0x%lx", + ctxt, subctxt, type, mapio, is_vmalloc, !!memdma, memaddr ?: (u64)memvirt, memlen, vma->vm_end - vma->vm_start, vma->vm_flags); } @@ -325,7 +320,8 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) memaddr = 0; void *memvirt = NULL; dma_addr_t memdma = 0; - u8 subctxt, mapio = 0, vmf = 0, type; + u8 subctxt, mapio = 0, type; + u8 is_vmalloc = 0; size_t memdmalen = 0; ssize_t memlen = 0; int ret = 0; @@ -348,7 +344,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) /* * vm_pgoff is used as a buffer selector cookie. Always mmap from * the beginning. - */ + */ vma->vm_pgoff = 0; flags = vma->vm_flags; @@ -367,7 +363,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) */ memlen = PAGE_ALIGN(uctxt->sc->credits * PIO_BLOCK_SIZE); flags &= ~VM_MAYREAD; - flags |= VM_DONTCOPY | VM_DONTEXPAND; + flags |= VM_DONTCOPY; vma->vm_page_prot = pgprot_writecombine(vma->vm_page_prot); mapio = 1; break; @@ -411,6 +407,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) memlen = rcvhdrq_size(uctxt); memvirt = uctxt->rcvhdrq; memdma = uctxt->rcvhdrq_dma; + flags |= VM_DONTEXPAND; break; case RCV_EGRBUF: { unsigned long vm_start_save; @@ -432,7 +429,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) ret = -EPERM; goto done; } - vm_flags_clear(vma, VM_MAYWRITE); + vm_flags_mod(vma, VM_DONTEXPAND, VM_MAYWRITE); /* * Mmap multiple separate allocations into a single vma. From * here, dma_mmap_coherent() calls dma_direct_mmap(), which @@ -448,7 +445,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) memvirt = uctxt->egrbufs.buffers[i].addr; memdma = uctxt->egrbufs.buffers[i].dma; vma->vm_end += memlen; - mmap_cdbg(ctxt, subctxt, type, mapio, vmf, memaddr, + mmap_cdbg(ctxt, subctxt, type, mapio, is_vmalloc, memaddr, memvirt, memdma, memlen, vma); ret = dma_mmap_coherent(&dd->pcidev->dev, vma, memvirt, memdma, memlen); @@ -477,7 +474,7 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) * user registers. */ memlen = PAGE_SIZE; - flags |= VM_DONTCOPY | VM_DONTEXPAND; + flags |= VM_DONTCOPY; vma->vm_page_prot = pgprot_noncached(vma->vm_page_prot); mapio = 1; break; @@ -486,15 +483,10 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) * Use the page where this context's flags are. User level * knows where it's own bitmap is within the page. */ - memaddr = (unsigned long) - (dd->events + uctxt_offset(uctxt)) & PAGE_MASK; + memvirt = dd->events + uctxt_offset(uctxt); + memvirt = (void *)(((uintptr_t)memvirt) & PAGE_MASK); memlen = PAGE_SIZE; - /* - * v3.7 removes VM_RESERVED but the effect is kept by - * using VM_IO. - */ - flags |= VM_IO | VM_DONTEXPAND; - vmf = 1; + is_vmalloc = 1; break; case STATUS: if (flags & VM_WRITE) { @@ -503,7 +495,6 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) } memaddr = kvirt_to_phys((void *)dd->status); memlen = PAGE_SIZE; - flags |= VM_IO | VM_DONTEXPAND; break; case RTAIL: if (!HFI1_CAP_IS_USET(DMA_RTAIL)) { @@ -522,25 +513,23 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) memvirt = (void *)hfi1_rcvhdrtail_kvaddr(uctxt); memdma = uctxt->rcvhdrqtailaddr_dma; flags &= ~VM_MAYWRITE; + flags |= VM_DONTEXPAND; break; case SUBCTXT_UREGS: - memaddr = (u64)uctxt->subctxt_uregbase; + memvirt = uctxt->subctxt_uregbase; memlen = PAGE_SIZE; - flags |= VM_IO | VM_DONTEXPAND; - vmf = 1; + is_vmalloc = 1; break; case SUBCTXT_RCV_HDRQ: - memaddr = (u64)uctxt->subctxt_rcvhdr_base; + memvirt = uctxt->subctxt_rcvhdr_base; memlen = rcvhdrq_size(uctxt) * uctxt->subctxt_cnt; - flags |= VM_IO | VM_DONTEXPAND; - vmf = 1; + is_vmalloc = 1; break; case SUBCTXT_EGRBUF: - memaddr = (u64)uctxt->subctxt_rcvegrbuf; + memvirt = uctxt->subctxt_rcvegrbuf; memlen = uctxt->egrbufs.size * uctxt->subctxt_cnt; - flags |= VM_IO | VM_DONTEXPAND; flags &= ~VM_MAYWRITE; - vmf = 1; + is_vmalloc = 1; break; case SDMA_COMP: { struct hfi1_user_sdma_comp_q *cq = fd->cq; @@ -549,10 +538,9 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) ret = -EFAULT; goto done; } - memaddr = (u64)cq->comps; + memvirt = cq->comps; memlen = PAGE_ALIGN(sizeof(*cq->comps) * cq->nentries); - flags |= VM_IO | VM_DONTEXPAND; - vmf = 1; + is_vmalloc = 1; break; } default: @@ -569,12 +557,10 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) } vm_flags_reset(vma, flags); - mmap_cdbg(ctxt, subctxt, type, mapio, vmf, memaddr, memvirt, memdma, + mmap_cdbg(ctxt, subctxt, type, mapio, is_vmalloc, memaddr, memvirt, memdma, memlen, vma); - if (vmf) { - vma->vm_pgoff = PFN_DOWN(memaddr); - vma->vm_ops = &vm_ops; - ret = 0; + if (is_vmalloc) { + ret = remap_vmalloc_range(vma, memvirt, 0); } else if (memdma) { ret = dma_mmap_coherent(&dd->pcidev->dev, vma, memvirt, memdma, @@ -599,24 +585,6 @@ static int hfi1_file_mmap(struct file *fp, struct vm_area_struct *vma) return ret; } -/* - * Local (non-chip) user memory is not mapped right away but as it is - * accessed by the user-level code. - */ -static vm_fault_t vma_fault(struct vm_fault *vmf) -{ - struct page *page; - - page = vmalloc_to_page((void *)(vmf->pgoff << PAGE_SHIFT)); - if (!page) - return VM_FAULT_SIGBUS; - - get_page(page); - vmf->page = page; - - return 0; -} - static __poll_t hfi1_poll(struct file *fp, struct poll_table_struct *pt) { struct hfi1_ctxtdata *uctxt; From 40342064fa3a736bcbbc74cb397b9618672916c3 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:21 +0100 Subject: [PATCH 0656/1012] selinux: reject writable opens of policy file, drop mmap shared/write check The policy file has no write method and is exposed read-only (S_IRUGO in selinux_files[]), yet sel_open_policy() performs no open mode check, so a CAP_DAC_OVERRIDE caller can open it O_RDWR. Reject FMODE_WRITE at open, as kernfs does. The file can then never be mapped with FMODE_WRITE, so do_mmap() always clears VM_MAYWRITE and VM_SHARED for MAP_SHARED mappings and the VM_SHARED check in sel_mmap_policy() cannot be reached. Remove it. This also stops sel_mmap_policy() clearing VM_MAYWRITE on a mapping that is neither a PFN map nor a mixed map, ahead of the core enforcing that only such mappings may do so. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-12-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Stephen Smalley Reviewed-by: Jann Horn Acked-by: Paul Moore Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- security/selinux/selinuxfs.c | 11 +++-------- 1 file changed, 3 insertions(+), 8 deletions(-) diff --git a/security/selinux/selinuxfs.c b/security/selinux/selinuxfs.c index c7d91476971cb5..545a6f89f9e763 100644 --- a/security/selinux/selinuxfs.c +++ b/security/selinux/selinuxfs.c @@ -340,6 +340,9 @@ static int sel_open_policy(struct inode *inode, struct file *filp) struct policy_load_memory *plm = NULL; int rc; + if (filp->f_mode & FMODE_WRITE) + return -EACCES; + rc = avc_has_perm(current_sid(), SECINITSID_SECURITY, SECCLASS_SECURITY, SECURITY__READ_POLICY, NULL); if (rc) @@ -424,14 +427,6 @@ static const struct vm_operations_struct sel_mmap_policy_ops = { static int sel_mmap_policy(struct file *filp, struct vm_area_struct *vma) { - if (vma->vm_flags & VM_SHARED) { - /* do not allow mprotect to make mapping writable */ - vm_flags_clear(vma, VM_MAYWRITE); - - if (vma->vm_flags & VM_WRITE) - return -EACCES; - } - vm_flags_set(vma, VM_DONTEXPAND | VM_DONTDUMP); vma->vm_ops = &sel_mmap_policy_ops; From 59ff900ebeb6796dbdb7fe728352b48299411e00 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:22 +0100 Subject: [PATCH 0657/1012] ALSA: pcm: use vm_insert_page() to map PCM status page There's no need to keep a fault handler around for this, instead map on mmap. While we're here, rename area to vma to be consistent. This correctly makes the mapping a mixed map mapping. This works towards establishing the invariant that only PFN mapped or mixed map mappings may clear the VM_MAYWRITE flag. The status page mapping clears VM_MAYWRITE, so it must be kernel-owned; the control page mapping remains writable and is left fault-based. The assumption is made that the struct pcm_mmap_status structure is at most a page in size, which is asserted as a build bug. This is safe to assume, as the size of the structure is 56 bytes at most. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-13-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Takashi Iwai Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- sound/core/pcm_native.c | 38 +++++++++++++------------------------- 1 file changed, 13 insertions(+), 25 deletions(-) diff --git a/sound/core/pcm_native.c b/sound/core/pcm_native.c index 6d32c12fb79bdf..af80488baaf427 100644 --- a/sound/core/pcm_native.c +++ b/sound/core/pcm_native.c @@ -3760,39 +3760,27 @@ static __poll_t snd_pcm_poll(struct file *file, poll_table *wait) /* * mmap status record */ -static vm_fault_t snd_pcm_mmap_status_fault(struct vm_fault *vmf) +static int snd_pcm_mmap_status(struct snd_pcm_substream *substream, struct file *file, + struct vm_area_struct *vma) { - struct snd_pcm_substream *substream = vmf->vma->vm_private_data; + const unsigned long size = vma->vm_end - vma->vm_start; struct snd_pcm_runtime *runtime; - - if (substream == NULL) - return VM_FAULT_SIGBUS; - runtime = substream->runtime; - vmf->page = virt_to_page(runtime->status); - get_page(vmf->page); - return 0; -} + struct page *page; -static const struct vm_operations_struct snd_pcm_vm_ops_status = -{ - .fault = snd_pcm_mmap_status_fault, -}; + BUILD_BUG_ON(sizeof(struct snd_pcm_mmap_status) > PAGE_SIZE); -static int snd_pcm_mmap_status(struct snd_pcm_substream *substream, struct file *file, - struct vm_area_struct *area) -{ - long size; - if (!(area->vm_flags & VM_READ)) + if (!(vma->vm_flags & VM_READ)) return -EINVAL; - size = area->vm_end - area->vm_start; - if (size != PAGE_ALIGN(sizeof(struct snd_pcm_mmap_status))) + if (size != PAGE_SIZE) return -EINVAL; - area->vm_ops = &snd_pcm_vm_ops_status; - area->vm_private_data = substream; - vm_flags_mod(area, VM_DONTEXPAND | VM_DONTDUMP, + + vm_flags_mod(vma, VM_DONTEXPAND | VM_DONTDUMP, VM_WRITE | VM_MAYWRITE); + vma->vm_page_prot = vm_get_page_prot(vma->vm_flags); - return 0; + runtime = substream->runtime; + page = virt_to_page(runtime->status); + return vm_insert_page(vma, vma->vm_start, page); } /* From 2473323484fca5b8862e104cc172199a1c3a5ab3 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:23 +0100 Subject: [PATCH 0658/1012] bpf: arena: mark arena_map_mmap() mappings VM_MIXEDMAP The bpf_map->ops->map_mmap callback invoked by bpf_map_mmap() can be set to one of ringbuf_map_mmap_kern(), ringbuf_map_mmap_user(), array_map_mmap() or arena_map_mmap(). It is convention in mm to mark mappings whose pages the kernel manages itself with VM_MIXEDMAP, so the core mm knows not to treat them as ordinary page cache or anonymous memory. The map_mmap callbacks ringbuf_map_mmap_kern() and ringbuf_map_mmap_user() use remap_vmalloc_range(), which ultimately invokes vm_insert_page() and so marks the ranges VM_MIXEDMAP, and array_map_mmap() sets VM_MIXEDMAP explicitly. However, the exception to this is arena_map_mmap(), which doesn't set the flag. This patch corrects this and updates the comment to reflect it. The pages are refcounted and vm_normal_page() finds them regardless of the flag, and VM_DONTEXPAND remains set (marking the memory as VM_SPECIAL and thus unmergeable). The one effect is that NUMA balancing now skips these VMAs, as it already does for the other bpf map mappings, which is the reason array_map_mmap() gives for setting the flag. The intent of this patch is to be able to establish the invariant that only PFN-mapped or mixed map ranges may clear the VM_MAYWRITE flag, as is done in bpf_map_mmap(). Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-14-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Emil Tsalapatis Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- kernel/bpf/arena.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/kernel/bpf/arena.c b/kernel/bpf/arena.c index 7b6847200b4312..b69fe5e343393d 100644 --- a/kernel/bpf/arena.c +++ b/kernel/bpf/arena.c @@ -620,8 +620,9 @@ static int arena_map_mmap(struct bpf_map *map, struct vm_area_struct *vma) * clears VM_MAYEXEC. Set VM_DONTEXPAND to avoid potential change * of user_vm_start. Set VM_DONTCOPY to prevent arena VMA from * being copied into the child process on fork. + * This is a kernel page so set VM_MIXEDMAP. */ - vm_flags_set(vma, VM_DONTEXPAND | VM_DONTCOPY); + vm_flags_set(vma, VM_MIXEDMAP | VM_DONTEXPAND | VM_DONTCOPY); vma->vm_ops = &arena_vm_ops; return 0; } From 3eb6def8e13fc2b58f3244fed179d478be59459d Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:24 +0100 Subject: [PATCH 0659/1012] mm/vma: add vma[_flags]_is_kernel_owned() predicates Rather than referring to VMA flags with uncertain meaning, add a new predicate that explicitly describes what possession of the VMA_PFNMAP_BIT or VMA_MIXEDMAP_BIT flags mean, and then refer to that function for determining VMA mergeability. Either flag means the contents of the mapping are owned by the kernel, usually a driver, rather than by the core mm: the memory may be MMIO, kernel-allocated pages or even ordinary pages the driver maps itself, but the core must not populate, reclaim, migrate, copy-on-write or merge the range on its own initiative. We initially also include VMA_IO_BIT here, as by implication, these must be kernel-owned. (mlock() also sets VMA_IO_BIT transiently on ordinary VMAs while locking them, which is addressed later in this series.) However the intent is to in future remove this, as no mapping should be marked as an I/O mapping without also being marked with VMA_PFNMAP_BIT. This forms the basis of further work intended to improve how we express VMA properties such as this. Also update the VMA userland tests to reflect the change. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-15-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 56 ++++++++++++++++++++++++++++++++- tools/testing/vma/include/dup.h | 29 ++++++++++++++++- 2 files changed, 83 insertions(+), 2 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 5602a89156775c..4f235e0ec385fb 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1613,6 +1613,44 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) return is_shared_maywrite(&vma->flags); } +/** + * vma_flags_is_kernel_owned() - Do the specified VMA flags indicate that the + * contents of the VMA are owned by the kernel rather than the core mm? + * @flags: The VMA flags to test. + * + * A kernel-owned mapping is one whose contents are established and controlled + * by the kernel, typically a driver, rather than by the core mm's fault and + * rmap machinery. + * + * The mapping may be memory-mapped I/O, kernel-allocated pages or ordinary + * pages the owner has chosen to map itself (shmem via a PFN map, for instance). + * + * In all cases the core mm must not populate, reclaim, migrate, copy-on-write + * or merge it of its own accord. + * + * Pages mapped this way are not necessarily reference counted or map counted. + * + * Returns: true if the flags indicate a kernel-owned mapping. + */ +static inline bool vma_flags_is_kernel_owned(const vma_flags_t *flags) +{ + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT, + VMA_IO_BIT); +} + +/** + * vma_is_kernel_owned() - Are the contents of @vma owned by the kernel? + * @vma: The VMA to test. + * + * See vma_flags_is_kernel_owned() for a description of this property. + * + * Returns: true if the VMA is kernel-owned. + */ +static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma) +{ + return vma_flags_is_kernel_owned(&vma->flags); +} + /** * vma_flags_can_merge() - Do the specified VMA flags permit the VMA to be * merged with another? @@ -1621,7 +1659,23 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) */ static inline bool vma_flags_can_merge(const vma_flags_t *flags) { - return !vma_flags_test_any_mask(flags, VMA_SPECIAL_FLAGS); + /* + * VMA merging assumes that a VMA's flags and fields completely describe + * its state. + * + * However, kernel-owned mappings may have established state upon mapping + * not embodied in any attribute of the VMA. + * + * Additionally, private (CoW) PFN maps encode the source PFN of the + * range in vma->vm_pgoff, which may otherwise cause spurious merges. + */ + if (vma_flags_is_kernel_owned(flags)) + return false; + /* VMA explicitly marked as being unmergeable. */ + if (vma_flags_test(flags, VMA_DONTEXPAND_BIT)) + return false; + + return true; } /** diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 1d5f6b3cbd21e8..d09148ce23054d 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1665,7 +1665,34 @@ static inline bool file_is_dev_zero(const struct file *file) return file && file->f_op == &zero_fops; } +static inline bool vma_flags_is_kernel_owned(const vma_flags_t *flags) +{ + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT, + VMA_IO_BIT); +} + +static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma) +{ + return vma_flags_is_kernel_owned(&vma->flags); +} + static inline bool vma_flags_can_merge(const vma_flags_t *flags) { - return !vma_flags_test_any_mask(flags, VMA_SPECIAL_FLAGS); + /* + * VMA merging assumes that a VMA's flags and fields completely describe + * its state. + * + * However, kernel-owned mappings may have established state upon mapping + * not embodied in any attribute of the VMA. + * + * Additionally, private (CoW) PFN maps encode the source PFN of the + * range in vma->vm_pgoff, which may otherwise cause spurious merges. + */ + if (vma_flags_is_kernel_owned(flags)) + return false; + /* VMA explicitly marked as being unmergeable. */ + if (vma_flags_test(flags, VMA_DONTEXPAND_BIT)) + return false; + + return true; } From b2a0a6f34d2096d1ff81853847ed4d0a4fcdda11 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:25 +0100 Subject: [PATCH 0660/1012] mm/vma: only allow mmap to clear VMA_MAYWRITE_BIT if kernel-owned For ordinary files the only way the VMA_MAYWRITE_BIT flag is cleared is if the underlying file is itself read-only. This means that mprotect() cannot mark a shared mapping of a read-only file as read/write, as doing so would violate the read only attribute, and permit writes. In general, we do not want file systems to be able to do this for read/write files. Doing so would violate fundamental user expectation of file attributes and likely break userspace. However, drivers pose a tricky problem here - the /dev/xxx file may be read/write but provide access to a resource which is fundamentally read-only. Therefore we must allow drivers to be able to clear VMA_MAYWRITE_BIT. To achieve both of these things, restrict this ability to kernel-owned mappings as identified by vma_flags_is_kernel_owned(). This constrains this ability to drivers which own the mapping's contents, whether memory-mapped I/O, kernel-allocated pages, or ordinary pages they map themselves, and so define its semantics. Every in-tree mmap hook which clears VMA_MAYWRITE_BIT, some twenty sites across drivers, filesystems and bpf, establishes a kernel-owned mapping, with usbmon and the ALSA PCM status page converted earlier in this series to do so. Note that drivers may, if they do not gate on VMA_SHARED_BIT, be able to disable MAP_PRIVATE-file-backed mapping CoW semantics. This is perhaps not always intended, but we retain this capacity to maintain existing behaviour. As all drivers which clear VMA_MAYWRITE_BIT establish kernel-owned mappings, no functional change is intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-16-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/vma.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/mm/vma.c b/mm/vma.c index b5f11b01fd9c1a..d9db60c418f281 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2810,6 +2810,11 @@ static int mmap_validate(unsigned long prev_start, unsigned long prev_end, if (WARN_ON_ONCE(!was_maywrite && is_maywrite)) return -EINVAL; + /* Only kernel-owned mappings may clear VMA_MAYWRITE_BIT. */ + if (!vma_flags_is_kernel_owned(curr_flags) && + WARN_ON_ONCE(was_maywrite && !is_maywrite)) + return -EINVAL; + return mmap_validate_vma_flags(curr_flags); } From fe7f1c86c69ddf0695cb99f8a24d6075e92bb1cd Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:26 +0100 Subject: [PATCH 0661/1012] mm/vma: add and use vma_[flags]_is_fixed_mapping This determines whether a VMA cannot be expanded or merged because what they mapped was determined to be a set size at mmap time. This typically refers to kernel-owned mappings, however VMA_DONTEXPAND_BIT is not reliably set alongside VMA_PFNMAP_BIT or VMA_MIXEDMAP_BIT, so we must explicitly test for this for now. We also explicitly test for VMA_PFNMAP_BIT as VMA_DONTEXPAND_BIT may not be set for VMA_PFNMAP_BIT's despite the one implying the other. Use this predicate in vma_flags_can_merge() and in check_prep_vma() in the mremap logic testing to see if mremap() can expand the VMA. The criteria for khugepaged and MADV_COLLAPSE eligibility in __thp_vma_allowable_orders() are precisely those for mergeability, so use vma_can_merge() there (with an expanded comment). This obviates the need for the VM_NO_KHUGEPAGED mask, so remove it. Hugetlb VMAs remain excluded from khugepaged as hugetlbfs always sets VMA_DONTEXPAND_BIT. Also update the userland VMA tests to reflect the change. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-17-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 39 +++++++++++++++++++++++++++++---- mm/huge_memory.c | 11 ++++++---- mm/mremap.c | 5 ++--- tools/testing/vma/include/dup.h | 16 +++++++++++++- 4 files changed, 59 insertions(+), 12 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 4f235e0ec385fb..a7fa4df6fd4747 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -601,9 +601,6 @@ enum { #define VMA_REMAP_FLAGS mk_vma_flags(VMA_IO_BIT, VMA_PFNMAP_BIT, \ VMA_DONTEXPAND_BIT, VMA_DONTDUMP_BIT) -/* This mask prevents VMA from being scanned with khugepaged */ -#define VM_NO_KHUGEPAGED (VM_SPECIAL | VM_HUGETLB) - /* This mask defines which mm->def_flags a process can inherit its parent */ #define VM_INIT_DEF_MASK VM_NOHUGEPAGE @@ -1651,6 +1648,40 @@ static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma) return vma_flags_is_kernel_owned(&vma->flags); } +/** + * vma_flags_is_fixed_mapping() - Do the specified VMA flags indicate that this + * is a fixed mapping that cannot be expanded or merged? + * @flags: The VMA flags to test. + * + * Fixed mappings are those whose size is set at the point of mmap (for + * instance, a kernel-owned mapping of a fixed range of memory), and thus + * cannot be expanded or merged. + * + * Returns: true if the flags indicate a fixed mapping. + */ +static inline bool vma_flags_is_fixed_mapping(const vma_flags_t *flags) +{ + /* + * VMA_PFNMAP_BIT should imply VMA_DONTEXPAND_BIT, but some callers set + * only the former. + */ + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_DONTEXPAND_BIT); +} + +/** + * vma_is_fixed_mapping() - Is this VMA a fixed mapping that cannot be + * expanded or merged? + * @vma: The VMA to test. + * + * See vma_flags_is_fixed_mapping() for a description of this property. + * + * Returns: true if the VMA maps a fixed mapping. + */ +static inline bool vma_is_fixed_mapping(const struct vm_area_struct *vma) +{ + return vma_flags_is_fixed_mapping(&vma->flags); +} + /** * vma_flags_can_merge() - Do the specified VMA flags permit the VMA to be * merged with another? @@ -1672,7 +1703,7 @@ static inline bool vma_flags_can_merge(const vma_flags_t *flags) if (vma_flags_is_kernel_owned(flags)) return false; /* VMA explicitly marked as being unmergeable. */ - if (vma_flags_test(flags, VMA_DONTEXPAND_BIT)) + if (vma_flags_is_fixed_mapping(flags)) return false; return true; diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 8f8bf60a22649a..6ee21854b6dfa9 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -212,11 +212,14 @@ unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma, return in_pf ? orders : 0; /* - * khugepaged special VMA and hugetlb VMA. - * Must be checked after dax since some dax mappings may have - * VM_MIXEDMAP set. + * khugepaged moves data from VMAs once collapsed, after they have been + * faulted in, relying on refaulting for file-backed memory. + * + * Kernel-owned mappings cannot be reliably reconstructed from page + * faults, and fixed mappings (including hugetlb) may not be marked as + * kernel-owned - precisely the mappings which cannot be merged. */ - if (!in_pf && !smaps && (vm_flags & VM_NO_KHUGEPAGED)) + if (!in_pf && !smaps && !vma_can_merge(vma)) return 0; /* diff --git a/mm/mremap.c b/mm/mremap.c index a06a9bf2de1d70..c5ee24784dfa67 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -1811,8 +1811,7 @@ static int check_prep_vma(struct vma_remap_struct *vrm) return -EINVAL; } - if ((vrm->flags & MREMAP_DONTUNMAP) && - vma_test_any(vma, VMA_DONTEXPAND_BIT, VMA_PFNMAP_BIT)) + if ((vrm->flags & MREMAP_DONTUNMAP) && vma_is_fixed_mapping(vma)) return -EINVAL; /* @@ -1850,7 +1849,7 @@ static int check_prep_vma(struct vma_remap_struct *vrm) if (pgoff + (new_len >> PAGE_SHIFT) < pgoff) return -EINVAL; - if (vma_test_any(vma, VMA_DONTEXPAND_BIT, VMA_PFNMAP_BIT)) + if (vma_is_fixed_mapping(vma)) return -EFAULT; if (!mlock_future_ok(mm, vma_test(vma, VMA_LOCKED_BIT), vrm->delta)) diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index d09148ce23054d..b8b1462ca71052 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1676,6 +1676,20 @@ static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma) return vma_flags_is_kernel_owned(&vma->flags); } +static inline bool vma_flags_is_fixed_mapping(const vma_flags_t *flags) +{ + /* + * VMA_PFNMAP_BIT should imply VMA_DONTEXPAND_BIT, but some callers set + * only the former. + */ + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_DONTEXPAND_BIT); +} + +static inline bool vma_is_fixed_mapping(const struct vm_area_struct *vma) +{ + return vma_flags_is_fixed_mapping(&vma->flags); +} + static inline bool vma_flags_can_merge(const vma_flags_t *flags) { /* @@ -1691,7 +1705,7 @@ static inline bool vma_flags_can_merge(const vma_flags_t *flags) if (vma_flags_is_kernel_owned(flags)) return false; /* VMA explicitly marked as being unmergeable. */ - if (vma_flags_test(flags, VMA_DONTEXPAND_BIT)) + if (vma_flags_is_fixed_mapping(flags)) return false; return true; From 366a25665faeab3c30b6af484adb07feff897cde Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:27 +0100 Subject: [PATCH 0662/1012] scsi: sg: convert mmap hook to mmap_prepare and rework Move from the deprecated mmap hook to the new mmap_prepare hook. We are mapping kernel pages here, so use the discontiguous kernel mapping mmap action to do so. Unwind the rather confusing loop and instead map as many pages as we can at one time. Note that we do not need to pay attention to rsv_schp->k_use_sg here, as the pages are populated for the length of the buffer at rsv_schp->page_order granularity as compound pages. The discontiguous kernel page mapping logic handles the compound pages for us. sfp->mmap_called keeps the buffer stable for us. As before it is never cleared, so a failed mmap also leaves it set. We also remove some useless vma, vma->vm_file NULL checks - these will always be non-NULL if you reached the mmap hook logic. We retain log output for consistency, but change what's output on page mapping to indicate that sg_discontig_get() does the work now. Note that we drop the VMA_IO_BIT flag for the VMA here. It was never necessary as we invoke alloc_pages() which gives us refcounted folios that are fine for GUP to access (VMA_IO_BIT would prevent that). Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-18-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- drivers/scsi/sg.c | 115 ++++++++++++++++++++-------------------------- 1 file changed, 51 insertions(+), 64 deletions(-) diff --git a/drivers/scsi/sg.c b/drivers/scsi/sg.c index 5408f002e6c01f..3f9e08725602ca 100644 --- a/drivers/scsi/sg.c +++ b/drivers/scsi/sg.c @@ -1212,85 +1212,72 @@ sg_fasync(int fd, struct file *filp, int mode) return fasync_helper(fd, filp, mode, &sfp->async_qp); } -static vm_fault_t -sg_vma_fault(struct vm_fault *vmf) +static int sg_discontig_init(void *vm_private_data, void **private) { - struct vm_area_struct *vma = vmf->vma; - Sg_fd *sfp; - unsigned long offset, len, sa; - Sg_scatter_hold *rsv_schp; - int k, length; - - if ((NULL == vma) || (!(sfp = (Sg_fd *) vma->vm_private_data))) - return VM_FAULT_SIGBUS; - rsv_schp = &sfp->reserve; - offset = vmf->pgoff << PAGE_SHIFT; - if (offset >= rsv_schp->bufflen) - return VM_FAULT_SIGBUS; - SCSI_LOG_TIMEOUT(3, sg_printk(KERN_INFO, sfp->parentdp, - "sg_vma_fault: offset=%lu, scatg=%d\n", - offset, rsv_schp->k_use_sg)); - sa = vma->vm_start; - length = 1 << (PAGE_SHIFT + rsv_schp->page_order); - for (k = 0; k < rsv_schp->k_use_sg && sa < vma->vm_end; k++) { - len = vma->vm_end - sa; - len = (len < length) ? len : length; - if (offset < len) { - struct page *page = rsv_schp->pages[k] + (offset >> PAGE_SHIFT); - get_page(page); /* increment page count */ - vmf->page = page; - return 0; /* success */ - } - sa += len; - offset -= len; + const unsigned long req_sz = (unsigned long)*private; + Sg_fd *sfp = vm_private_data; + Sg_scatter_hold *rsv_schp = &sfp->reserve; + int err = 0; + + mutex_lock(&sfp->f_mutex); + if (req_sz > rsv_schp->bufflen) { + err = -ENOMEM; /* cannot map more than reserved buffer */ + goto out; + } + sfp->mmap_called = 1; /* Prevents changes to buffer size. */ +out: + mutex_unlock(&sfp->f_mutex); + return err; +} + +static int +sg_discontig_get(struct discontig_kernel_page_state *state) +{ + Sg_fd *sfp = state->vm_private_data; + Sg_scatter_hold *rsv_schp = &sfp->reserve; + const unsigned int order = rsv_schp->page_order; + const pgoff_t nr_pages = state->nr_pages_mapped; + + if (nr_pages >= (rsv_schp->bufflen >> PAGE_SHIFT)) { + discontig_kernel_map_abort(state); + return 0; } - return VM_FAULT_SIGBUS; + SCSI_LOG_TIMEOUT(3, sg_printk(KERN_INFO, sfp->parentdp, + "%s: offset=%lu, scatg=%d\n", __func__, + nr_pages << PAGE_SHIFT, rsv_schp->k_use_sg)); + + discontig_kernel_map_page(state, rsv_schp->pages[nr_pages >> order]); + return 0; } -static const struct vm_operations_struct sg_mmap_vm_ops = { - .fault = sg_vma_fault, +static const struct discontig_kernel_page_ops sg_discontig_ops = { + .init = sg_discontig_init, + .get = sg_discontig_get, }; static int -sg_mmap(struct file *filp, struct vm_area_struct *vma) +sg_mmap_prepare(struct vm_area_desc *desc) { - Sg_fd *sfp; - unsigned long req_sz, len, sa; - Sg_scatter_hold *rsv_schp; - int k, length; - int ret = 0; + Sg_fd *sfp = desc->file->private_data; + const unsigned long req_sz = vma_desc_size(desc); - if ((!filp) || (!vma) || (!(sfp = (Sg_fd *) filp->private_data))) + if (!sfp) return -ENXIO; - req_sz = vma->vm_end - vma->vm_start; + SCSI_LOG_TIMEOUT(3, sg_printk(KERN_INFO, sfp->parentdp, "sg_mmap starting, vm_start=%p, len=%d\n", - (void *) vma->vm_start, (int) req_sz)); - if (vma->vm_pgoff) + (void *) desc->start, (int) req_sz)); + + if (desc->pgoff) return -EINVAL; /* want no offset */ - rsv_schp = &sfp->reserve; - mutex_lock(&sfp->f_mutex); - if (req_sz > rsv_schp->bufflen) { - ret = -ENOMEM; /* cannot map more than reserved buffer */ - goto out; - } - sa = vma->vm_start; - length = 1 << (PAGE_SHIFT + rsv_schp->page_order); - for (k = 0; k < rsv_schp->k_use_sg && sa < vma->vm_end; k++) { - len = vma->vm_end - sa; - len = (len < length) ? len : length; - sa += len; - } + vma_desc_set_flags(desc, VMA_DONTEXPAND_BIT, VMA_DONTDUMP_BIT); + desc->private_data = sfp; - sfp->mmap_called = 1; - vm_flags_set(vma, VM_IO | VM_DONTEXPAND | VM_DONTDUMP); - vma->vm_private_data = sfp; - vma->vm_ops = &sg_mmap_vm_ops; -out: - mutex_unlock(&sfp->f_mutex); - return ret; + mmap_action_map_discontig_kernel_pages(desc, (void *)req_sz, + &sg_discontig_ops); + return 0; } static void @@ -1415,7 +1402,7 @@ static const struct file_operations sg_fops = { .unlocked_ioctl = sg_ioctl, .compat_ioctl = compat_ptr_ioctl, .open = sg_open, - .mmap = sg_mmap, + .mmap_prepare = sg_mmap_prepare, .release = sg_release, .fasync = sg_fasync, }; From 968ee8be6be354918cc752576b2e52e2e20d1c0f Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:28 +0100 Subject: [PATCH 0663/1012] fbdev: defio: assert FBINFO_VIRTFB, drop VM_IO, add VM_MIXEDMAP Currently all drivers which use defio allocate system memory. All of them also set FBINFO_VIRTFB, other than ssd1307fb, however this driver allocates system RAM, so simply failed to set this flag when it ought to. This patch sets FBINFO_VIRTFB on ssd1307fb probe, then drops setting VM_IO in fb_deferred_io_mmap() and instead requires FBINFO_VIRTFB to be set, erroring out with a kernel warning if not. The logic requires a page from the driver and since commit 1ecbc7dd2902 ("fbdev/deferred-io: Always call get_page() for framebuffer pages") has always required it to be refcounted, so this was implicitly already the case. Finally this patch sets VM_MIXEDMAP, as the logic is mapping kernel-allocated memory so this is appropriate. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-19-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Thomas Zimmermann Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- drivers/video/fbdev/core/fb_defio.c | 6 +++--- drivers/video/fbdev/ssd1307fb.c | 2 ++ 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/drivers/video/fbdev/core/fb_defio.c b/drivers/video/fbdev/core/fb_defio.c index fd00b86e1ae602..fb359ecc396619 100644 --- a/drivers/video/fbdev/core/fb_defio.c +++ b/drivers/video/fbdev/core/fb_defio.c @@ -366,13 +366,13 @@ int fb_deferred_io_mmap(struct fb_info *info, struct vm_area_struct *vma) { vma->vm_page_prot = pgprot_decrypted(vma->vm_page_prot); + if (WARN_ON_ONCE(!(info->flags & FBINFO_VIRTFB))) + return -EINVAL; if (!try_module_get(THIS_MODULE)) return -EINVAL; vma->vm_ops = &fb_deferred_io_vm_ops; - vm_flags_set(vma, VM_DONTEXPAND | VM_DONTDUMP); - if (!(info->flags & FBINFO_VIRTFB)) - vm_flags_set(vma, VM_IO); + vm_flags_set(vma, VM_MIXEDMAP | VM_DONTEXPAND | VM_DONTDUMP); vma->vm_private_data = info->fbdefio_state; fb_deferred_io_state_get(info->fbdefio_state); /* released in vma->vm_ops->close() */ diff --git a/drivers/video/fbdev/ssd1307fb.c b/drivers/video/fbdev/ssd1307fb.c index 4d185c75428438..db61d710fccb01 100644 --- a/drivers/video/fbdev/ssd1307fb.c +++ b/drivers/video/fbdev/ssd1307fb.c @@ -767,6 +767,8 @@ static int ssd1307fb_probe(struct i2c_client *client) info->fix.smem_start = __pa(vmem); info->fix.smem_len = vmem_size; + info->flags = FBINFO_VIRTFB; + fb_deferred_io_init(info); i2c_set_clientdata(client, info); From 24b25eb9fee3b4a5506df1f933f085bcb57f9d6f Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:29 +0100 Subject: [PATCH 0664/1012] HSI: cmt_speech: convert mmap hook to mmap_prepare, refactor Use the mmap_prepare in favour of the deprecated mmap hook as part of the work to convert one to another. Since this is simply a refcounted kernel page that has been allocated, it should not be marked VM_IO and should be inserted using the kernel page insertion mechanism, so convert it to do this instead. Use the VMA descriptor's private data field as a scratch buffer to store the page in - this stays valid throughout the kernel page mapping operation. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-20-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- drivers/hsi/clients/cmt_speech.c | 33 +++++++++----------------------- 1 file changed, 9 insertions(+), 24 deletions(-) diff --git a/drivers/hsi/clients/cmt_speech.c b/drivers/hsi/clients/cmt_speech.c index 7226677ebde7ac..801697b74d4f8b 100644 --- a/drivers/hsi/clients/cmt_speech.c +++ b/drivers/hsi/clients/cmt_speech.c @@ -1084,22 +1084,6 @@ static void cs_hsi_stop(struct cs_hsi_iface *hi) kfree(hi); } -static vm_fault_t cs_char_vma_fault(struct vm_fault *vmf) -{ - struct cs_char *csdata = vmf->vma->vm_private_data; - struct page *page; - - page = virt_to_page((void *)csdata->mmap_base); - get_page(page); - vmf->page = page; - - return 0; -} - -static const struct vm_operations_struct cs_char_vm_ops = { - .fault = cs_char_vma_fault, -}; - static int cs_char_fasync(int fd, struct file *file, int on) { struct cs_char *csdata = file->private_data; @@ -1256,18 +1240,19 @@ static long cs_char_ioctl(struct file *file, unsigned int cmd, return r; } -static int cs_char_mmap(struct file *file, struct vm_area_struct *vma) +static int cs_char_mmap_prepare(struct vm_area_desc *desc) { - if (vma->vm_end < vma->vm_start) - return -EINVAL; + struct file *file = desc->file; + struct cs_char *csdata = file->private_data; + struct page **pages = (struct page **)&desc->private_data; - if (vma_pages(vma) != 1) + if (vma_desc_pages(desc) != 1) return -EINVAL; - vm_flags_set(vma, VM_IO | VM_DONTDUMP | VM_DONTEXPAND); - vma->vm_ops = &cs_char_vm_ops; - vma->vm_private_data = file->private_data; + vma_desc_set_flags(desc, VMA_DONTDUMP_BIT, VMA_DONTEXPAND_BIT); + *pages = virt_to_page((void *)csdata->mmap_base); + mmap_action_map_kernel_pages_full(desc, pages); return 0; } @@ -1353,7 +1338,7 @@ static const struct file_operations cs_char_fops = { .write = cs_char_write, .poll = cs_char_poll, .unlocked_ioctl = cs_char_ioctl, - .mmap = cs_char_mmap, + .mmap_prepare = cs_char_mmap_prepare, .open = cs_char_open, .release = cs_char_release, .fasync = cs_char_fasync, From 995db66f5ed8e3f869622b732d2ed91bbce8b786 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:30 +0100 Subject: [PATCH 0665/1012] mm/gup: error out early on !VMA_MAYREAD_BIT VMAs When populating a VMA range via the aptly named populate_vma_page_range() an unreadable VMA will always eventually fail with -EFAULT. That a VMA is accessible is always checked, however VMA_MAYREAD_BIT is not. All user mappings always have VMA_MAYREAD_BIT set, so this check only impacts kernel mappings. It is implemented specifically to disallow population of uprobes XOL mappings which are exec-only. A nasty interaction with these mappings may occur if they are mlocked, so actively disallow this early. This allows a subsequent commit to remove the VM_IO check in __mm_populate() which otherwise requires non-MMIO mappings to be wrongly flagged simply as a workaround. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-21-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/gup.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/mm/gup.c b/mm/gup.c index c2dfcb4744bc3f..e6310a7cc05b2b 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -1836,6 +1836,10 @@ long populate_vma_page_range(struct vm_area_struct *vma, if (!vma_is_accessible(vma)) return -EFAULT; + /* Unreadable VMAs also cannot be faulted in. */ + if (!vma_test(vma, VMA_MAYREAD_BIT)) + return -EFAULT; + gup_flags = FOLL_TOUCH; /* * We want to touch writable mappings with a write fault in order From 366ef864f21868c7adeb5c3b02780645e93fdfb2 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:31 +0100 Subject: [PATCH 0666/1012] uprobes: remove VM_IO, set VM_MIXEDMAP for mapped kernel pages These are not MMIO pages so VMA_IO_BIT is an inappropriate flag to set. Instead, set them VMA_MIXEDMAP_BIT as they are kernel mappings and this is the appropriate flag to set for those. This provides the semantics required - no VMA merging is permitted, but does not prevent GUP. However this has no meaningful impact as these are refcounted and thus can be GUPed. A previous commit already prevented __mm_populate() from being invoked on XOL areas, which VMA_IO_BIT was previously relied upon to do, so that is no longer required. Both VMAs set a VMA name, so always_dump_vma() returns true before vma_dump_size() reaches its VMA_IO_BIT check, and thus there is no change in core dump behaviour. Change this for both the core xol_add_vma() function and the x86-specific get_uprobe_trampoline() function. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-22-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- arch/x86/kernel/uprobes.c | 2 +- kernel/events/uprobes.c | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/arch/x86/kernel/uprobes.c b/arch/x86/kernel/uprobes.c index 65a2de82ecd292..0f60c0d076b620 100644 --- a/arch/x86/kernel/uprobes.c +++ b/arch/x86/kernel/uprobes.c @@ -715,7 +715,7 @@ static struct vm_area_struct *get_uprobe_trampoline(struct mm_struct *mm, unsign *new_mapping = true; return _install_special_mapping(mm, vaddr, PAGE_SIZE, - VM_READ|VM_EXEC|VM_MAYEXEC|VM_MAYREAD|VM_IO, + VM_READ|VM_EXEC|VM_MAYEXEC|VM_MAYREAD|VM_MIXEDMAP, &tramp_mapping); } diff --git a/kernel/events/uprobes.c b/kernel/events/uprobes.c index 7709ea88247785..b89cc5cee00274 100644 --- a/kernel/events/uprobes.c +++ b/kernel/events/uprobes.c @@ -1726,8 +1726,8 @@ static int xol_add_vma(struct mm_struct *mm, struct xol_area *area) } vma = _install_special_mapping(mm, area->vaddr, PAGE_SIZE, - VM_EXEC|VM_MAYEXEC|VM_DONTCOPY|VM_IO| - VM_SEALED_SYSMAP, + VM_EXEC|VM_MAYEXEC|VM_DONTCOPY| + VM_MIXEDMAP|VM_SEALED_SYSMAP, &xol_mapping); if (IS_ERR(vma)) { ret = PTR_ERR(vma); From 36f5f40b144432a5c0280c275f445ff6eb9db1b7 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:32 +0100 Subject: [PATCH 0667/1012] mm/mlock: clear VMA_LOCKED_MASK over mmap callback Currently there's a confusing mess around VMA_LOCKED_BIT and VMA_LOCKONFAULT_BIT. It is permitted for drivers to set any flags they like, with the VMA already possessing lock flags. This results in the absurd situation of a VMA possessing both VMA_SPECIAL_FLAGS and VMA_LOCKED_MASK flags, which is not permitted. This has resulted in mlock_vma_folio() having a very silly check for this scenario to work around it. There is no need for this - just clear the flags before invoking the hook and reinstate them afterwards if they are required. Nothing relies upon this being set during the mmap operation. mmap_prepare is unaffected by this so requires no fix. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-23-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- mm/internal.h | 9 +-------- mm/vma.c | 14 ++++++++++++++ 2 files changed, 15 insertions(+), 8 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index a1970ff52ad7b9..1794b10ffeb4f3 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -971,14 +971,7 @@ void mlock_folio(struct folio *folio); static inline void mlock_vma_folio(struct folio *folio, struct vm_area_struct *vma) { - /* - * The VM_SPECIAL check here serves two purposes. - * 1) VM_IO check prevents migration from double-counting during mlock. - * 2) Although mmap_region() and mlock_fixup() take care that VM_LOCKED - * is never left set on a VM_SPECIAL vma, there is an interval while - * file->f_op->mmap() is using vm_insert_page(s), when VM_LOCKED may - * still be set while VM_SPECIAL bits are added: so ignore it then. - */ + /* The VM_IO check prevents migration from double-counting during mlock. */ if (unlikely((vma->vm_flags & (VM_LOCKED|VM_SPECIAL)) == VM_LOCKED)) mlock_folio(folio); } diff --git a/mm/vma.c b/mm/vma.c index d9db60c418f281..ddaf152d7ee8b4 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2622,6 +2622,11 @@ static int __mmap_new_file_vma(struct mmap_state *map, if (!map->vm_file->f_op->mmap) return 0; + /* + * Driver-specified flags may make the lock flags invalid, so clear + * VMA_LOCKED_MASK and reinstate it afterwards if appropriate. + */ + vma_clear_flags_mask(vma, VMA_LOCKED_MASK); error = mmap_file(vma->vm_file, vma); map->vm_file = vma->vm_file; @@ -2638,6 +2643,15 @@ static int __mmap_new_file_vma(struct mmap_state *map, return error; } + /* If VMA flags still valid for locked mask, reinstate. */ + if (vma_supports_mlock(vma)) { + const vma_flags_t mask = + vma_flags_and_mask(&map->vma_flags, + VMA_LOCKED_MASK); + + vma_set_flags_mask(vma, mask); + } + map->vma_flags = vma->flags; return 0; From afec1cda909bc6c590e0ca1a11859a08b3171ebc Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:33 +0100 Subject: [PATCH 0668/1012] mm/mlock: eliminate weird VMA_IO_BIT abuse and simplify When performing mlock() or munlock() otherwise normal VMAs have VMA_IO_BIT solely to fix a race with migration which might otherwise double-count mlock VMAs. This is unnecessary - at the point of applying folio mlock state, whether setting or clearing PG_mlocked, we know whether or not we are locking. Solve this in two ways - thread a boolean through the page table walk indicating whether a lock or unlock is being performed, and run a locking walk with VMA_LOCKONFAULT_BIT set and VMA_LOCKED_BIT cleared. This state never occurs otherwise, as VMA_LOCKONFAULT_BIT always implies VMA_LOCKED_BIT. These are also always cleared together. Then, update folio_add_lru_vma() and mlock_folio() to check only for VMA_LOCKED_BIT, and update try_to_unmap_one() to check for VMA_LOCKED_MASK instead. Also remove the useless invocation of allow_mlock_munlock() which simply returns true if unlocking and instead rename it to allow_mlock() and only call it when locking. Finally, with the other mlock abuse of VMA_IO_BIT addressed, update mlock_vma_folio() and folio_add_lru_vma() to simply test for VMA_LOCKED_BIT. munlock_vma_folio() tests VMA_LOCKED_MASK instead, as an unmap racing with the locking walk must still munlock folios the walk has already counted. While here, also replace some deprecated VMA flag predicates. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-24-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/folio.c | 2 +- mm/internal.h | 10 +++++++--- mm/mlock.c | 51 +++++++++++++++++++-------------------------------- mm/rmap.c | 4 +++- 4 files changed, 30 insertions(+), 37 deletions(-) diff --git a/mm/folio.c b/mm/folio.c index 47a437e0f7fde6..35e242b48870b0 100644 --- a/mm/folio.c +++ b/mm/folio.c @@ -505,7 +505,7 @@ void folio_add_lru_vma(struct folio *folio, struct vm_area_struct *vma) { VM_BUG_ON_FOLIO(folio_test_lru(folio), folio); - if (unlikely((vma->vm_flags & (VM_LOCKED | VM_SPECIAL)) == VM_LOCKED)) + if (vma_test(vma, VMA_LOCKED_BIT)) mlock_new_folio(folio); else folio_add_lru(folio); diff --git a/mm/internal.h b/mm/internal.h index 1794b10ffeb4f3..1bf6517cf38932 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -971,8 +971,7 @@ void mlock_folio(struct folio *folio); static inline void mlock_vma_folio(struct folio *folio, struct vm_area_struct *vma) { - /* The VM_IO check prevents migration from double-counting during mlock. */ - if (unlikely((vma->vm_flags & (VM_LOCKED|VM_SPECIAL)) == VM_LOCKED)) + if (vma_test(vma, VMA_LOCKED_BIT)) mlock_folio(folio); } @@ -989,7 +988,12 @@ static inline void munlock_vma_folio(struct folio *folio, * always munlock the folio and page reclaim will correct it * if it's wrong. */ - if (unlikely(vma->vm_flags & VM_LOCKED)) + /* + * VMA_LOCKONFAULT_BIT alone marks an mlock walk in progress, see + * mlock_vma_pages_range(). An unmap racing with the walk must still + * munlock folios the walk has already counted. + */ + if (unlikely(vma_test_any_mask(vma, VMA_LOCKED_MASK))) munlock_folio(folio); } diff --git a/mm/mlock.c b/mm/mlock.c index 39215a3eab1fbf..4235a1518fc9e4 100644 --- a/mm/mlock.c +++ b/mm/mlock.c @@ -316,22 +316,10 @@ static inline unsigned int folio_mlock_step(struct folio *folio, return folio_pte_batch(folio, pte, ptent, count); } -static inline bool allow_mlock_munlock(struct folio *folio, +static inline bool allow_mlock(struct folio *folio, struct vm_area_struct *vma, unsigned long start, unsigned long end, unsigned int step) { - /* - * For unlock, allow munlock large folio which is partially - * mapped to VMA. As it's possible that large folio is - * mlocked and VMA is split later. - * - * During memory pressure, such kind of large folio can - * be split. And the pages are not in VM_LOCKed VMA - * can be reclaimed. - */ - if (!vma_test(vma, VMA_LOCKED_BIT)) - return true; - /* folio_within_range() cannot take KSM, but any small folio is OK */ if (!folio_test_large(folio)) return true; @@ -352,6 +340,7 @@ static int mlock_pte_range(pmd_t *pmd, unsigned long addr, { struct vm_area_struct *vma = walk->vma; + const bool lock = walk->private; spinlock_t *ptl; pte_t *start_pte, *pte; pte_t ptent; @@ -368,7 +357,7 @@ static int mlock_pte_range(pmd_t *pmd, unsigned long addr, folio = pmd_folio(*pmd); if (folio_is_zone_device(folio)) goto out; - if (vma_test(vma, VMA_LOCKED_BIT)) + if (lock) mlock_folio(folio); else munlock_folio(folio); @@ -390,10 +379,10 @@ static int mlock_pte_range(pmd_t *pmd, unsigned long addr, continue; step = folio_mlock_step(folio, pte, addr, end); - if (!allow_mlock_munlock(folio, vma, start, end, step)) + if (lock && !allow_mlock(folio, vma, start, end, step)) goto next_entry; - if (vma_test(vma, VMA_LOCKED_BIT)) + if (lock) mlock_folio(folio); else munlock_folio(folio); @@ -428,31 +417,29 @@ static void mlock_vma_pages_range(struct vm_area_struct *vma, .pmd_entry = mlock_pte_range, .walk_lock = PGWALK_WRLOCK_VERIFY, }; + const bool lock = vma_flags_test(new_vma_flags, VMA_LOCKED_BIT); + vma_flags_t walk_flags = *new_vma_flags; /* - * There is a slight chance that concurrent page migration, - * or page reclaim finding a page of this now-VMA_LOCKED_BIT vma, - * will call mlock_vma_folio() and raise page's mlock_count: - * double counting, leaving the page unevictable indefinitely. - * Communicate this danger to mlock_vma_folio() with VMA_IO_BIT, - * which is a VMA_SPECIAL_FLAGS flag not allowed on VMA_LOCKED_BIT vmas. - * mmap_lock is held in write mode here, so this weird - * combination should not be visible to other mmap_lock users; - * but WRITE_ONCE so rmap walkers must see VMA_IO_BIT if VMA_LOCKED_BIT. + * LOCKONFAULT without LOCKED never otherwise occurs: it marks a walk in + * progress so that rmap-side callers, which test VMA_LOCKED_BIT, do not + * count folios, while try_to_unmap_one(), which tests VMA_LOCKED_MASK, + * still refuses to unmap them. */ - if (vma_flags_test(new_vma_flags, VMA_LOCKED_BIT)) - vma_flags_set(new_vma_flags, VMA_IO_BIT); + if (lock) { + vma_flags_clear(&walk_flags, VMA_LOCKED_BIT); + vma_flags_set(&walk_flags, VMA_LOCKONFAULT_BIT); + } + vma_start_write(vma); - vma_flags_reset_once(vma, new_vma_flags); + vma_flags_reset_once(vma, &walk_flags); lru_add_drain(); - walk_page_range_vma(vma, start, end, &mlock_walk_ops, NULL); + walk_page_range_vma(vma, start, end, &mlock_walk_ops, (void *)lock); lru_add_drain(); - if (vma_flags_test(new_vma_flags, VMA_IO_BIT)) { - vma_flags_clear(new_vma_flags, VMA_IO_BIT); + if (lock) vma_flags_reset_once(vma, new_vma_flags); - } } /* diff --git a/mm/rmap.c b/mm/rmap.c index 5332c52909be18..6661bc11ce658b 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -2239,9 +2239,11 @@ static bool try_to_unmap_one(struct folio *folio, struct vm_area_struct *vma, /* * If the folio is in an mlock()d vma, we must not swap it out. + * VMA_LOCKONFAULT_BIT alone marks an mlock walk in progress, see + * mlock_vma_pages_range(). */ if (!(flags & TTU_IGNORE_MLOCK) && - (vma->vm_flags & VM_LOCKED)) { + vma_test_any_mask(vma, VMA_LOCKED_MASK)) { ptes++; /* From 4952d3b80554ac523842209f4d669d2bbade5f46 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:34 +0100 Subject: [PATCH 0669/1012] mm/vma: enforce that only kernel-owned mappings may set VMA_IO_BIT It makes no sense for a mapping whose contents the kernel does not own to specify that the range is MMIO. Prior to this patch, all in-tree drivers which did so have been updated such that they are marked as kernel-owned. The check WARNs and fails the mmap for any out-of-tree driver that still sets VMA_IO_BIT without a kernel mapping. No functional change intended for in-tree code. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-25-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/vma.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/mm/vma.c b/mm/vma.c index ddaf152d7ee8b4..4ca017610fe564 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2802,6 +2802,12 @@ static int mmap_validate_vma_flags(const vma_flags_t *flags) return -EINVAL; #endif + if (!vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT)) { + /* Only kernel-owned mappings may set VMA_IO_BIT. */ + if (WARN_ON_ONCE(vma_flags_test(flags, VMA_IO_BIT))) + return -EINVAL; + } + return 0; } From 9d3c45f4392bcbcc18b0b2f527ddc2c61d0ca53f Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:35 +0100 Subject: [PATCH 0670/1012] mm: remove VMA_IO_BIT check in vma[_flags]_is_kernel_owned() We have now made it such that every driver which sets VMA_IO_BIT marks it as kernel-owned. However, vma_flags_is_kernel_owned() currently checks for VMA_IO_BIT. This was a product of drivers previously marking a range as kernel-owned by setting VMA_IO_BIT alone. Fix this by removing the VMA_IO_BIT check in vma_flags_is_kernel_owned(), and update mmap_validate_vma_flags() to use vma_flags_is_kernel_owned() rather than open-coding the VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT check. This change means that vma[_flags]_can_merge() doesn't check VMA_IO_BIT any longer (which is now redundant) as it calls vma_flags_is_kernel_owned(). Now that the predicate means precisely VMA_PFNMAP_BIT or VMA_MIXEDMAP_BIT, also use it at the other sites which open-code that pair, so the intent is stated rather than the flags, with no functional change: zap_special_vma_range() only zaps kernel-owned mappings, as drivers use it to tear down ranges they established themselves. The mprotect() arch PFN modification check applies to kernel-owned mappings, which may map PFNs without struct pages. NUMA balancing skips VM_MIXEDMAP mappings having already excluded VM_IO and VM_PFNMAP mappings via vma_migratable(), so it skips exactly the kernel-owned mappings - say so. Finally, update the VMA userland merge 'special' flag tests to no longer assert that VMA_IO_BIT prevents merge as VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT now suffices. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-26-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- include/linux/mm.h | 3 +-- kernel/sched/fair.c | 2 +- mm/memory.c | 6 +++--- mm/mprotect.c | 3 +-- mm/vma.c | 2 +- tools/testing/vma/include/dup.h | 3 +-- tools/testing/vma/tests/merge.c | 10 ++-------- 7 files changed, 10 insertions(+), 19 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index a7fa4df6fd4747..e0fe10e0375993 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1631,8 +1631,7 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) */ static inline bool vma_flags_is_kernel_owned(const vma_flags_t *flags) { - return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT, - VMA_IO_BIT); + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT); } /** diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 57360f5cdde4fa..3a8730a627c897 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -4417,7 +4417,7 @@ static void task_numa_work(struct callback_head *work) for (; vma; vma = vma_next(&vmi)) { if (!vma_migratable(vma) || !vma_policy_mof(vma) || - is_vm_hugetlb_page(vma) || (vma->vm_flags & VM_MIXEDMAP)) { + is_vm_hugetlb_page(vma) || vma_is_kernel_owned(vma)) { trace_sched_skip_vma_numa(mm, vma, NUMAB_SKIP_UNSUITABLE); continue; } diff --git a/mm/memory.c b/mm/memory.c index 45b21bb04a18b9..1e6cd2e504089a 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2343,19 +2343,19 @@ void zap_vma_range(struct vm_area_struct *vma, unsigned long address, } /** - * zap_special_vma_range - zap all page table entries in a special vma range + * zap_special_vma_range - zap all page table entries in a kernel-owned VMA * @vma: the vma covering the range to zap * @address: starting address of the range to zap * @size: number of bytes to zap * * This function does nothing when the provided address range is not fully - * contained in @vma, or when the @vma is not VM_PFNMAP or VM_MIXEDMAP. + * contained in @vma, or when @vma is not kernel-owned. */ void zap_special_vma_range(struct vm_area_struct *vma, unsigned long address, unsigned long size) { if (!range_in_vma(vma, address, address + size) || - !(vma->vm_flags & (VM_PFNMAP | VM_MIXEDMAP))) + !vma_is_kernel_owned(vma)) return; zap_vma_range(vma, address, size); diff --git a/mm/mprotect.c b/mm/mprotect.c index 2888ee638d872a..fe32fd87cf5cd7 100644 --- a/mm/mprotect.c +++ b/mm/mprotect.c @@ -783,8 +783,7 @@ mprotect_fixup(struct vma_iterator *vmi, struct mmu_gather *tlb, * uncommon case, so doesn't need to be very optimized. */ if (arch_has_pfn_modify_check() && - vma_flags_test_any(&old_vma_flags, VMA_PFNMAP_BIT, - VMA_MIXEDMAP_BIT) && + vma_flags_is_kernel_owned(&old_vma_flags) && !vma_flags_test_any_mask(&new_vma_flags, VMA_ACCESS_FLAGS)) { pgprot_t new_pgprot = vm_get_page_prot(newflags); diff --git a/mm/vma.c b/mm/vma.c index 4ca017610fe564..48d6d7a50bb628 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -2802,7 +2802,7 @@ static int mmap_validate_vma_flags(const vma_flags_t *flags) return -EINVAL; #endif - if (!vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT)) { + if (!vma_flags_is_kernel_owned(flags)) { /* Only kernel-owned mappings may set VMA_IO_BIT. */ if (WARN_ON_ONCE(vma_flags_test(flags, VMA_IO_BIT))) return -EINVAL; diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index b8b1462ca71052..97d3bf6cd5b712 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1667,8 +1667,7 @@ static inline bool file_is_dev_zero(const struct file *file) static inline bool vma_flags_is_kernel_owned(const vma_flags_t *flags) { - return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT, - VMA_IO_BIT); + return vma_flags_test_any(flags, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT); } static inline bool vma_is_kernel_owned(const struct vm_area_struct *vma) diff --git a/tools/testing/vma/tests/merge.c b/tools/testing/vma/tests/merge.c index acaab282939c0b..b26f1a66a17074 100644 --- a/tools/testing/vma/tests/merge.c +++ b/tools/testing/vma/tests/merge.c @@ -496,17 +496,11 @@ static bool test_vma_merge_special_flags(void) .mm = &mm, .vmi = &vmi, }; - vma_flag_t special_flags[] = { VMA_IO_BIT, VMA_DONTEXPAND_BIT, + vma_flag_t special_flags[] = { VMA_DONTEXPAND_BIT, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT }; - vma_flags_t all_special_flags = EMPTY_VMA_FLAGS; int i; struct vm_area_struct *vma_left, *vma; - /* Make sure there aren't new VM_SPECIAL flags. */ - for (i = 0; i < ARRAY_SIZE(special_flags); i++) - vma_flags_set(&all_special_flags, special_flags[i]); - ASSERT_FLAGS_SAME_MASK(&all_special_flags, VMA_SPECIAL_FLAGS); - /* * 01234 * AAA @@ -520,7 +514,7 @@ static bool test_vma_merge_special_flags(void) * 01234 * AAA* * - * This should merge if not for the VM_SPECIAL flag. + * This should merge if not for the 'special' flag. */ vmg_set_range(&vmg, 0x3000, 0x4000, 3, vma_flags); for (i = 0; i < ARRAY_SIZE(special_flags); i++) { From 9edf328bee646626dd3737504402677a0d9e1f33 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:36 +0100 Subject: [PATCH 0671/1012] mm: remove hugetlb_inline.h This header really makes little sense - every place it is included mm.h is also included, and the header itself includes mm.h, so it does nothing to reduce header size. It also oddly does an #ifdef around checking VMA_HUGETLB_BIT, however VMA_HUGETLB_BIT is unconditionally available, and will never be set if hugetlb is not enabled. Simply remove the header, eliminate the odd ifdeffery and place the predicates in mm.h. The naming of these predicates is odd, but to keep changes separate, we will address this in a separate patch. The file was never put into MAINTAINERS so there's no change required there. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-27-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- drivers/gpu/drm/drm_gpusvm.c | 2 +- include/asm-generic/tlb.h | 2 +- include/linux/hugetlb.h | 1 - include/linux/hugetlb_inline.h | 28 ---------------------------- include/linux/mm.h | 11 +++++++++++ include/linux/pagemap.h | 1 - include/linux/userfaultfd_k.h | 1 - kernel/sched/fair.c | 1 - mm/vma_internal.h | 1 - 9 files changed, 13 insertions(+), 35 deletions(-) delete mode 100644 include/linux/hugetlb_inline.h diff --git a/drivers/gpu/drm/drm_gpusvm.c b/drivers/gpu/drm/drm_gpusvm.c index a93eee7ddb9e95..793dacec210034 100644 --- a/drivers/gpu/drm/drm_gpusvm.c +++ b/drivers/gpu/drm/drm_gpusvm.c @@ -9,9 +9,9 @@ #include #include #include -#include #include #include +#include #include #include diff --git a/include/asm-generic/tlb.h b/include/asm-generic/tlb.h index 9d827076db1969..bb7b05e1009963 100644 --- a/include/asm-generic/tlb.h +++ b/include/asm-generic/tlb.h @@ -11,9 +11,9 @@ #ifndef _ASM_GENERIC__TLB_H #define _ASM_GENERIC__TLB_H +#include #include #include -#include #include #include diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 80a5a03e9cee72..d7e6563cef753c 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -7,7 +7,6 @@ #include #include #include -#include #include #include #include diff --git a/include/linux/hugetlb_inline.h b/include/linux/hugetlb_inline.h deleted file mode 100644 index 5c29cd3223a1e4..00000000000000 --- a/include/linux/hugetlb_inline.h +++ /dev/null @@ -1,28 +0,0 @@ -/* SPDX-License-Identifier: GPL-2.0 */ -#ifndef _LINUX_HUGETLB_INLINE_H -#define _LINUX_HUGETLB_INLINE_H - -#include - -#ifdef CONFIG_HUGETLB_PAGE - -static inline bool is_vma_hugetlb_flags(const vma_flags_t *flags) -{ - return vma_flags_test(flags, VMA_HUGETLB_BIT); -} - -#else - -static inline bool is_vma_hugetlb_flags(const vma_flags_t *flags) -{ - return false; -} - -#endif - -static inline bool is_vm_hugetlb_page(const struct vm_area_struct *vma) -{ - return is_vma_hugetlb_flags(&vma->flags); -} - -#endif diff --git a/include/linux/mm.h b/include/linux/mm.h index e0fe10e0375993..04eee802912faf 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1610,6 +1610,17 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) return is_shared_maywrite(&vma->flags); } +static inline bool is_vma_hugetlb_flags(const vma_flags_t *flags) +{ + return IS_ENABLED(CONFIG_HUGETLB_PAGE) && + vma_flags_test(flags, VMA_HUGETLB_BIT); +} + +static inline bool is_vm_hugetlb_page(const struct vm_area_struct *vma) +{ + return is_vma_hugetlb_flags(&vma->flags); +} + /** * vma_flags_is_kernel_owned() - Do the specified VMA flags indicate that the * contents of the VMA are owned by the kernel rather than the core mm? diff --git a/include/linux/pagemap.h b/include/linux/pagemap.h index bcbb0afe1a6816..73af18a3736706 100644 --- a/include/linux/pagemap.h +++ b/include/linux/pagemap.h @@ -14,7 +14,6 @@ #include #include #include /* for in_interrupt() */ -#include struct folio_batch; diff --git a/include/linux/userfaultfd_k.h b/include/linux/userfaultfd_k.h index a4351cffc60ce3..a14b8a9ffb7b1c 100644 --- a/include/linux/userfaultfd_k.h +++ b/include/linux/userfaultfd_k.h @@ -18,7 +18,6 @@ #include #include #include -#include /* The set of all possible UFFD-related VM flags. */ #define __VM_UFFD_FLAGS (VM_UFFD_MISSING | VM_UFFD_MINOR | \ diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index 3a8730a627c897..e2b00e56d76e16 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -22,7 +22,6 @@ */ #include #include -#include #include #include #include diff --git a/mm/vma_internal.h b/mm/vma_internal.h index 4d300e7bbaf4c2..4f73f0a4db796b 100644 --- a/mm/vma_internal.h +++ b/mm/vma_internal.h @@ -18,7 +18,6 @@ #include #include #include -#include #include #include #include From 6f28fbdc899e36df5f8573ae22b2f7adf0765b50 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:37 +0100 Subject: [PATCH 0672/1012] mm: rename is_vm_hugetlb_page() to vma_is_hugetlb() The is_vm_hugetlb_page() predicate is badly named - the mapping can span more than a page and it is inconsistent with other VMA predicates that typically are prefixed by vma_. Rename to vma_is_hugetlb() for consistency, and while we're here update some VM_BUG_ON_VMA() to VM_WARN_ON_ONCE_VMA() as to avoid unnecessary oopses. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-28-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Marc Zyngier Acked-by: Claudio Imbrenda Acked-by: Anup Patel Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- arch/arm64/kvm/mmu.c | 4 ++-- arch/powerpc/mm/book3s64/radix_tlb.c | 6 +++--- arch/powerpc/mm/nohash/e500_hugetlbpage.c | 2 +- arch/powerpc/mm/nohash/tlb.c | 2 +- arch/riscv/kvm/mmu.c | 2 +- arch/riscv/mm/tlbflush.c | 2 +- arch/s390/mm/gmap_helpers.c | 6 +++--- arch/sparc/mm/init_64.c | 2 +- drivers/gpu/drm/drm_gpusvm.c | 2 +- fs/coredump.c | 2 +- fs/hugetlbfs/inode.c | 2 +- fs/proc/task_mmu.c | 8 +++---- include/asm-generic/tlb.h | 2 +- include/linux/hugetlb.h | 4 ++-- include/linux/mm.h | 19 ++++++++++++++--- include/linux/rmap.h | 2 +- kernel/events/core.c | 2 +- kernel/sched/fair.c | 2 +- mm/gup.c | 4 ++-- mm/huge_memory.c | 2 +- mm/hugetlb.c | 14 ++++++------ mm/internal.h | 2 +- mm/madvise.c | 4 ++-- mm/memory.c | 12 +++++------ mm/mempolicy.c | 2 +- mm/migrate_device.c | 2 +- mm/mmap.c | 2 +- mm/mmu_gather.c | 2 +- mm/mprotect.c | 2 +- mm/mremap.c | 6 +++--- mm/page_vma_mapped.c | 4 ++-- mm/pagewalk.c | 2 +- mm/swapfile.c | 2 +- mm/userfaultfd.c | 26 +++++++++++------------ mm/vma.c | 8 +++---- mm/vmscan.c | 2 +- tools/testing/vma/include/stubs.h | 2 +- 37 files changed, 92 insertions(+), 79 deletions(-) diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c index 2d44cd6a5aed90..b8e28d1b5461d5 100644 --- a/arch/arm64/kvm/mmu.c +++ b/arch/arm64/kvm/mmu.c @@ -1472,13 +1472,13 @@ static int get_vma_page_shift(struct vm_area_struct *vma, unsigned long hva) { unsigned long pa; - if (is_vm_hugetlb_page(vma) && !(vma->vm_flags & VM_PFNMAP)) + if (vma_is_hugetlb(vma) && !(vma->vm_flags & VM_PFNMAP)) return huge_page_shift(hstate_vma(vma)); if (!(vma->vm_flags & VM_PFNMAP)) return PAGE_SHIFT; - VM_BUG_ON(is_vm_hugetlb_page(vma)); + VM_BUG_ON(vma_is_hugetlb(vma)); pa = (vma->vm_pgoff << PAGE_SHIFT) + (hva - vma->vm_start); diff --git a/arch/powerpc/mm/book3s64/radix_tlb.c b/arch/powerpc/mm/book3s64/radix_tlb.c index 7de5760164a90f..b4603a98224b32 100644 --- a/arch/powerpc/mm/book3s64/radix_tlb.c +++ b/arch/powerpc/mm/book3s64/radix_tlb.c @@ -627,7 +627,7 @@ void radix__local_flush_tlb_page(struct vm_area_struct *vma, unsigned long vmadd { #ifdef CONFIG_HUGETLB_PAGE /* need the return fix for nohash.c */ - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) return radix__local_flush_hugetlb_page(vma, vmaddr); #endif radix__local_flush_tlb_page_psize(vma->vm_mm, vmaddr, mmu_virtual_psize); @@ -945,7 +945,7 @@ void radix__flush_tlb_page_psize(struct mm_struct *mm, unsigned long vmaddr, void radix__flush_tlb_page(struct vm_area_struct *vma, unsigned long vmaddr) { #ifdef CONFIG_HUGETLB_PAGE - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) return radix__flush_hugetlb_page(vma, vmaddr); #endif radix__flush_tlb_page_psize(vma->vm_mm, vmaddr, mmu_virtual_psize); @@ -1113,7 +1113,7 @@ void radix__flush_tlb_range(struct vm_area_struct *vma, unsigned long start, { #ifdef CONFIG_HUGETLB_PAGE - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) return radix__flush_hugetlb_tlb_range(vma, start, end); #endif diff --git a/arch/powerpc/mm/nohash/e500_hugetlbpage.c b/arch/powerpc/mm/nohash/e500_hugetlbpage.c index a134d28a0e4d39..b87623f04be53c 100644 --- a/arch/powerpc/mm/nohash/e500_hugetlbpage.c +++ b/arch/powerpc/mm/nohash/e500_hugetlbpage.c @@ -180,7 +180,7 @@ book3e_hugetlb_preload(struct vm_area_struct *vma, unsigned long ea, pte_t pte) */ void __update_mmu_cache(struct vm_area_struct *vma, unsigned long address, pte_t *ptep) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) book3e_hugetlb_preload(vma, address, *ptep); } diff --git a/arch/powerpc/mm/nohash/tlb.c b/arch/powerpc/mm/nohash/tlb.c index 0a650742f3a008..07a2db16c2b153 100644 --- a/arch/powerpc/mm/nohash/tlb.c +++ b/arch/powerpc/mm/nohash/tlb.c @@ -278,7 +278,7 @@ void __flush_tlb_page(struct mm_struct *mm, unsigned long vmaddr, void flush_tlb_page(struct vm_area_struct *vma, unsigned long vmaddr) { #ifdef CONFIG_HUGETLB_PAGE - if (vma && is_vm_hugetlb_page(vma)) + if (vma && vma_is_hugetlb(vma)) flush_hugetlb_page(vma, vmaddr); #endif diff --git a/arch/riscv/kvm/mmu.c b/arch/riscv/kvm/mmu.c index 3e955d808743b4..371fccaf9df08a 100644 --- a/arch/riscv/kvm/mmu.c +++ b/arch/riscv/kvm/mmu.c @@ -665,7 +665,7 @@ int kvm_riscv_mmu_map(struct kvm_vcpu *vcpu, struct kvm_memory_slot *memslot, return -EFAULT; } - is_hugetlb = is_vm_hugetlb_page(vma); + is_hugetlb = vma_is_hugetlb(vma); if (is_hugetlb) vma_pageshift = huge_page_shift(hstate_vma(vma)); else diff --git a/arch/riscv/mm/tlbflush.c b/arch/riscv/mm/tlbflush.c index 962db300a16659..a74a7d5258aa1d 100644 --- a/arch/riscv/mm/tlbflush.c +++ b/arch/riscv/mm/tlbflush.c @@ -149,7 +149,7 @@ void flush_tlb_range(struct vm_area_struct *vma, unsigned long start, { unsigned long stride_size; - if (!is_vm_hugetlb_page(vma)) { + if (!vma_is_hugetlb(vma)) { stride_size = PAGE_SIZE; } else { stride_size = huge_page_size(hstate_vma(vma)); diff --git a/arch/s390/mm/gmap_helpers.c b/arch/s390/mm/gmap_helpers.c index ff63ffb1dbd29c..3f6783b93e679f 100644 --- a/arch/s390/mm/gmap_helpers.c +++ b/arch/s390/mm/gmap_helpers.c @@ -102,7 +102,7 @@ __context_unsafe(/* pte_unmap_unlock() not instrumented */) /* Find the vm address for the guest address */ vma = vma_lookup(mm, vmaddr); - if (!vma || is_vm_hugetlb_page(vma)) + if (!vma || vma_is_hugetlb(vma)) return; /* Get pointer to the page table entry */ @@ -139,7 +139,7 @@ void gmap_helper_discard(struct mm_struct *mm, unsigned long vmaddr, unsigned lo vma = find_vma_intersection(mm, vmaddr, end); if (!vma) return; - if (!is_vm_hugetlb_page(vma)) + if (!vma_is_hugetlb(vma)) zap_vma_range(vma, vmaddr, min(end, vma->vm_end) - vmaddr); vmaddr = vma->vm_end; } @@ -247,7 +247,7 @@ static int __gmap_helper_unshare_zeropages(struct mm_struct *mm) * proof to catch unexpected zeropages in other mappings and * fail. */ - if ((vma->vm_flags & VM_PFNMAP) || is_vm_hugetlb_page(vma)) + if ((vma->vm_flags & VM_PFNMAP) || vma_is_hugetlb(vma)) continue; addr = vma->vm_start; diff --git a/arch/sparc/mm/init_64.c b/arch/sparc/mm/init_64.c index 103db4683b165e..9bbccb5d23a8f1 100644 --- a/arch/sparc/mm/init_64.c +++ b/arch/sparc/mm/init_64.c @@ -413,7 +413,7 @@ void update_mmu_cache_range(struct vm_fault *vmf, struct vm_area_struct *vma, if (mm->context.hugetlb_pte_count || mm->context.thp_pte_count) { unsigned long hugepage_size = PAGE_SIZE; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) hugepage_size = huge_page_size(hstate_vma(vma)); if (hugepage_size >= PUD_SIZE) { diff --git a/drivers/gpu/drm/drm_gpusvm.c b/drivers/gpu/drm/drm_gpusvm.c index 793dacec210034..a1d4989b0b616f 100644 --- a/drivers/gpu/drm/drm_gpusvm.c +++ b/drivers/gpu/drm/drm_gpusvm.c @@ -1142,7 +1142,7 @@ drm_gpusvm_range_find_or_insert(struct drm_gpusvm *gpusvm, * have to change. */ migrate_devmem = ctx->devmem_possible && - vma_is_anonymous(vas) && !is_vm_hugetlb_page(vas); + vma_is_anonymous(vas) && !vma_is_hugetlb(vas); chunk_size = drm_gpusvm_range_chunk_size(gpusvm, notifier, vas, fault_addr, gpuva_start, diff --git a/fs/coredump.c b/fs/coredump.c index 6114839f5178b0..5820cb8ec88e70 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -1608,7 +1608,7 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, } /* Hugetlb memory check */ - if (is_vm_hugetlb_page(vma)) { + if (vma_is_hugetlb(vma)) { if ((vma->vm_flags & VM_SHARED) && FILTER(HUGETLB_SHARED)) goto whole; if (!(vma->vm_flags & VM_SHARED) && FILTER(HUGETLB_PRIVATE)) diff --git a/fs/hugetlbfs/inode.c b/fs/hugetlbfs/inode.c index 7611a8470ea265..ba7097d5720c07 100644 --- a/fs/hugetlbfs/inode.c +++ b/fs/hugetlbfs/inode.c @@ -108,7 +108,7 @@ static int hugetlbfs_file_mmap(struct file *file, struct vm_area_struct *vma) * vma address alignment (but not the pgoff alignment) has * already been checked by prepare_hugepage_range. If you add * any error returns here, do so after setting VM_HUGETLB, so - * is_vm_hugetlb_page tests below unmap_region go the right + * vma_is_hugetlb tests below unmap_region go the right * way when do_mmap unwinds (may be important on powerpc * and ia64). */ diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index e671b4fd8dedd9..565e6446bd3127 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -3015,7 +3015,7 @@ static int pagemap_scan_pte_hole(unsigned long addr, unsigned long end, * hugetlb differs, see pagemap_hugetlb_category(). */ categories = p->cur_vma_category; - if (userfaultfd_wp(vma) && !is_vm_hugetlb_page(vma)) + if (userfaultfd_wp(vma) && !vma_is_hugetlb(vma)) categories |= PAGE_IS_WRITTEN; if (!pagemap_scan_is_interesting_page(categories, p)) @@ -3028,7 +3028,7 @@ static int pagemap_scan_pte_hole(unsigned long addr, unsigned long end, if (~p->arg.flags & PM_SCAN_WP_MATCHING) return ret; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) err = pagemap_scan_hugetlb_hole_wp(vma, addr, end); else err = uffd_wp_range(vma, addr, end - addr, true); @@ -3470,7 +3470,7 @@ static int show_numa_map(struct seq_file *m, void *v) seq_puts(m, " stack"); } - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) seq_puts(m, " huge"); /* Skip walking pages if gate VMA */ @@ -3499,7 +3499,7 @@ static int show_numa_map(struct seq_file *m, void *v) if (md->swapcache) seq_printf(m, " swapcache=%lu", md->swapcache); - if (md->active < md->pages && !is_vm_hugetlb_page(vma)) + if (md->active < md->pages && !vma_is_hugetlb(vma)) seq_printf(m, " active=%lu", md->active); if (md->writeback) diff --git a/include/asm-generic/tlb.h b/include/asm-generic/tlb.h index bb7b05e1009963..fbe1114e9a443d 100644 --- a/include/asm-generic/tlb.h +++ b/include/asm-generic/tlb.h @@ -438,7 +438,7 @@ tlb_update_vma_flags(struct mmu_gather *tlb, struct vm_area_struct *vma) * We rely on tlb_end_vma() to issue a flush, such that when we reset * these values the batch is empty. */ - tlb->vma_huge = is_vm_hugetlb_page(vma); + tlb->vma_huge = vma_is_hugetlb(vma); tlb->vma_exec = !!(vma->vm_flags & VM_EXEC); /* diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index d7e6563cef753c..24727ece20fe52 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -251,14 +251,14 @@ extern void __hugetlb_zap_end(struct vm_area_struct *vma, static inline void hugetlb_zap_begin(struct vm_area_struct *vma, unsigned long *start, unsigned long *end) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) __hugetlb_zap_begin(vma, start, end); } static inline void hugetlb_zap_end(struct vm_area_struct *vma, struct zap_details *details) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) __hugetlb_zap_end(vma, details); } diff --git a/include/linux/mm.h b/include/linux/mm.h index 04eee802912faf..dbd4c1a70a2359 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1610,15 +1610,28 @@ static inline bool vma_is_shared_maywrite(const struct vm_area_struct *vma) return is_shared_maywrite(&vma->flags); } -static inline bool is_vma_hugetlb_flags(const vma_flags_t *flags) +/** + * vma_flags_is_hugetlb() - Do the specified VMA flags indicate that the + * VMA is a hugetlb mapping? + * @flags: The VMA flags to test. + * + * Returns: true if the flags indicate a hugetlb mapping, false otherwise. + */ +static inline bool vma_flags_is_hugetlb(const vma_flags_t *flags) { return IS_ENABLED(CONFIG_HUGETLB_PAGE) && vma_flags_test(flags, VMA_HUGETLB_BIT); } -static inline bool is_vm_hugetlb_page(const struct vm_area_struct *vma) +/** + * vma_is_hugetlb() - Is @vma a hugetlb mapping? + * @vma: The VMA to test. + * + * Returns: true if @vma is a hugetlb mapping, false otherwise. + */ +static inline bool vma_is_hugetlb(const struct vm_area_struct *vma) { - return is_vma_hugetlb_flags(&vma->flags); + return vma_flags_is_hugetlb(&vma->flags); } /** diff --git a/include/linux/rmap.h b/include/linux/rmap.h index 0b332770abeed5..74cca0e3c72641 100644 --- a/include/linux/rmap.h +++ b/include/linux/rmap.h @@ -888,7 +888,7 @@ struct page_vma_mapped_walk { static inline void page_vma_mapped_walk_done(struct page_vma_mapped_walk *pvmw) { /* HugeTLB pte is set to the relevant page table entry without pte_mapped. */ - if (pvmw->pte && !is_vm_hugetlb_page(pvmw->vma)) + if (pvmw->pte && !vma_is_hugetlb(pvmw->vma)) pte_unmap(pvmw->pte); if (pvmw->ptl) spin_unlock(pvmw->ptl); diff --git a/kernel/events/core.c b/kernel/events/core.c index 634d2ccbab82d8..601e8d944c240c 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -9818,7 +9818,7 @@ static void perf_event_mmap_event(struct perf_mmap_event *mmap_event) if (vma->vm_flags & VM_LOCKED) flags |= MAP_LOCKED; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) flags |= MAP_HUGETLB; if (file) { diff --git a/kernel/sched/fair.c b/kernel/sched/fair.c index e2b00e56d76e16..8d38c3b7d7920b 100644 --- a/kernel/sched/fair.c +++ b/kernel/sched/fair.c @@ -4416,7 +4416,7 @@ static void task_numa_work(struct callback_head *work) for (; vma; vma = vma_next(&vmi)) { if (!vma_migratable(vma) || !vma_policy_mof(vma) || - is_vm_hugetlb_page(vma) || vma_is_kernel_owned(vma)) { + vma_is_hugetlb(vma) || vma_is_kernel_owned(vma)) { trace_sched_skip_vma_numa(mm, vma, NUMAB_SKIP_UNSUITABLE); continue; } diff --git a/mm/gup.c b/mm/gup.c index e6310a7cc05b2b..f166acf794e308 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -621,7 +621,7 @@ static struct page *no_page_table(struct vm_area_struct *vma, * But we can only make this optimization where a hole would surely * be zero-filled if handle_mm_fault() actually did handle it. */ - if (is_vm_hugetlb_page(vma)) { + if (vma_is_hugetlb(vma)) { struct hstate *h = hstate_vma(vma); if (!hugetlbfs_pagecache_present(h, vma, address)) @@ -1213,7 +1213,7 @@ static int check_vma_flags(struct vm_area_struct *vma, unsigned long gup_flags) if ((gup_flags & FOLL_LONGTERM) && vma_is_fsdax(vma)) return -EOPNOTSUPP; - if ((gup_flags & FOLL_SPLIT_PMD) && is_vm_hugetlb_page(vma)) + if ((gup_flags & FOLL_SPLIT_PMD) && vma_is_hugetlb(vma)) return -EOPNOTSUPP; if (vma_is_secretmem(vma)) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 6ee21854b6dfa9..05c17a01551df1 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -4792,7 +4792,7 @@ static inline bool vma_not_suitable_for_thp_split(struct vm_area_struct *vma) return true; if (vma_test(vma, VMA_IO_BIT)) return true; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) return true; return false; diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 7b27c3c5c3e58e..1b53ba991d360a 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -1147,7 +1147,7 @@ static inline struct resv_map *inode_resv_map(struct inode *inode) static struct resv_map *vma_resv_map(struct vm_area_struct *vma) { - VM_BUG_ON_VMA(!is_vm_hugetlb_page(vma), vma); + VM_WARN_ON_ONCE_VMA(!vma_is_hugetlb(vma), vma); if (vma->vm_flags & VM_MAYSHARE) { struct address_space *mapping = vma->vm_file->f_mapping; struct inode *inode = mapping->host; @@ -1162,7 +1162,7 @@ static struct resv_map *vma_resv_map(struct vm_area_struct *vma) static void set_vma_resv_map(struct vm_area_struct *vma, struct resv_map *map) { - VM_WARN_ON_ONCE_VMA(!is_vm_hugetlb_page(vma), vma); + VM_WARN_ON_ONCE_VMA(!vma_is_hugetlb(vma), vma); VM_WARN_ON_ONCE_VMA(vma_test(vma, VMA_MAYSHARE_BIT), vma); set_vma_private_data(vma, (unsigned long)map); @@ -1170,7 +1170,7 @@ static void set_vma_resv_map(struct vm_area_struct *vma, struct resv_map *map) static void set_vma_resv_flags(struct vm_area_struct *vma, unsigned long flags) { - VM_WARN_ON_ONCE_VMA(!is_vm_hugetlb_page(vma), vma); + VM_WARN_ON_ONCE_VMA(!vma_is_hugetlb(vma), vma); VM_WARN_ON_ONCE_VMA(vma_test(vma, VMA_MAYSHARE_BIT), vma); set_vma_private_data(vma, get_vma_private_data(vma) | flags); @@ -1178,7 +1178,7 @@ static void set_vma_resv_flags(struct vm_area_struct *vma, unsigned long flags) static int is_vma_resv_set(struct vm_area_struct *vma, unsigned long flag) { - VM_BUG_ON_VMA(!is_vm_hugetlb_page(vma), vma); + VM_WARN_ON_ONCE_VMA(!vma_is_hugetlb(vma), vma); return (get_vma_private_data(vma) & flag) != 0; } @@ -1192,7 +1192,7 @@ bool __vma_private_lock(struct vm_area_struct *vma) void hugetlb_dup_vma_private(struct vm_area_struct *vma) { - VM_BUG_ON_VMA(!is_vm_hugetlb_page(vma), vma); + VM_WARN_ON_ONCE_VMA(!vma_is_hugetlb(vma), vma); /* * Clear vm_private_data * - For shared mappings this is a per-vma semaphore that may be @@ -5276,7 +5276,7 @@ void __unmap_hugepage_range(struct mmu_gather *tlb, struct vm_area_struct *vma, unsigned long last_addr_mask; i_mmap_assert_write_locked(vma->vm_file->f_mapping); - WARN_ON(!is_vm_hugetlb_page(vma)); + WARN_ON(!vma_is_hugetlb(vma)); BUG_ON(start & ~huge_page_mask(h)); BUG_ON(end & ~huge_page_mask(h)); @@ -7502,6 +7502,6 @@ void hugetlb_unshare_all_pmds(struct vm_area_struct *vma) */ void fixup_hugetlb_reservations(struct vm_area_struct *vma) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) clear_vma_resv_huge_pages(vma); } diff --git a/mm/internal.h b/mm/internal.h index 1bf6517cf38932..c63df7b7d77272 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -1117,7 +1117,7 @@ static inline bool vma_supports_mlock(const struct vm_area_struct *vma) return false; if (vma_test_single_mask(vma, VMA_DROPPABLE)) return false; - if (vma_is_dax(vma) || is_vm_hugetlb_page(vma)) + if (vma_is_dax(vma) || vma_is_hugetlb(vma)) return false; return vma != get_gate_vma(current->mm); } diff --git a/mm/madvise.c b/mm/madvise.c index 1cfb0433229310..f7d03e8988e214 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -881,7 +881,7 @@ bool madvise_dontneed_free_valid_vma(struct madvise_behavior *madv_behavior) int behavior = madv_behavior->behavior; struct madvise_behavior_range *range = &madv_behavior->range; - if (!is_vm_hugetlb_page(vma)) { + if (!vma_is_hugetlb(vma)) { unsigned int forbidden = VM_PFNMAP; if (behavior != MADV_DONTNEED_LOCKED) @@ -1579,7 +1579,7 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior) new_flags |= VM_DONTDUMP; break; case MADV_DODUMP: - if ((!is_vm_hugetlb_page(vma) && (new_flags & VM_SPECIAL)) || + if ((!vma_is_hugetlb(vma) && (new_flags & VM_SPECIAL)) || (new_flags & VM_DROPPABLE)) return -EINVAL; new_flags &= ~VM_DONTDUMP; diff --git a/mm/memory.c b/mm/memory.c index 1e6cd2e504089a..6c011979401aab 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -1564,7 +1564,7 @@ copy_page_range(struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma) if (!vma_needs_copy(dst_vma, src_vma)) return 0; - if (is_vm_hugetlb_page(src_vma)) + if (vma_is_hugetlb(src_vma)) return copy_hugetlb_page_range(dst_mm, src_mm, dst_vma, src_vma); /* @@ -2178,7 +2178,7 @@ static void __zap_vma_range(struct mmu_gather *tlb, struct vm_area_struct *vma, if (vma->vm_file && !reaping) uprobe_munmap(vma, start, end); - if (unlikely(is_vm_hugetlb_page(vma))) { + if (unlikely(vma_is_hugetlb(vma))) { zap_flags_t zap_flags = details ? details->zap_flags : 0; VM_WARN_ON_ONCE(reaping); @@ -2313,7 +2313,7 @@ void zap_vma_range_batched(struct mmu_gather *tlb, */ __zap_vma_range(tlb, vma, address, end, details); mmu_notifier_invalidate_range_end(&range); - if (is_vm_hugetlb_page(vma)) { + if (vma_is_hugetlb(vma)) { /* * flush tlb and free resources before hugetlb_zap_end(), to * avoid concurrent page faults' allocation failure. @@ -6933,7 +6933,7 @@ vm_fault_t handle_mm_fault(struct vm_area_struct *vma, unsigned long address, lru_gen_enter_fault(vma); - if (unlikely(is_vm_hugetlb_page(vma))) + if (unlikely(vma_is_hugetlb(vma))) ret = hugetlb_fault(vma->vm_mm, vma, address, flags); else ret = __handle_mm_fault(vma, address, flags); @@ -7803,12 +7803,12 @@ void ptlock_free(struct ptdesc *ptdesc) void vma_pgtable_walk_begin(struct vm_area_struct *vma) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) hugetlb_vma_lock_read(vma); } void vma_pgtable_walk_end(struct vm_area_struct *vma) { - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) hugetlb_vma_unlock_read(vma); } diff --git a/mm/mempolicy.c b/mm/mempolicy.c index e9860fb9f73f8d..70298fded1b4a8 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -2018,7 +2018,7 @@ bool vma_migratable(struct vm_area_struct *vma) if (vma_is_dax(vma)) return false; - if (is_vm_hugetlb_page(vma) && + if (vma_is_hugetlb(vma) && !hugepage_migration_supported(hstate_vma(vma))) return false; diff --git a/mm/migrate_device.c b/mm/migrate_device.c index 0c437004329d9c..c38cbaaef5a492 100644 --- a/mm/migrate_device.c +++ b/mm/migrate_device.c @@ -743,7 +743,7 @@ int migrate_vma_setup(struct migrate_vma *args) args->start &= PAGE_MASK; args->end &= PAGE_MASK; - if (!args->vma || is_vm_hugetlb_page(args->vma) || + if (!args->vma || vma_is_hugetlb(args->vma) || (args->vma->vm_flags & VM_SPECIAL) || vma_is_dax(args->vma)) return -EINVAL; if (nr_pages <= 0) diff --git a/mm/mmap.c b/mm/mmap.c index 4bf26b0f1e6e36..98449f364af1c4 100644 --- a/mm/mmap.c +++ b/mm/mmap.c @@ -1786,7 +1786,7 @@ __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm) /* * Copy/update hugetlb private vma information. */ - if (is_vm_hugetlb_page(tmp)) + if (vma_is_hugetlb(tmp)) hugetlb_dup_vma_private(tmp); /* diff --git a/mm/mmu_gather.c b/mm/mmu_gather.c index 2a72a9686773a3..9f353f0e2ef4d2 100644 --- a/mm/mmu_gather.c +++ b/mm/mmu_gather.c @@ -480,7 +480,7 @@ void tlb_gather_mmu_vma(struct mmu_gather *tlb, struct vm_area_struct *vma) { tlb_gather_mmu(tlb, vma->vm_mm); tlb_update_vma_flags(tlb, vma); - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) /* All entries have the same size. */ tlb_change_page_size(tlb, huge_page_size(hstate_vma(vma))); } diff --git a/mm/mprotect.c b/mm/mprotect.c index fe32fd87cf5cd7..a1b6d29bf03908 100644 --- a/mm/mprotect.c +++ b/mm/mprotect.c @@ -717,7 +717,7 @@ long change_protection(struct mmu_gather *tlb, (cp_flags & MM_CP_UFFD_RWP)) newprot = PAGE_NONE; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) pages = hugetlb_change_protection(vma, start, end, newprot, cp_flags); else diff --git a/mm/mremap.c b/mm/mremap.c index c5ee24784dfa67..49dc25d8a34dc2 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -812,7 +812,7 @@ unsigned long move_page_tables(struct pagetable_move_control *pmc) if (!pmc->len_in) return 0; - if (is_vm_hugetlb_page(pmc->old)) + if (vma_is_hugetlb(pmc->old)) return move_hugetlb_page_tables(pmc->old, pmc->new, pmc->old_addr, pmc->new_addr, pmc->len_in); @@ -1758,7 +1758,7 @@ static bool vma_multi_allowed(struct vm_area_struct *vma) /* Known good. */ if (vma_is_shmem(vma)) return true; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) return true; if (file->f_op->get_unmapped_area == thp_get_unmapped_area) return true; @@ -1781,7 +1781,7 @@ static int check_prep_vma(struct vma_remap_struct *vrm) return -EPERM; /* Align to hugetlb page size, if required. */ - if (is_vm_hugetlb_page(vma) && !align_hugetlb(vrm)) + if (vma_is_hugetlb(vma) && !align_hugetlb(vrm)) return -EINVAL; vrm_set_delta(vrm); diff --git a/mm/page_vma_mapped.c b/mm/page_vma_mapped.c index 28e306fdb3a5b8..8408aee7571b56 100644 --- a/mm/page_vma_mapped.c +++ b/mm/page_vma_mapped.c @@ -109,7 +109,7 @@ static bool check_pte(struct page_vma_mapped_walk *pvmw, unsigned long pte_nr) unsigned long pfn; pte_t ptent; - if (is_vm_hugetlb_page(pvmw->vma)) + if (vma_is_hugetlb(pvmw->vma)) ptent = huge_ptep_get(pvmw->vma->vm_mm, pvmw->address, pvmw->pte); else @@ -206,7 +206,7 @@ bool page_vma_mapped_walk(struct page_vma_mapped_walk *pvmw) if (pvmw->pmd && !pvmw->pte) return not_found(pvmw); - if (unlikely(is_vm_hugetlb_page(vma))) { + if (unlikely(vma_is_hugetlb(vma))) { struct hstate *hstate = hstate_vma(vma); unsigned long size = huge_page_size(hstate); /* The only possible mapping was handled on last iteration */ diff --git a/mm/pagewalk.c b/mm/pagewalk.c index 7411702a37f58d..e6493bbe6919e5 100644 --- a/mm/pagewalk.c +++ b/mm/pagewalk.c @@ -408,7 +408,7 @@ static int __walk_page_range(unsigned long start, unsigned long end, int err = 0; struct vm_area_struct *vma = walk->vma; const struct mm_walk_ops *ops = walk->ops; - bool is_hugetlb = is_vm_hugetlb_page(vma); + bool is_hugetlb = vma_is_hugetlb(vma); /* We do not support hugetlb PTE installation. */ if (ops->install_pte && is_hugetlb) diff --git a/mm/swapfile.c b/mm/swapfile.c index 2cd0d0ba966c38..c1c5fbb3c909d3 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -2707,7 +2707,7 @@ static int unuse_mm(struct mm_struct *mm, unsigned int type) if (check_stable_address_space(mm)) goto unlock; for_each_vma(vmi, vma) { - if (vma->anon_vma && !is_vm_hugetlb_page(vma)) { + if (vma->anon_vma && !vma_is_hugetlb(vma)) { ret = unuse_vma(vma, type); if (ret) break; diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index bf50bff3838aa4..215b993d6df4fd 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -237,7 +237,7 @@ static int mfill_get_vma(struct mfill_state *state) if ((flags & MFILL_ATOMIC_WP) && !(dst_vma->vm_flags & VM_UFFD_WP)) goto out_unlock; - if (is_vm_hugetlb_page(dst_vma)) + if (vma_is_hugetlb(dst_vma)) return 0; ops = vma_uffd_ops(dst_vma); @@ -804,7 +804,7 @@ static __always_inline ssize_t mfill_atomic_hugetlb( } err = -ENOENT; - if (!is_vm_hugetlb_page(dst_vma)) + if (!vma_is_hugetlb(dst_vma)) goto out_unlock_vma; err = -EINVAL; @@ -967,7 +967,7 @@ static __always_inline ssize_t mfill_atomic(struct userfaultfd_ctx *ctx, /* * If this is a HUGETLB vma, pass off to appropriate routine */ - if (is_vm_hugetlb_page(state.vma)) + if (vma_is_hugetlb(state.vma)) return mfill_atomic_hugetlb(ctx, state.vma, dst_start, src_start, len, flags); @@ -1114,7 +1114,7 @@ static int mwriteprotect_range(struct userfaultfd_ctx *ctx, unsigned long start, break; } - if (is_vm_hugetlb_page(dst_vma)) { + if (vma_is_hugetlb(dst_vma)) { err = -EINVAL; page_mask = vma_kernel_pagesize(dst_vma) - 1; if ((start & page_mask) || (len & page_mask)) @@ -1172,7 +1172,7 @@ int mrwprotect_range(struct userfaultfd_ctx *ctx, unsigned long start, if (!userfaultfd_rwp(dst_vma)) return -ENOENT; - if (is_vm_hugetlb_page(dst_vma)) { + if (vma_is_hugetlb(dst_vma)) { unsigned long page_mask; page_mask = vma_kernel_pagesize(dst_vma) - 1; @@ -2150,7 +2150,7 @@ static bool vma_can_userfault(struct vm_area_struct *vma, vm_flags_t vm_flags, if (vma->vm_flags & (VM_DROPPABLE | VM_SHADOW_STACK)) return false; - if (!is_vm_hugetlb_page(vma) && (vma->vm_flags & VM_SPECIAL)) + if (!vma_is_hugetlb(vma) && (vma->vm_flags & VM_SPECIAL)) return false; vm_flags &= __VM_UFFD_FLAGS; @@ -2320,7 +2320,7 @@ static int userfaultfd_register_range(struct userfaultfd_ctx *ctx, */ userfaultfd_set_ctx(vma, ctx, vm_flags); - if (is_vm_hugetlb_page(vma) && uffd_disable_huge_pmd_share(vma)) + if (vma_is_hugetlb(vma) && uffd_disable_huge_pmd_share(vma)) hugetlb_unshare_all_pmds(vma); skip: @@ -2896,7 +2896,7 @@ vm_fault_t handle_userfault(struct vm_fault *vmf, unsigned long reason) * (sleepable) vma lock can modify the current task state, that * must be before explicitly calling set_current_state(). */ - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) hugetlb_vma_lock_read(vma); spin_lock_irq(&ctx->fault_pending_wqh.lock); @@ -2913,7 +2913,7 @@ vm_fault_t handle_userfault(struct vm_fault *vmf, unsigned long reason) set_current_state(blocking_state); spin_unlock_irq(&ctx->fault_pending_wqh.lock); - if (is_vm_hugetlb_page(vma)) { + if (vma_is_hugetlb(vma)) { must_wait = userfaultfd_huge_must_wait(ctx, vmf, reason); hugetlb_vma_unlock_read(vma); } else { @@ -3745,7 +3745,7 @@ static int userfaultfd_register(struct userfaultfd_ctx *ctx, * If the first vma contains huge pages, make sure start address * is aligned to huge page size. */ - if (is_vm_hugetlb_page(vma)) { + if (vma_is_hugetlb(vma)) { unsigned long vma_hpagesize = vma_kernel_pagesize(vma); if (start & (vma_hpagesize - 1)) @@ -3796,7 +3796,7 @@ static int userfaultfd_register(struct userfaultfd_ctx *ctx, * If this vma contains ending address, and huge pages * check alignment. */ - if (is_vm_hugetlb_page(cur) && end <= cur->vm_end && + if (vma_is_hugetlb(cur) && end <= cur->vm_end && end > cur->vm_start) { unsigned long vma_hpagesize = vma_kernel_pagesize(cur); @@ -3832,7 +3832,7 @@ static int userfaultfd_register(struct userfaultfd_ctx *ctx, /* * Note vmas containing huge pages */ - if (is_vm_hugetlb_page(cur)) + if (vma_is_hugetlb(cur)) basic_ioctls = true; found = true; @@ -3918,7 +3918,7 @@ static int userfaultfd_unregister(struct userfaultfd_ctx *ctx, * If the first vma contains huge pages, make sure start address * is aligned to huge page size. */ - if (is_vm_hugetlb_page(vma)) { + if (vma_is_hugetlb(vma)) { unsigned long vma_hpagesize = vma_kernel_pagesize(vma); if (start & (vma_hpagesize - 1)) diff --git a/mm/vma.c b/mm/vma.c index 48d6d7a50bb628..0862d0861d0128 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -599,7 +599,7 @@ __split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, * boundary. */ vma_adjust_trans_huge(vma, vma->vm_start, addr, NULL); - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) hugetlb_split(vma, addr); if (new_below) { @@ -2251,7 +2251,7 @@ bool vma_wants_writenotify(struct vm_area_struct *vma, pgprot_t vm_page_prot) * Do we need to track softdirty? hugetlb does not support softdirty * tracking yet. */ - if (vma_soft_dirty_enabled(vma) && !is_vm_hugetlb_page(vma)) + if (vma_soft_dirty_enabled(vma) && !vma_is_hugetlb(vma)) return true; /* Do we need write faults for uffd-wp tracking? */ @@ -2370,7 +2370,7 @@ int mm_take_all_locks(struct mm_struct *mm) if (signal_pending(current)) goto out_unlock; if (vma->vm_file && vma->vm_file->f_mapping && - is_vm_hugetlb_page(vma)) + vma_is_hugetlb(vma)) vm_lock_mapping(mm, vma->vm_file->f_mapping); } @@ -2379,7 +2379,7 @@ int mm_take_all_locks(struct mm_struct *mm) if (signal_pending(current)) goto out_unlock; if (vma->vm_file && vma->vm_file->f_mapping && - !is_vm_hugetlb_page(vma)) + !vma_is_hugetlb(vma)) vm_lock_mapping(mm, vma->vm_file->f_mapping); } diff --git a/mm/vmscan.c b/mm/vmscan.c index dd6261c862794f..9fc4282da0aab5 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3474,7 +3474,7 @@ static int should_skip_vma(unsigned long start, unsigned long end, struct mm_wal if (!vma_is_accessible(vma)) return true; - if (is_vm_hugetlb_page(vma)) + if (vma_is_hugetlb(vma)) return true; if (!vma_has_recency(vma)) diff --git a/tools/testing/vma/include/stubs.h b/tools/testing/vma/include/stubs.h index d6136e19a8af3d..48d1dc53df42cb 100644 --- a/tools/testing/vma/include/stubs.h +++ b/tools/testing/vma/include/stubs.h @@ -193,7 +193,7 @@ static inline bool mapping_can_writeback(struct address_space *mapping) return true; } -static inline bool is_vm_hugetlb_page(struct vm_area_struct *vma) +static inline bool vma_is_hugetlb(struct vm_area_struct *vma) { return false; } From 6942fbac058875dcb8af0a15cede87f15041763a Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:38 +0100 Subject: [PATCH 0673/1012] mm: drop some redundant checks around hugetlb VMAs Adjust code which inadvertently perform redundant checks on hugetlb VMAs and clean them up: * hugetlb VMAs have VMA_DONTEXPAND_BIT set so a VMA_SPECIAL_FLAGS check suffices. (migrate_vma_setup() regains an explicit hugetlb test later in the series, once VMA_SPECIAL_FLAGS is removed.) * hugetlb VMAs unconditionally set vma->vm_ops, so they are never anonymous. * hugetlb VMAs do not set VMA_PFNMAP_BIT so checking for this is redundant. While we're here also drop a VM_BUG_ON() which the simplified check above makes unreachable, and use the new VMA flag API. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-29-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Marc Zyngier Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- arch/arm64/kvm/mmu.c | 4 +--- drivers/gpu/drm/drm_gpusvm.c | 3 +-- mm/migrate_device.c | 4 ++-- 3 files changed, 4 insertions(+), 7 deletions(-) diff --git a/arch/arm64/kvm/mmu.c b/arch/arm64/kvm/mmu.c index b8e28d1b5461d5..73c8492e85c4e0 100644 --- a/arch/arm64/kvm/mmu.c +++ b/arch/arm64/kvm/mmu.c @@ -1472,14 +1472,12 @@ static int get_vma_page_shift(struct vm_area_struct *vma, unsigned long hva) { unsigned long pa; - if (vma_is_hugetlb(vma) && !(vma->vm_flags & VM_PFNMAP)) + if (vma_is_hugetlb(vma)) return huge_page_shift(hstate_vma(vma)); if (!(vma->vm_flags & VM_PFNMAP)) return PAGE_SHIFT; - VM_BUG_ON(vma_is_hugetlb(vma)); - pa = (vma->vm_pgoff << PAGE_SHIFT) + (hva - vma->vm_start); #ifndef __PAGETABLE_PMD_FOLDED diff --git a/drivers/gpu/drm/drm_gpusvm.c b/drivers/gpu/drm/drm_gpusvm.c index a1d4989b0b616f..fab34fea99c2fe 100644 --- a/drivers/gpu/drm/drm_gpusvm.c +++ b/drivers/gpu/drm/drm_gpusvm.c @@ -1141,8 +1141,7 @@ drm_gpusvm_range_find_or_insert(struct drm_gpusvm *gpusvm, * limitations. If/when migrate_vma_* add more support, this logic will * have to change. */ - migrate_devmem = ctx->devmem_possible && - vma_is_anonymous(vas) && !vma_is_hugetlb(vas); + migrate_devmem = ctx->devmem_possible && vma_is_anonymous(vas); chunk_size = drm_gpusvm_range_chunk_size(gpusvm, notifier, vas, fault_addr, gpuva_start, diff --git a/mm/migrate_device.c b/mm/migrate_device.c index c38cbaaef5a492..b9c453c28795f3 100644 --- a/mm/migrate_device.c +++ b/mm/migrate_device.c @@ -743,8 +743,8 @@ int migrate_vma_setup(struct migrate_vma *args) args->start &= PAGE_MASK; args->end &= PAGE_MASK; - if (!args->vma || vma_is_hugetlb(args->vma) || - (args->vma->vm_flags & VM_SPECIAL) || vma_is_dax(args->vma)) + if (!args->vma || vma_test_any_mask(args->vma, VMA_SPECIAL_FLAGS) || + vma_is_dax(args->vma)) return -EINVAL; if (nr_pages <= 0) return -EINVAL; From 73b89e1141ed69dcb502705b810faaf2cd847019 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:39 +0100 Subject: [PATCH 0674/1012] mm/madvise: update is_valid_guard_vma() to use vma_can_merge() We currently disallow the installation of lightweight guard regions in VMAs whose flags intersect VMA_SPECIAL_FLAGS or VMA_HUGETLB_BIT, or VMA_LOCKED_BIT unless allow_locked is set. hugetlb VMAs set VMA_DONTEXPAND_BIT so this was already redundant, VMA_SPECIAL_FLAGS already sufficed. However, now that VMA_IO_BIT is only set if VMA_PFNMAP or VMA_MIXEDMAP_BIT is set, this check collapses to being the equivalent of !vma_can_merge(). Update is_valid_guard_vma() to reflect this. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-30-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/madvise.c | 20 +++++++++++++------- 1 file changed, 13 insertions(+), 7 deletions(-) diff --git a/mm/madvise.c b/mm/madvise.c index f7d03e8988e214..7a038837b2a8b7 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -1222,19 +1222,25 @@ static long madvise_remove(struct madvise_behavior *madv_behavior) return error; } -static bool is_valid_guard_vma(struct vm_area_struct *vma, bool allow_locked) +static bool is_valid_guard_vma(const struct vm_area_struct *vma, + bool allow_locked) { - vm_flags_t disallowed = VM_SPECIAL | VM_HUGETLB; - /* - * A user could lock after setting a guard range but that's fine, as + * A user could lock after setting a guard range but that's fine as * they'd not be able to fault in. The issue arises when we try to zap * existing locked VMAs. We don't want to do that. */ - if (!allow_locked) - disallowed |= VM_LOCKED; + if (!allow_locked && vma_test(vma, VMA_LOCKED_BIT)) + return false; + /* + * Guard regions require a VMA whose page tables are managed solely by + * the core, which is also what merging requires, so disallow any flags + * that would prevent a merge. + */ + if (!vma_can_merge(vma)) + return false; - return !(vma->vm_flags & disallowed); + return true; } static bool is_guard_pte_marker(pte_t ptent) From e58dd8de8f75e672b86687a64cbfcad0f59aaeb2 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:40 +0100 Subject: [PATCH 0675/1012] mm/vma: introduce vma[_flags]_is_persistent() Introduce vma[_flags]_is_persistent() for the purposes of identifying mappings that are persistent in the sense that bytes to the mapping stay there, and bytes read from the mapping are the same unless changed by actions taken by userland. Kernel-owned mappings do not fall into this category, as their owner may change the contents without the user having initiated it, and nor of course does memory-mapped I/O. We exclude fixed mappings as these are singled out as being unmergeable and so cannot be guaranteed to persist user data. hugetlb mappings are fixed mappings, but their contents are entirely the user's, so they are explicitly carved out as persistent, as the MADV_DODUMP check already does. It excludes droppable mappings, which by their nature are ephemeral. Use this functionality to update the madvise MADV_DODUMP check to test for persistence rather than open-coding this. This replaces the VM_SPECIAL check which means it no longer checks for VMA_IO_BIT, however this is safe as we have established the invariant that only kernel-owned mappings may set VMA_IO_BIT, so we implicitly include these. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-31-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 44 ++++++++++++++++++++++++++++++++++++++++++++ mm/madvise.c | 4 ++-- 2 files changed, 46 insertions(+), 2 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index dbd4c1a70a2359..eaf3a4110f91ed 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1742,6 +1742,50 @@ static inline bool vma_can_merge(const struct vm_area_struct *vma) return vma_flags_can_merge(&vma->flags); } +/** + * vma_flags_is_persistent() - Do the specified VMA flags imply that the VMA + * contains persistent data? + * @flags: The VMA flags to test. + * + * Persistent in the sense that - if you write bytes to the mapping - do they + * stay written? + * + * If the kernel or a device could write to the memory independently of + * userland, or the kernel could arbitrarily discard it, then it is not + * persistent. + * + * Returns: true if the flags imply this VMA is persistent, otherwise false. + */ +static inline bool vma_flags_is_persistent(const vma_flags_t *flags) +{ + /* hugetlb is a fixed mapping, but its contents are the user's own. */ + if (vma_flags_is_hugetlb(flags)) + return true; + /* + * MMIO mappings may not store what is written and may be changed by the + * device. Kernel-owned and fixed mappings may be changed by their owner + * without the user having initiated it. + */ + if (vma_flags_is_kernel_owned(flags) || + vma_flags_is_fixed_mapping(flags)) + return false; + /* Droppable memory is discardable by definition. */ + return !vma_flags_test_single_mask(flags, VMA_DROPPABLE); +} + +/** + * vma_is_persistent() - Does the VMA contain persistent data? + * @vma: The VMA to test. + * + * See vma_flags_is_persistent() for details. + * + * Returns: true if the VMA is persistent, otherwise false. + */ +static inline bool vma_is_persistent(const struct vm_area_struct *vma) +{ + return vma_flags_is_persistent(&vma->flags); +} + /** * vma_kernel_pagesize - Default page size granularity for this VMA. * @vma: The user mapping. diff --git a/mm/madvise.c b/mm/madvise.c index 7a038837b2a8b7..f5307b3c2191f6 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -1585,8 +1585,8 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior) new_flags |= VM_DONTDUMP; break; case MADV_DODUMP: - if ((!vma_is_hugetlb(vma) && (new_flags & VM_SPECIAL)) || - (new_flags & VM_DROPPABLE)) + /* Non-persistent memory cannot be dumped. */ + if (!vma_is_persistent(vma)) return -EINVAL; new_flags &= ~VM_DONTDUMP; break; From 94ef9d53bc6d888313e3d057fe7efa46e05515e5 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:41 +0100 Subject: [PATCH 0676/1012] mm/uffd: use predicates for userfaultfd checks Rather than directly checking VMA flags, use the newly introduced vma_is_kernel_owned() and vma_is_persistent() helpers in userfaultfd when assessing VMA suitability for userfaultfd and UFFDIO_MOVE. Update vma_move_compatible() so it's expressed in terms of VMA characteristics rather than arbitrary flags. Additionally, update the use of the deprecated VMA flag API when checking VMA_SHADOW_STACK_BIT. A VMA_IO_BIT check is no longer required but that is fine as a hard invariant has been established that only kernel-owned mappings may set VMA_IO_BIT so the check is now redundant. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-32-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/userfaultfd.c | 21 +++++++++++++++------ 1 file changed, 15 insertions(+), 6 deletions(-) diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index 215b993d6df4fd..17ecbb0ceddf11 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -1755,10 +1755,18 @@ static inline bool move_splits_huge_pmd(unsigned long dst_addr, } #endif -static inline bool vma_move_compatible(struct vm_area_struct *vma) +static inline bool vma_move_compatible(const struct vm_area_struct *vma) { - return !(vma->vm_flags & (VM_PFNMAP | VM_IO | VM_HUGETLB | - VM_MIXEDMAP | VM_SHADOW_STACK)); + /* uffd is generally incompatible with kernel-owned mappings. */ + if (vma_is_kernel_owned(vma)) + return false; + /* The shadow stack should not be written to by userspace. */ + if (vma_test_single_mask(vma, VMA_SHADOW_STACK)) + return false; + /* hugetlb mappings cannot be safely moved. */ + if (vma_is_hugetlb(vma)) + return false; + return true; } static int validate_move_areas(struct userfaultfd_ctx *ctx, @@ -2147,10 +2155,11 @@ static bool vma_can_userfault(struct vm_area_struct *vma, vm_flags_t vm_flags, { const struct vm_uffd_ops *ops = vma_uffd_ops(vma); - if (vma->vm_flags & (VM_DROPPABLE | VM_SHADOW_STACK)) + /* Non-persistent memory is inherently not controllable by userspace. */ + if (!vma_is_persistent(vma)) return false; - - if (!vma_is_hugetlb(vma) && (vma->vm_flags & VM_SPECIAL)) + /* The shadow stack should not be written to by userspace. */ + if (vma_test_single_mask(vma, VMA_SHADOW_STACK)) return false; vm_flags &= __VM_UFFD_FLAGS; From 7d5600b30ae30bb0b117e9c0e387b8a3bd908cc5 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:42 +0100 Subject: [PATCH 0677/1012] mm/madvise: use predicates for madvise(..., MADV_DOFORK) Make it clear what we're blocking in MADV_DOFORK. Previously we simply disallowed VM_SPECIAL i.e. kernel-owned mappings, fixed mappings and VMA_IO_BIT. Now the invariant is established that only kernel-owned mappings can set VMA_IO_BIT, the VMA_IO_BIT check is redundant. The rest is equivalent to testing for a kernel-owned or fixed mapping, i.e. exactly the same check as whether the VMA is permitted to be merged. This was established by commit 0b2758f48f22 ("Require (reasonably) normal mappings for MADV_DOFORK") containing my hands-down favourite call out of all time. Express the same thing differently - if we wouldn't be allowed to merge it, then we aren't allowed to manipulate CoW behaviour on fork. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-33-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/madvise.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/madvise.c b/mm/madvise.c index f5307b3c2191f6..80ea991ce350a3 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -1566,7 +1566,7 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior) new_flags |= VM_DONTCOPY; break; case MADV_DOFORK: - if (new_flags & VM_SPECIAL) + if (!vma_can_merge(vma)) return -EINVAL; new_flags &= ~VM_DONTCOPY; break; From 70a8fd31a3207c619a89204d453144e1319619e9 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:43 +0100 Subject: [PATCH 0678/1012] mm: eliminate VMA_SPECIAL_FLAGS usage when hugetlb explicitly tested It is now an invariant that VMA_IO_BIT is not set except by kernel-owned mappings, so each existing VMA_SPECIAL_FLAGS test need only test for VMA_DONTEXPAND_BIT, VMA_PFNMAP_BIT and VMA_MIXEDMAP_BIT. This is precisely a test for a kernel-owned or fixed mapping. Update a number of callsites which already explicitly handle hugetlb mappings. vma_supports_mlock() and ksm_compatible() also explicitly bail on droppable mappings - detecting kernel-owned, fixed or droppable mappings is handled by vma_is_persistent(), so in these cases use this predicate. should_skip_vma() tests for locked, kernel-owned or fixed memory (having already excluded hugetlb mappings) so simply test for those there. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-34-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/internal.h | 4 +--- mm/ksm.c | 4 +--- mm/vmscan.c | 3 ++- 3 files changed, 4 insertions(+), 7 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index c63df7b7d77272..3b9fdb826162df 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -1113,9 +1113,7 @@ static inline struct file *maybe_unlock_mmap_for_io(struct vm_fault *vmf, static inline bool vma_supports_mlock(const struct vm_area_struct *vma) { - if (vma_test_any_mask(vma, VMA_SPECIAL_FLAGS)) - return false; - if (vma_test_single_mask(vma, VMA_DROPPABLE)) + if (!vma_is_persistent(vma)) return false; if (vma_is_dax(vma) || vma_is_hugetlb(vma)) return false; diff --git a/mm/ksm.c b/mm/ksm.c index 624f37975e1295..f80372bfd4b2fc 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -747,9 +747,7 @@ static bool ksm_compatible(const struct file *file, vma_flags_t vma_flags) if (vma_flags_test_any(&vma_flags, VMA_SHARED_BIT, VMA_MAYSHARE_BIT, VMA_HUGETLB_BIT)) return false; - if (vma_flags_test_single_mask(&vma_flags, VMA_DROPPABLE)) - return false; - if (vma_flags_test_any_mask(&vma_flags, VMA_SPECIAL_FLAGS)) + if (!vma_flags_is_persistent(&vma_flags)) return false; if (file_is_dax(file)) return false; diff --git a/mm/vmscan.c b/mm/vmscan.c index 9fc4282da0aab5..76f6ece5f20134 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3480,7 +3480,8 @@ static int should_skip_vma(unsigned long start, unsigned long end, struct mm_wal if (!vma_has_recency(vma)) return true; - if (vma->vm_flags & (VM_LOCKED | VM_SPECIAL)) + if (vma_test(vma, VMA_LOCKED_BIT) || vma_is_kernel_owned(vma) || + vma_is_fixed_mapping(vma)) return true; if (vma == get_gate_vma(vma->vm_mm)) From e5b71f7f462d5864aed0b7f6ed83dada11f65d8d Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:44 +0100 Subject: [PATCH 0679/1012] mm: eliminate VMA_SPECIAL_FLAGS check in lru_gen_look_around() A kernel-owned or fixed mapping is one which sets VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT or VMA_DONTEXPAND_BIT, which is precisely what VMA_SPECIAL_FLAGS tests for other than VMA_IO_BIT, which is safe to drop as only kernel-owned mappings may set it. Using these predicates rather than VMA_SPECIAL_FLAGS makes the check self-documenting and helps eliminate the confusion around 'special' flags. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-35-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/vmscan.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 76f6ece5f20134..836f50814ffae5 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -4421,8 +4421,8 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) if (spin_is_contended(pvmw->ptl)) return true; - /* exclude special VMAs containing anon pages from COW */ - if (vma->vm_flags & VM_SPECIAL) + /* exclude kernel-owned and fixed VMAs containing anon pages from COW */ + if (vma_is_kernel_owned(vma) || vma_is_fixed_mapping(vma)) return true; /* avoid taking the LRU lock under the PTL when possible */ From f0c76005fd27ba77927f7cc2a3c424d2b8d79a9b Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:45 +0100 Subject: [PATCH 0680/1012] mm: avoid use of VMA_SPECIAL_FLAGS in migrate_vma_setup() Now we have the expressive vma_is_kernel_owned() and vma_is_fixed_mapping() predicates, use them to determine whether to proceed with migration. This drops the VMA_IO_BIT test, which is safe as only kernel-owned mappings may set it. hugetlb mappings remain excluded, as they are fixed mappings. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-36-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- mm/migrate_device.c | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/mm/migrate_device.c b/mm/migrate_device.c index b9c453c28795f3..b74c0ae4276826 100644 --- a/mm/migrate_device.c +++ b/mm/migrate_device.c @@ -739,19 +739,21 @@ static void migrate_vma_unmap(struct migrate_vma *migrate) */ int migrate_vma_setup(struct migrate_vma *args) { + const struct vm_area_struct *vma = args->vma; long nr_pages = (args->end - args->start) >> PAGE_SHIFT; args->start &= PAGE_MASK; args->end &= PAGE_MASK; - if (!args->vma || vma_test_any_mask(args->vma, VMA_SPECIAL_FLAGS) || - vma_is_dax(args->vma)) + if (!vma) + return -EINVAL; + if (vma_is_kernel_owned(vma) || vma_is_fixed_mapping(vma) || + vma_is_dax(vma)) return -EINVAL; if (nr_pages <= 0) return -EINVAL; - if (args->start < args->vma->vm_start || - args->start >= args->vma->vm_end) + if (args->start < vma->vm_start || args->start >= vma->vm_end) return -EINVAL; - if (args->end <= args->vma->vm_start || args->end > args->vma->vm_end) + if (args->end <= vma->vm_start || args->end > vma->vm_end) return -EINVAL; if (!args->src || !args->dst) return -EINVAL; From 917ce234cbfda7a0a1bcea9d93b99afbade0db01 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:46 +0100 Subject: [PATCH 0681/1012] mm: eliminate VM_SPECIAL, VMA_SPECIAL_FLAGS Every user of the VM_SPECIAL or VMA_SPECIAL_FLAGS has now been converted to predicates which explicitly express what is actually being checked for rather than the nebulous concept of possessing 'special' VMA flags. In any case 'special' is not so special a term of art in mm - it includes VDSO/VVAR mappings, special in the sense of vm_normal_folio() and probably other cases too. Therefore make things less special by eliminating these now unused flags. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-37-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: Zi Yan Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie --- include/linux/mm.h | 8 -------- tools/testing/vma/include/dup.h | 8 -------- 2 files changed, 16 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index eaf3a4110f91ed..448384fdb594d2 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -577,14 +577,6 @@ enum { #define VM_ACCESS_FLAGS (VM_READ | VM_WRITE | VM_EXEC) #define VMA_ACCESS_FLAGS mk_vma_flags(VMA_READ_BIT, VMA_WRITE_BIT, VMA_EXEC_BIT) -/* - * Special vmas that are non-mergable, non-mlock()able. - */ - -#define VMA_SPECIAL_FLAGS mk_vma_flags(VMA_IO_BIT, VMA_DONTEXPAND_BIT, \ - VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT) -#define VM_SPECIAL vma_flags_to_legacy(VMA_SPECIAL_FLAGS) - /* * Physically remapped pages are special. Tell the * rest of the world about it: diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index 97d3bf6cd5b712..c21f67decab58d 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -352,14 +352,6 @@ enum { #define VM_ACCESS_FLAGS (VM_READ | VM_WRITE | VM_EXEC) #define VMA_ACCESS_FLAGS mk_vma_flags(VMA_READ_BIT, VMA_WRITE_BIT, VMA_EXEC_BIT) -/* - * Special vmas that are non-mergable, non-mlock()able. - */ -#define VM_SPECIAL (VM_IO | VM_DONTEXPAND | VM_PFNMAP | VM_MIXEDMAP) - -#define VMA_SPECIAL_FLAGS mk_vma_flags(VMA_IO_BIT, VMA_DONTEXPAND_BIT, \ - VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT) - #define VMA_REMAP_FLAGS mk_vma_flags(VMA_IO_BIT, VMA_PFNMAP_BIT, \ VMA_DONTEXPAND_BIT, VMA_DONTDUMP_BIT) From c1d718f89a97d1b2baf60ea96fedfb5c804d80fe Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:47 +0100 Subject: [PATCH 0682/1012] fuse: dax: do not set VM_MIXEDMAP Commit e1fb4a086495 ("dax: remove VM_MIXEDMAP for fsdax and device dax") prevented fsdax and device-dax from setting VM_MIXEDMAP, as DAX no longer relies on it to direct core mm paths. The fuse DAX implementation, added later, copied the old pattern and still sets it. Fuse DAX maps pages the same way fsdax does, via dax_iomap_fault() and ultimately vmf_insert_page_mkwrite() and vmf_insert_folio_pmd(), which insert ordinary refcounted pages and so do not require VM_MIXEDMAP. Setting it only serves to mark the mapping as kernel-owned, making fuse DAX the sole DAX implementation whose mappings are unmergeable, cannot be mlock()'d, eagerly copy page tables on fork and reject MADV_DOFORK and MADV_DODUMP. It also requires vma_is_special_huge() in mm/huge_memory.c to carve DAX out of its kernel-owned check explicitly. There is no reason for fuse DAX to keep on using this flag so drop it. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-38-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- fs/fuse/dax.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/fuse/dax.c b/fs/fuse/dax.c index 85cdf0199bc0b8..a5994f1c637d92 100644 --- a/fs/fuse/dax.c +++ b/fs/fuse/dax.c @@ -826,7 +826,7 @@ int fuse_dax_mmap(struct file *file, struct vm_area_struct *vma) { file_accessed(file); vma->vm_ops = &fuse_dax_vm_ops; - vm_flags_set(vma, VM_MIXEDMAP | VM_HUGEPAGE); + vma_set_flags(vma, VMA_HUGEPAGE_BIT); return 0; } From 2154041376ddf8298d4832e145c982723ddf39d0 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:48 +0100 Subject: [PATCH 0683/1012] mm/huge_memory: remove vma_is_special_huge() vma_is_special_huge() tests whether either the VMA_PFNMAP_BIT or VMA_MIXEDMAP_BIT is set (i.e. whether the VMA is a kernel-owned mapping), but with a DAX carve-out. DAX however no longer sets VMA_MIXEDMAP_BIT, so this carve-out is no longer required. Therefore test for vma_is_kernel_owned() instead and also drop the VMA_IO_BIT check, as it is now redundant since it is enforced that only kernel-owned mappings can set this flag. This also eliminates another overloaded use of 'special' within mm. No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-39-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- mm/huge_memory.c | 18 ++++-------------- 1 file changed, 4 insertions(+), 14 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index 05c17a01551df1..ec37a63b8a2ec7 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -110,14 +110,6 @@ static inline bool file_thp_enabled(const struct vm_area_struct *vma) return S_ISREG(inode->i_mode); } -/* If returns true, we are unable to access the VMA's folios. */ -static bool vma_is_special_huge(const struct vm_area_struct *vma) -{ - if (vma_is_dax(vma)) - return false; - return vma_test_any(vma, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT); -} - static bool vma_file_bypass_thp_tuneables(const struct vm_area_struct *vma, enum tva_type type) { @@ -192,7 +184,7 @@ unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma, /* Check the intersection of requested and supported orders. */ if (vma_is_anonymous(vma)) supported_orders = THP_ORDERS_ALL_ANON; - else if (vma_is_dax(vma) || vma_is_special_huge(vma)) + else if (vma_is_dax(vma) || vma_is_kernel_owned(vma)) supported_orders = THP_ORDERS_ALL_SPECIAL_DAX; else supported_orders = THP_ORDERS_ALL_FILE_DEFAULT; @@ -3065,7 +3057,7 @@ int zap_huge_pud(struct mmu_gather *tlb, struct vm_area_struct *vma, orig_pud = pudp_huge_get_and_clear_full(vma, addr, pud, tlb->fullmm); arch_check_zapped_pud(vma, orig_pud); tlb_remove_pud_tlb_entry(tlb, pud, addr); - if (vma_is_special_huge(vma)) { + if (vma_is_kernel_owned(vma)) { spin_unlock(ptl); /* No zero page support yet */ } else { @@ -3221,7 +3213,7 @@ static void __split_huge_pmd_locked(struct vm_area_struct *vma, pmd_t *pmd, */ if (arch_needs_pgtable_deposit()) zap_deposited_table(mm, pmd); - if (vma_is_special_huge(vma)) + if (vma_is_kernel_owned(vma)) return; if (unlikely(pmd_is_migration_entry(old_pmd))) { const softleaf_t old_entry = softleaf_from_pmd(old_pmd); @@ -4788,9 +4780,7 @@ static inline bool vma_not_suitable_for_thp_split(struct vm_area_struct *vma) { if (vma_is_dax(vma)) return true; - if (vma_is_special_huge(vma)) - return true; - if (vma_test(vma, VMA_IO_BIT)) + if (vma_is_kernel_owned(vma)) return true; if (vma_is_hugetlb(vma)) return true; From 70fd34de90fb69fd0f44ebfb72f0cb1e7fe539aa Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 17:22:49 +0100 Subject: [PATCH 0684/1012] mm/vma: introduce and use vma[_flags]_can_gup() GUP cannot be used for VMAs which set VMA_IO_BIT - because memory-mapped I/O must not be accessed on the user's behalf - or VMA_PFNMAP_BIT - because PFN maps have no folios which the kernel is permitted to access. Rather than keeping these checks open-coded, abstract them to vma_flags_can_gup() and its VMA wrapper vma_can_gup(). A number of other places make the same check to decide whether a mapping can be populated or accessed as GUP would, so update those too. While here, drop a reference to 'special' and replace a use of the deprecated VMA flags API in vma_dump_size(). No functional change intended. Link: https://lore.kernel.org/20260917-b4-mmap-prepare-vma-flag-sanify-v3-40-4583d8a23bca@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexei Starovoitov Cc: Alistair Popple Cc: Al Viro Cc: Andreas Larsson Cc: Andrii Nakryiko Cc: "Aneesh Kumar K.V" Cc: Anup Patel Cc: Arnaldo Carvalho de Melo Cc: Arnd Bergmann Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: "Borislav Petkov (AMD)" Cc: Byungchul Park Cc: Catalin Marinas Cc: Chengming Zhou Cc: Chris Li Cc: Christian Borntraeger Cc: Christian Brauner Cc: Claudio Imbrenda Cc: Dave Airlie Cc: Dave Hansen Cc: David Hildenbrand Cc: David S. Miller Cc: Dennis Dalessandro Cc: Dev Jain Cc: Doug Gilbert Cc: Eduard Zingerman Cc: Emil Tsalapatis Cc: Gerald Schaefer Cc: Greg Kroah-Hartman Cc: Gregory Price Cc: Harry Yoo Cc: Heiko Carstens Cc: Helge Deller Cc: "Huang, Ying" Cc: Ingo Molnar Cc: James Bottomley Cc: Jan Kara Cc: Jann Horn Cc: Janosch Frank Cc: Jaroslav Kysela Cc: Jason Gunthorpe Cc: Jaya Kumar Cc: Johannes Weiner Cc: John Hubbard Cc: Jonathan Corbet Cc: Joshua Hahn Cc: Juri Lelli Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Kumar Kartikeya Dwivedi Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Maarten Lankhorst Cc: Madhavan Srinivasan Cc: Marc Rutland Cc: Marc Zyngier Cc: "Masami Hiramatsu (Google)" Cc: Matthew Brost Cc: Matthew Wilcox (Oracle) Cc: Maxime Ripard Cc: Michal Hocko Cc: Michal Hocko Cc: Mike Rapoport Cc: Miklos Szeredi Cc: Muchun Song Cc: Namhyung kim Cc: Nhat Pham Cc: Nicholas Piggin Cc: Oleg Nesterov Cc: Oscar Salvador Cc: Palmer Dabbelt Cc: Paul Moore Cc: Pedro Falcato Cc: Peter Xu Cc: Peter Zijlstra Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Sebastian Reichel Cc: Shakeel Butt Cc: Stephen Smalley Cc: Suren Baghdasaryan Cc: Takashi Iwai (SUSE) Cc: Takashi Iwai Cc: Thomas Zimemrmann Cc: Vasily Gorbik Cc: Vincent Guittot Cc: Vlastimil Babka Cc: Wei Xu Cc: Will Deacon Cc: Yuanchu Xie Cc: Zi Yan --- fs/coredump.c | 4 ++-- include/linux/mm.h | 29 +++++++++++++++++++++++++++++ mm/gup.c | 7 +++---- mm/hmm.c | 3 +-- mm/memory.c | 14 ++++++++------ mm/mempolicy.c | 3 ++- 6 files changed, 45 insertions(+), 15 deletions(-) diff --git a/fs/coredump.c b/fs/coredump.c index 5820cb8ec88e70..4d03826dd24981 100644 --- a/fs/coredump.c +++ b/fs/coredump.c @@ -1616,8 +1616,8 @@ static unsigned long vma_dump_size(struct vm_area_struct *vma, return 0; } - /* Do not dump I/O mapped devices or special mappings */ - if (vma->vm_flags & VM_IO) + /* Do not dump memory-mapped I/O, which may have side effects on read. */ + if (vma_test(vma, VMA_IO_BIT)) return 0; /* By default, dump shared memory if mapped from an anonymous file. */ diff --git a/include/linux/mm.h b/include/linux/mm.h index 448384fdb594d2..4fd47cc796a6c1 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -1778,6 +1778,35 @@ static inline bool vma_is_persistent(const struct vm_area_struct *vma) return vma_flags_is_persistent(&vma->flags); } +/** + * vma_flags_can_gup() - Do the specified VMA flags permit GUP to access the + * mapping's pages? + * @flags: The VMA flags to test. + * + * GUP cannot obtain pages from a PFN map (VMA_PFNMAP_BIT), which may have no + * struct pages behind it, and must not provide access to memory-mapped I/O + * (VMA_IO_BIT). + * + * Returns: true if GUP may access pages from the mapping, otherwise false. + */ +static inline bool vma_flags_can_gup(const vma_flags_t *flags) +{ + return !vma_flags_test_any(flags, VMA_IO_BIT, VMA_PFNMAP_BIT); +} + +/** + * vma_can_gup() - May GUP obtain pages from @vma? + * @vma: The VMA to test. + * + * See vma_flags_can_gup() for details. + * + * Returns: true if GUP may access pages from the mapping, otherwise false. + */ +static inline bool vma_can_gup(const struct vm_area_struct *vma) +{ + return vma_flags_can_gup(&vma->flags); +} + /** * vma_kernel_pagesize - Default page size granularity for this VMA. * @vma: The user mapping. diff --git a/mm/gup.c b/mm/gup.c index f166acf794e308..8e9ef5ee7498c6 100644 --- a/mm/gup.c +++ b/mm/gup.c @@ -1204,7 +1204,7 @@ static int check_vma_flags(struct vm_area_struct *vma, unsigned long gup_flags) int foreign = (gup_flags & FOLL_REMOTE); bool vma_anon = vma_is_anonymous(vma); - if (vm_flags & (VM_IO | VM_PFNMAP)) + if (!vma_can_gup(vma)) return -EFAULT; if ((gup_flags & FOLL_ANON) && !vma_anon) @@ -1955,7 +1955,7 @@ int __mm_populate(unsigned long start, unsigned long len, int ignore_errors) * range with the first VMA. Also, skip undesirable VMA types. */ nend = min(end, vma->vm_end); - if (vma->vm_flags & (VM_IO | VM_PFNMAP)) + if (!vma_can_gup(vma)) continue; if (nstart < vma->vm_start) nstart = vma->vm_start; @@ -2017,8 +2017,7 @@ static long __get_user_pages_locked(struct mm_struct *mm, unsigned long start, break; /* protect what we can, including chardevs */ - if ((vma->vm_flags & (VM_IO | VM_PFNMAP)) || - !(vm_flags & vma->vm_flags)) + if (!vma_can_gup(vma) || !(vm_flags & vma->vm_flags)) break; if (pages) { diff --git a/mm/hmm.c b/mm/hmm.c index 2f1e98c6b6440b..e9569b82a1f0cc 100644 --- a/mm/hmm.c +++ b/mm/hmm.c @@ -595,8 +595,7 @@ static int hmm_vma_walk_test(unsigned long start, unsigned long end, struct hmm_range *range = hmm_vma_walk->range; struct vm_area_struct *vma = walk->vma; - if (!(vma->vm_flags & (VM_IO | VM_PFNMAP)) && - vma->vm_flags & VM_READ) + if (vma_can_gup(vma) && vma_test(vma, VMA_READ_BIT)) return 0; /* diff --git a/mm/memory.c b/mm/memory.c index 6c011979401aab..338fce99e71197 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2417,11 +2417,11 @@ static bool vm_mixed_zeropage_allowed(struct vm_area_struct *vma) * be problematic as soon as the zeropage gets replaced by a different * page due to vma->vm_ops->pfn_mkwrite, because what's mapped would * now differ to what GUP looked up. FSDAX is incompatible to - * FOLL_LONGTERM and VM_IO is incompatible to GUP completely (see - * check_vma_flags). + * FOLL_LONGTERM and memory-mapped I/O is incompatible to GUP completely + * (see vma_can_gup()). */ return vma->vm_ops && vma->vm_ops->pfn_mkwrite && - (vma_is_fsdax(vma) || vma->vm_flags & VM_IO); + (vma_is_fsdax(vma) || vma_test(vma, VMA_IO_BIT)); } static int validate_page_before_insert(struct vm_area_struct *vma, @@ -7116,7 +7116,8 @@ int follow_pfnmap_start(struct follow_pfnmap_args *args) if (unlikely(address < vma->vm_start || address >= vma->vm_end)) goto out; - if (!(vma->vm_flags & (VM_IO | VM_PFNMAP))) + /* Only mappings GUP cannot handle are followed here. */ + if (vma_can_gup(vma)) goto out; retry: pgdp = pgd_offset(mm, address); @@ -7316,8 +7317,9 @@ static int __access_remote_vm(struct mm_struct *mm, unsigned long addr, } /* - * Check if this is a VM_IO | VM_PFNMAP VMA, which - * we can access using slightly different code. + * GUP failed, perhaps because this is a mapping it + * cannot handle (see vma_can_gup()) - such mappings may + * provide access via vm_ops->access() instead. */ bytes = 0; #ifdef CONFIG_HAVE_IOREMAP_PROT diff --git a/mm/mempolicy.c b/mm/mempolicy.c index 70298fded1b4a8..40744658483b27 100644 --- a/mm/mempolicy.c +++ b/mm/mempolicy.c @@ -2008,7 +2008,8 @@ SYSCALL_DEFINE5(get_mempolicy, int __user *, policy, bool vma_migratable(struct vm_area_struct *vma) { - if (vma->vm_flags & (VM_IO | VM_PFNMAP)) + /* Pages which GUP cannot obtain cannot be migrated either. */ + if (!vma_can_gup(vma)) return false; /* From 602baa2ffe1cdf040fe8717af6b7463b2628d2bc Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:18 -0700 Subject: [PATCH 0685/1012] mm/damon/sysfs-schemes: read sysfs_filter->addr_range only once Patch series "mm/damon: move damos filter range arguments validation to core". DAMOS filter range arguments are being validated in the DAMON sysfs interface. Some of those have minor time-of-check to time-of-use (TOCTOU) bugs. In future, other callers might have duplicated validations with similar bugs. Fix the bugs and further refactor the code to move the validation to the core layer. Also add kunit test cases for the validations. Patches 1 and 2 fix the existing TOCTOU bugs. Patches 3 and 4 adds the validation in the core layer. Patch 5 drops the replicated validation in DAMON sysfs interface. Patch 6 further cleanup the code. Patches 7 and 8 extends kunit tests to test the validation. This patch (of 8): DAMON sysfs interface reads the user-provided address range arguments for addr type DAMOS filter twice. Once for validation, and once again for assignments to the variable that will be passed to the core layer. If the user updates the argument in parallel, an invalid address range could be passed to the core layer. Avoid it by doing the assignments first, and then validating the assigned variables before passing those to the core layer. User impact of the bug should be trivial. From the core layer's perspective, the invalid address range is not really invalid. It just works as having a weird address range. No critical issues such as a crash or a leak could happen. And sane users ain't do such parallel arguments update anyway. If they do, such racy behavior is arguably somewhat expected and deserved. That said, there is no reason to keep such races. Link: https://lore.kernel.org/20260914142327.92510-1-sj@kernel.org Link: https://lore.kernel.org/20260914142327.92510-2-sj@kernel.org Fixes: 2f1abcfccd86 ("mm/damon/sysfs-schemes: support address range type DAMOS filter") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/sysfs-schemes.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index 3de4d804e049f8..3c1c1cb387fec2 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -2831,12 +2831,12 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, return err; } } else if (filter->type == DAMOS_FILTER_TYPE_ADDR) { - if (sysfs_filter->addr_range.end < - sysfs_filter->addr_range.start) { + filter->addr_range = sysfs_filter->addr_range; + if (filter->addr_range.end < + filter->addr_range.start) { damos_destroy_filter(filter); return -EINVAL; } - filter->addr_range = sysfs_filter->addr_range; } else if (filter->type == DAMOS_FILTER_TYPE_TARGET) { filter->target_idx = sysfs_filter->target_idx; } else if (filter->type == DAMOS_FILTER_TYPE_HUGEPAGE_SIZE) { From 1a7605db770a2ada470cef3ab449bef09801e262 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:19 -0700 Subject: [PATCH 0686/1012] mm/damon/sysfs-schemes: read sysfs_filter->sz_range only once DAMON sysfs interface reads the user-provided size range arguments for hugepage_size type DAMOS filter twice. Once for validation, and once again for assignments to the variable that will be passed to the core layer. If the user updates the arguments in parallel, an invalid size range could be passed to the core layer. Avoid it by doing the assignments first, and then validating the assigned variables before passing those to the core layer. User impact of the bug should be trivial. From the core layer's perspective, the invalid size range is not really invalid. It just works as having a weird size range. No critical issues such as a crash or a leak could happen. And sane users ain't do such parallel arguments update anyway. If they do, such racy behavior is arguably somewhat expected and deserved. That said, there is no reason to keep such races. Link: https://lore.kernel.org/20260914142327.92510-3-sj@kernel.org Fixes: ea1f204ba29a ("mm/damon/sysfs-schemes: add files for setting damos_filter->sz_range") Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/sysfs-schemes.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index 3c1c1cb387fec2..8c8ab82c8facc9 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -2840,13 +2840,12 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, } else if (filter->type == DAMOS_FILTER_TYPE_TARGET) { filter->target_idx = sysfs_filter->target_idx; } else if (filter->type == DAMOS_FILTER_TYPE_HUGEPAGE_SIZE) { - if (sysfs_filter->range_min > - sysfs_filter->range_max) { + filter->sz_range.min = sysfs_filter->range_min; + filter->sz_range.max = sysfs_filter->range_max; + if (filter->range_min > filter->range_max) { damos_destroy_filter(filter); return -EINVAL; } - filter->sz_range.min = sysfs_filter->range_min; - filter->sz_range.max = sysfs_filter->range_max; } else if (filter->type == DAMOS_FILTER_TYPE_PROBE_HITS_WSUM) { filter->range_min = sysfs_filter->range_min; filter->range_max = sysfs_filter->range_max; From d991a9f99c1e701bd600754035a1f5966cda5619 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:20 -0700 Subject: [PATCH 0687/1012] mm/damon/core: return an error from damos_commit_filter_arg() damos_commit_filter_arg() is supposed to always succeed. It may not in future, for example, if the given filter is invalid. Prepare the case by modifying its signature to return an error when it failed. Also pipe the return value to its callers and let them handle the error. Link: https://lore.kernel.org/20260914142327.92510-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Kunwu Chan Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/core.c | 41 ++++++++++++++++++++++++++++------------- 1 file changed, 28 insertions(+), 13 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 38383ced3dd6c0..1932a50252079b 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1317,7 +1317,7 @@ static struct damos_filter *damos_nth_ops_filter(int n, struct damos *s) return NULL; } -static void damos_commit_filter_arg( +static int damos_commit_filter_arg( struct damos_filter *dst, struct damos_filter *src) { switch (dst->type) { @@ -1340,28 +1340,32 @@ static void damos_commit_filter_arg( default: break; } + return 0; } -static void damos_commit_filter( +static int damos_commit_filter( struct damos_filter *dst, struct damos_filter *src) { dst->type = src->type; dst->matching = src->matching; dst->allow = src->allow; - damos_commit_filter_arg(dst, src); + return damos_commit_filter_arg(dst, src); } static int damos_commit_core_filters(struct damos *dst, struct damos *src) { struct damos_filter *dst_filter, *next, *src_filter, *new_filter; - int i = 0, j = 0; + int i = 0, j = 0, err; damos_for_each_core_filter_safe(dst_filter, next, dst) { src_filter = damos_nth_core_filter(i++, src); - if (src_filter) - damos_commit_filter(dst_filter, src_filter); - else + if (src_filter) { + err = damos_commit_filter(dst_filter, src_filter); + if (err) + return err; + } else { damos_destroy_filter(dst_filter); + } } damos_for_each_core_filter_safe(src_filter, next, src) { @@ -1373,7 +1377,11 @@ static int damos_commit_core_filters(struct damos *dst, struct damos *src) src_filter->allow); if (!new_filter) return -ENOMEM; - damos_commit_filter_arg(new_filter, src_filter); + err = damos_commit_filter_arg(new_filter, src_filter); + if (err) { + damos_destroy_filter(new_filter); + return err; + } damos_add_filter(dst, new_filter); } return 0; @@ -1382,14 +1390,17 @@ static int damos_commit_core_filters(struct damos *dst, struct damos *src) static int damos_commit_ops_filters(struct damos *dst, struct damos *src) { struct damos_filter *dst_filter, *next, *src_filter, *new_filter; - int i = 0, j = 0; + int i = 0, j = 0, err; damos_for_each_ops_filter_safe(dst_filter, next, dst) { src_filter = damos_nth_ops_filter(i++, src); - if (src_filter) - damos_commit_filter(dst_filter, src_filter); - else + if (src_filter) { + err = damos_commit_filter(dst_filter, src_filter); + if (err) + return err; + } else { damos_destroy_filter(dst_filter); + } } damos_for_each_ops_filter_safe(src_filter, next, src) { @@ -1401,7 +1412,11 @@ static int damos_commit_ops_filters(struct damos *dst, struct damos *src) src_filter->allow); if (!new_filter) return -ENOMEM; - damos_commit_filter_arg(new_filter, src_filter); + err = damos_commit_filter_arg(new_filter, src_filter); + if (err) { + damos_destroy_filter(new_filter); + return err; + } damos_add_filter(dst, new_filter); } return 0; From 283d4e0502e2f568e6634f4b08e16ebaa3fa9ab1 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:21 -0700 Subject: [PATCH 0688/1012] mm/damon/core: disallow max < min damos filter range arguments commit damos_commit_filter_arg() receives range arguments for a few types of DAMOS filters. It allows any range including max < min range. It is fine for the logic, but makes no sense to support it. Actually DAMON sysfs interface is doing the validation on its own. To avoid duplicated validations in multiple DAMON API callers, it would be better to do the validation in the core layer. Add a validation of the given range. Link: https://lore.kernel.org/20260914142327.92510-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/core.c | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 1932a50252079b..5fdacb9dcee5f6 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1325,15 +1325,21 @@ static int damos_commit_filter_arg( dst->memcg_id = src->memcg_id; break; case DAMOS_FILTER_TYPE_ADDR: + if (src->addr_range.end < src->addr_range.start) + return -EINVAL; dst->addr_range = src->addr_range; break; case DAMOS_FILTER_TYPE_TARGET: dst->target_idx = src->target_idx; break; case DAMOS_FILTER_TYPE_HUGEPAGE_SIZE: + if (src->sz_range.max < src->sz_range.min) + return -EINVAL; dst->sz_range = src->sz_range; break; case DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: + if (src->range_max < src->range_min) + return -EINVAL; dst->range_min = src->range_min; dst->range_max = src->range_max; break; From 5c570547fcaf9173feb0fe30965ab2b16d84dd4b Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:22 -0700 Subject: [PATCH 0689/1012] mm/damon/sysfs-schemes: drop centralized filter range arg validations DAMON sysfs interface is validating wrong range arguments for DAMOS filters. Now the core layer is doing the same validation. Drop the duplicated validation in DAMON sysfs interface. Link: https://lore.kernel.org/20260914142327.92510-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/sysfs-schemes.c | 13 ------------- 1 file changed, 13 deletions(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index 8c8ab82c8facc9..a4ec5d54cfbd14 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -2832,27 +2832,14 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, } } else if (filter->type == DAMOS_FILTER_TYPE_ADDR) { filter->addr_range = sysfs_filter->addr_range; - if (filter->addr_range.end < - filter->addr_range.start) { - damos_destroy_filter(filter); - return -EINVAL; - } } else if (filter->type == DAMOS_FILTER_TYPE_TARGET) { filter->target_idx = sysfs_filter->target_idx; } else if (filter->type == DAMOS_FILTER_TYPE_HUGEPAGE_SIZE) { filter->sz_range.min = sysfs_filter->range_min; filter->sz_range.max = sysfs_filter->range_max; - if (filter->range_min > filter->range_max) { - damos_destroy_filter(filter); - return -EINVAL; - } } else if (filter->type == DAMOS_FILTER_TYPE_PROBE_HITS_WSUM) { filter->range_min = sysfs_filter->range_min; filter->range_max = sysfs_filter->range_max; - if (filter->range_min > filter->range_max) { - damos_destroy_filter(filter); - return -EINVAL; - } } damos_add_filter(scheme, filter); From 1f8425ce4100f840ab641f1af795471af5a727f2 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:23 -0700 Subject: [PATCH 0690/1012] mm/damon/sysfs-schemes: use switch-case in add_scheme_filters() damon_sysfs_add_scheeme_filters() has long if-else chains for DAMOS filter types. Convert the code to use switch-case, which would be cleaner and more efficient. Link: https://lore.kernel.org/20260914142327.92510-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/sysfs-schemes.c | 18 +++++++++++++----- 1 file changed, 13 insertions(+), 5 deletions(-) diff --git a/mm/damon/sysfs-schemes.c b/mm/damon/sysfs-schemes.c index a4ec5d54cfbd14..bfb6f0bc3f2138 100644 --- a/mm/damon/sysfs-schemes.c +++ b/mm/damon/sysfs-schemes.c @@ -2822,7 +2822,8 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, if (!filter) return -ENOMEM; - if (filter->type == DAMOS_FILTER_TYPE_MEMCG) { + switch (filter->type) { + case DAMOS_FILTER_TYPE_MEMCG: err = damon_sysfs_memcg_path_to_id( sysfs_filter->memcg_path, &filter->memcg_id); @@ -2830,16 +2831,23 @@ static int damon_sysfs_add_scheme_filters(struct damos *scheme, damos_destroy_filter(filter); return err; } - } else if (filter->type == DAMOS_FILTER_TYPE_ADDR) { + break; + case DAMOS_FILTER_TYPE_ADDR: filter->addr_range = sysfs_filter->addr_range; - } else if (filter->type == DAMOS_FILTER_TYPE_TARGET) { + break; + case DAMOS_FILTER_TYPE_TARGET: filter->target_idx = sysfs_filter->target_idx; - } else if (filter->type == DAMOS_FILTER_TYPE_HUGEPAGE_SIZE) { + break; + case DAMOS_FILTER_TYPE_HUGEPAGE_SIZE: filter->sz_range.min = sysfs_filter->range_min; filter->sz_range.max = sysfs_filter->range_max; - } else if (filter->type == DAMOS_FILTER_TYPE_PROBE_HITS_WSUM) { + break; + case DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: filter->range_min = sysfs_filter->range_min; filter->range_max = sysfs_filter->range_max; + break; + default: + break; } damos_add_filter(scheme, filter); From d255baf5f8d07bd3074dec46ebfadd854ed19ef4 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:24 -0700 Subject: [PATCH 0691/1012] mm/damon/core-kunit: extend damos_commit_filter_for() for wrong input damos_commit_filter_for() supposes damos_commit_filter() to always succeed with given inputs. damos_commit_filter() could return an error for invalid inputs. Existing callers always pass only valid inputs, but they may pass invalid inputs in future, for test purposes. Extend the function to be able to be used for wrong inputs-caused error testing. Link: https://lore.kernel.org/20260914142327.92510-8-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/tests/core-kunit.h | 24 +++++++++++++++--------- 1 file changed, 15 insertions(+), 9 deletions(-) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index c01e6a75cadc1e..2b0931cf6fb326 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1116,9 +1116,15 @@ static void damos_test_commit_dests(struct kunit *test) } static void damos_test_commit_filter_for(struct kunit *test, - struct damos_filter *dst, struct damos_filter *src) + struct damos_filter *dst, struct damos_filter *src, + bool expect_fail) { - damos_commit_filter(dst, src); + int err; + + err = damos_commit_filter(dst, src); + KUNIT_EXPECT_EQ(test, err != 0, expect_fail); + if (expect_fail) + return; KUNIT_EXPECT_EQ(test, dst->type, src->type); KUNIT_EXPECT_EQ(test, dst->matching, src->matching); KUNIT_EXPECT_EQ(test, dst->allow, src->allow); @@ -1157,47 +1163,47 @@ static void damos_test_commit_filter(struct kunit *test) .type = DAMOS_FILTER_TYPE_ANON, .matching = true, .allow = true, - }); + }, false); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_MEMCG, .matching = false, .allow = false, .memcg_id = 123, - }); + }, false); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_YOUNG, .matching = true, .allow = true, - }); + }, false); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_HUGEPAGE_SIZE, .matching = false, .allow = false, .sz_range = {.min = 234, .max = 345}, - }); + }, false); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_UNMAPPED, .matching = true, .allow = true, - }); + }, false); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_ADDR, .matching = false, .allow = false, .addr_range = {.start = 456, .end = 567}, - }); + }, false); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_TARGET, .matching = true, .allow = true, .target_idx = 6, - }); + }, false); } static void damos_test_help_initailize_scheme(struct damos *scheme) From 79f56f9d15b904cd06b0bc9cd392eb4db5bf0c06 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Mon, 14 Sep 2026 07:23:25 -0700 Subject: [PATCH 0692/1012] mm/damon/core-kunit: test invalid damos filter commits Add test cases for testing the validation of damos filter arguments in commit time. Link: https://lore.kernel.org/20260914142327.92510-9-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: Usama Arif Cc: --- mm/damon/tests/core-kunit.h | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 2b0931cf6fb326..527abc25706167 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1184,6 +1184,13 @@ static void damos_test_commit_filter(struct kunit *test) .allow = false, .sz_range = {.min = 234, .max = 345}, }, false); + damos_test_commit_filter_for(test, &dst, + &(struct damos_filter){ + .type = DAMOS_FILTER_TYPE_HUGEPAGE_SIZE, + .matching = false, + .allow = false, + .sz_range = {.min = 456, .max = 123}, + }, true); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_UNMAPPED, @@ -1197,6 +1204,13 @@ static void damos_test_commit_filter(struct kunit *test) .allow = false, .addr_range = {.start = 456, .end = 567}, }, false); + damos_test_commit_filter_for(test, &dst, + &(struct damos_filter){ + .type = DAMOS_FILTER_TYPE_ADDR, + .matching = false, + .allow = false, + .addr_range = {.start = 567, .end = 456}, + }, true); damos_test_commit_filter_for(test, &dst, &(struct damos_filter){ .type = DAMOS_FILTER_TYPE_TARGET, From ef8ee1cdf9d012609b181fd04b6d2da4824ff56e Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Mon, 14 Sep 2026 07:19:46 -0700 Subject: [PATCH 0693/1012] selftests/damon: stop kdamond on error exits of no-op commit test Patch series "mm/damon: misc improvements in tests and documents". Add various improvements to DAMON tests and documents. Patch 1 and 2 from Zenghui Yu (Huawei) let DAMON selftest to clean up its state and output after running the tests. Patch 3 from Eva Kurchatova makes a few selftest be able to run with Python's safe path mode. Patch 4 from Kunwu Chan improves coverage of DAMON kunit tests. Finally, patches 5 and 6 from Liew Rui Yan clarify and fix typos on documents for DAMOS quota behaviors. Below are notes that can be removed in the final commit message. This is a batched repost of DAMON patches that were individually posted. Please refer to each patch for changelog. This patch (of 6): The sysfs_no_op_commit_break test starts a kdamond via sysfs, but its error paths (e.g., drgn not installed) exit without stopping it. The leaked kdamond then makes subsequent tests, e.g. lru_sort.sh and reclaim.sh, skip with "Another kdamond is running". Wrap the post-start logic in try-finally so that kdamonds.stop() is executed on every exit path. Link: https://lore.kernel.org/20260914141952.91465-1-sj@kernel.org Link: https://lore.kernel.org/20260914141952.91465-2-sj@kernel.org Fixes: 10725cd2b09a ("selftests/damon: test no-op commit broke DAMON status") Signed-off-by: Zenghui Yu (Huawei) Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Assisted-by: GLM-5.3 OpenCode Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Eva Kurchatova Cc: Kunwu Chan Cc: Lian Wang Cc: Liew Rui Yan --- .../damon/sysfs_no_op_commit_break.py | 35 ++++++++++--------- 1 file changed, 18 insertions(+), 17 deletions(-) diff --git a/tools/testing/selftests/damon/sysfs_no_op_commit_break.py b/tools/testing/selftests/damon/sysfs_no_op_commit_break.py index 2c65cffe6b5450..81774ddc55ac45 100755 --- a/tools/testing/selftests/damon/sysfs_no_op_commit_break.py +++ b/tools/testing/selftests/damon/sysfs_no_op_commit_break.py @@ -47,26 +47,27 @@ def main(): print('kdamond start failed: %s' % err) exit(1) - before_commit_status, err = \ - dump_damon_status_dict(kdamonds.kdamonds[0].pid) - if err is not None: - print('before-commit status dump failed: %s' % err) - exit(1) + try: + before_commit_status, err = \ + dump_damon_status_dict(kdamonds.kdamonds[0].pid) + if err is not None: + print('before-commit status dump failed: %s' % err) + exit(1) - kdamonds.kdamonds[0].commit() + kdamonds.kdamonds[0].commit() - after_commit_status, err = \ - dump_damon_status_dict(kdamonds.kdamonds[0].pid) - if err is not None: - print('after-commit status dump failed: %s' % err) - exit(1) - - if before_commit_status != after_commit_status: - print(f'before: {json.dumps(before_commit_status, indent=2)}') - print(f'after: {json.dumps(after_commit_status, indent=2)}') - exit(1) + after_commit_status, err = \ + dump_damon_status_dict(kdamonds.kdamonds[0].pid) + if err is not None: + print('after-commit status dump failed: %s' % err) + exit(1) - kdamonds.stop() + if before_commit_status != after_commit_status: + print(f'before: {json.dumps(before_commit_status, indent=2)}') + print(f'after: {json.dumps(after_commit_status, indent=2)}') + exit(1) + finally: + kdamonds.stop() if __name__ == '__main__': main() From 0235cd3553f86bd68534e342ca9222b7c8fb9eea Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Mon, 14 Sep 2026 07:19:47 -0700 Subject: [PATCH 0694/1012] selftests/damon: ignore test-generated damon_dump_output sysfs.py and sysfs_no_op_commit_break.py dump the DAMON status collected via drgn_dump_damon_status.py into a damon_dump_output file in the working directory. Ignore it so that running the tests in-tree does not pollute git status. Link: https://lore.kernel.org/20260914141952.91465-3-sj@kernel.org Signed-off-by: Zenghui Yu (Huawei) Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Eva Kurchatova Cc: Jonathan Corbet Cc: Kunwu Chan Cc: Liam R. Howlett Cc: Lian Wang Cc: Liew Rui Yan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/damon/.gitignore | 1 + 1 file changed, 1 insertion(+) diff --git a/tools/testing/selftests/damon/.gitignore b/tools/testing/selftests/damon/.gitignore index 2f0297657c8167..226bd7c4e07286 100644 --- a/tools/testing/selftests/damon/.gitignore +++ b/tools/testing/selftests/damon/.gitignore @@ -1,3 +1,4 @@ # SPDX-License-Identifier: GPL-2.0-only access_memory access_memory_even +damon_dump_output From aa152af8cd930f9eb23ba437e4c3ef0846c34b49 Mon Sep 17 00:00:00 2001 From: Eva Kurchatova Date: Mon, 14 Sep 2026 07:19:48 -0700 Subject: [PATCH 0695/1012] selftests/damon: add script dir to sys.path for PYTHONSAFEPATH compatibility Running these tests under Python's safe path mode, either with -P on the interpreter command line or with PYTHONSAFEPATH set in the environment, stops the script's own directory being prepended to sys.path. Some distributions (RHEL, for example) build the tests with -P in the shebang. This breaks all 7 DAMON Python selftests that import the _damon_sysfs helper module located in the same directory: ModuleNotFoundError: No module named '_damon_sysfs' Fix this by explicitly adding the script's directory to sys.path before importing _damon_sysfs, following the same pattern used in commit c3b3eb565bd7 ("tools: ynl: add script dir to sys.path") which fixed the identical issue for the YNL tools. Link: https://lore.kernel.org/20260914141952.91465-4-sj@kernel.org Signed-off-by: Eva Kurchatova Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Kunwu Chan Cc: Liam R. Howlett Cc: Lian Wang Cc: Liew Rui Yan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: "Zenghui Yu (Huawei)" --- tools/testing/selftests/damon/damon_nr_regions.py | 3 +++ tools/testing/selftests/damon/damos_apply_interval.py | 3 +++ tools/testing/selftests/damon/damos_quota.py | 3 +++ tools/testing/selftests/damon/damos_quota_goal.py | 3 +++ tools/testing/selftests/damon/damos_tried_regions.py | 3 +++ .../selftests/damon/sysfs_update_schemes_tried_regions_hang.py | 3 +++ .../damon/sysfs_update_schemes_tried_regions_wss_estimation.py | 3 +++ 7 files changed, 21 insertions(+) diff --git a/tools/testing/selftests/damon/damon_nr_regions.py b/tools/testing/selftests/damon/damon_nr_regions.py index 58f3291fed12a4..e55239813c654e 100755 --- a/tools/testing/selftests/damon/damon_nr_regions.py +++ b/tools/testing/selftests/damon/damon_nr_regions.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def test_nr_regions(real_nr_regions, min_nr_regions, max_nr_regions): diff --git a/tools/testing/selftests/damon/damos_apply_interval.py b/tools/testing/selftests/damon/damos_apply_interval.py index 0f2f36584e48cb..0bf7768b2006ac 100755 --- a/tools/testing/selftests/damon/damos_apply_interval.py +++ b/tools/testing/selftests/damon/damos_apply_interval.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def main(): diff --git a/tools/testing/selftests/damon/damos_quota.py b/tools/testing/selftests/damon/damos_quota.py index 57c4937aaed285..879115a499bf8f 100755 --- a/tools/testing/selftests/damon/damos_quota.py +++ b/tools/testing/selftests/damon/damos_quota.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def main(): diff --git a/tools/testing/selftests/damon/damos_quota_goal.py b/tools/testing/selftests/damon/damos_quota_goal.py index 661e4ba4765ae1..fed033a0afdf9c 100755 --- a/tools/testing/selftests/damon/damos_quota_goal.py +++ b/tools/testing/selftests/damon/damos_quota_goal.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def main(): diff --git a/tools/testing/selftests/damon/damos_tried_regions.py b/tools/testing/selftests/damon/damos_tried_regions.py index d6472e6a6e082b..6941f87c10b1ce 100755 --- a/tools/testing/selftests/damon/damos_tried_regions.py +++ b/tools/testing/selftests/damon/damos_tried_regions.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def main(): diff --git a/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_hang.py b/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_hang.py index 28c887a0108fde..625761c243b58b 100755 --- a/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_hang.py +++ b/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_hang.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def main(): diff --git a/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py b/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py index 16fdc6e7fc566a..36e7ae5f826d82 100755 --- a/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py +++ b/tools/testing/selftests/damon/sysfs_update_schemes_tried_regions_wss_estimation.py @@ -1,9 +1,12 @@ #!/usr/bin/env python3 # SPDX-License-Identifier: GPL-2.0 +import os import subprocess +import sys import time +sys.path.append(os.path.dirname(os.path.abspath(__file__))) import _damon_sysfs def pass_wss_estimation(sz_region): From 8af945b6b028cc1a6474e8709a41a1b6f100bf5a Mon Sep 17 00:00:00 2001 From: Kunwu Chan Date: Mon, 14 Sep 2026 07:19:49 -0700 Subject: [PATCH 0696/1012] mm/damon/tests/core-kunit: improve nr_samples_per_aggr test isolation Test the zero sample interval case with a non-zero aggregation interval, and keep the zero/zero case to cover the zero-result fallback. Also make the overflow case use an explicit sample interval so that it does not depend on the zero sample interval fallback. Link: https://lore.kernel.org/20260914141952.91465-5-sj@kernel.org Signed-off-by: Kunwu Chan Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Lian Wang Reviewed-by: SJ Park Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Eva Kurchatova Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Liew Rui Yan Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: "Zenghui Yu (Huawei)" --- mm/damon/tests/core-kunit.h | 19 +++++++++++++++---- 1 file changed, 15 insertions(+), 4 deletions(-) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 527abc25706167..5da84caf4124d5 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -627,12 +627,20 @@ static void damon_test_set_regions(struct kunit *test) static void damon_test_nr_samples_per_aggr(struct kunit *test) { - struct damon_attrs attrs = { + struct damon_attrs attrs; + + /* Zero sample interval is treated as one. */ + attrs = (struct damon_attrs){ .sample_interval = 0, - .aggr_interval = 0, + .aggr_interval = 5000, }; + KUNIT_EXPECT_EQ(test, damon_nr_samples_per_aggr(&attrs), 5000); - /* Zero aggregation interval doesn't cause division by zero */ + /* Zero sample and aggregation intervals cover the zero-result fallback. */ + attrs = (struct damon_attrs){ + .sample_interval = 0, + .aggr_interval = 0, + }; KUNIT_EXPECT_EQ(test, damon_nr_samples_per_aggr(&attrs), 1); /* @@ -640,7 +648,10 @@ static void damon_test_nr_samples_per_aggr(struct kunit *test) * overflow */ if (ULONG_MAX > UINT_MAX) { - attrs.aggr_interval = (unsigned long)UINT_MAX + 1; + attrs = (struct damon_attrs){ + .sample_interval = 1, + .aggr_interval = (unsigned long)UINT_MAX + 1, + }; KUNIT_EXPECT_EQ(test, damon_nr_samples_per_aggr(&attrs), UINT_MAX); } From c1c9707b027a4d89de8cd0367e3745978cff4501 Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 14 Sep 2026 07:19:50 -0700 Subject: [PATCH 0697/1012] Docs/mm/damon/design: clarify when qt_exceeds increases qt_exceeds counts how many times a scheme's quota has been exceeded. When using the temporal auto-tuning algorithm, the effective size quota becomes zero once the goal is [over-]achieved. In this case, the quotas are still set, so qt_exceeds keeps increasing once per quota reset interval while the goal stays achieved. Clarify this behavior in the design document. Link: https://lore.kernel.org/20260914141952.91465-6-sj@kernel.org Signed-off-by: Liew Rui Yan Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Eva Kurchatova Cc: Jonathan Corbet Cc: Kunwu Chan Cc: Liam R. Howlett Cc: Lian Wang Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: "Zenghui Yu (Huawei)" --- Documentation/mm/damon/design.rst | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index cb116a82ff20b4..ad09416dfb21a4 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -696,7 +696,9 @@ There are two such tuning algorithms that users can select as they need. fast as possible, using maximum allowed quota, but only for a temporal short time. When the quota is under-achieved, this algorithm keeps tuning quota to a maximum allowed one. Once the quota is [over]-achieved, this sets the - quota zero. Useful for deterministic control required environments. + quota zero. Useful for deterministic control required environments. Note + that the zero quota is a valid quota, and therefore ``qt_exceeds`` :ref:`stat + ` will keep increasing in this case. The goal can be specified with five parameters, namely ``target_metric``, ``target_value``, ``current_value``, ``nid`` and ``path``. The auto-tuning From f1edcf20145355160ce43153aeac7315c844309c Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 14 Sep 2026 07:19:51 -0700 Subject: [PATCH 0698/1012] Docs/mm/damon/design: fix typos in temporal auto-tuning algorithm section Correct incorrect references of "quota" to "goal" in the temporal auto-tuning algorithm description and fix the square bracket formatting. Link: https://lore.kernel.org/20260914141952.91465-7-sj@kernel.org Signed-off-by: Liew Rui Yan Reviewed-by: SJ Park Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Eva Kurchatova Cc: Jonathan Corbet Cc: Kunwu Chan Cc: Liam R. Howlett Cc: Lian Wang Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: "Zenghui Yu (Huawei)" --- Documentation/mm/damon/design.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index ad09416dfb21a4..707170bcd1b33c 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -694,8 +694,8 @@ There are two such tuning algorithms that users can select as they need. This is the default selection. If unsure, use this. - ``temporal``: More straightforward algorithm. Tries to achieve the goal as fast as possible, using maximum allowed quota, but only for a temporal short - time. When the quota is under-achieved, this algorithm keeps tuning quota to - a maximum allowed one. Once the quota is [over]-achieved, this sets the + time. When the goal is under-achieved, this algorithm keeps tuning quota to + a maximum allowed one. Once the goal is [over-]achieved, this sets the quota zero. Useful for deterministic control required environments. Note that the zero quota is a valid quota, and therefore ``qt_exceeds`` :ref:`stat ` will keep increasing in this case. From 1a6d6fabffb3e290c4dfb70ac9cd41372883739c Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Sat, 12 Sep 2026 08:28:22 -0400 Subject: [PATCH 0699/1012] proc/task_mmu: handle special PMDs in clear_refs and pagemap mshv_vtl_low can install a special PMD from a PFN supplied in the mmap file offset. That PFN need not have a memmap entry. Reading /proc/PID/pagemap or writing /proc/PID/clear_refs for such a mapping oopses the kernel. With a test driver mapping the PFN at the 1 TiB mark: echo 1 > /proc/self/clear_refs BUG: unable to handle page fault for address: fffffb5b80000008 #PF: supervisor read access in kernel mode #PF: error_code(0x0000) - not-present page RIP: 0010:clear_refs_pte_range+0xb1/0x1e0 Call Trace: walk_pgd_range+0x50a/0xad0 __walk_page_range+0x6a/0x1d0 walk_page_range_mm_unsafe+0x193/0x230 clear_refs_write+0x18e/0x3f0 pread(pagemap_fd, buf, 512 * 8, addr / PAGE_SIZE * 8) BUG: unable to handle page fault for address: fffff57840000008 RIP: 0010:pagemap_pmd_range+0x3cf/0x6b0 Call Trace: walk_pgd_range+0x50a/0xad0 __walk_page_range+0x6a/0x1d0 walk_page_range_mm_unsafe+0x193/0x230 pagemap_read+0x1dc/0x350 clear_refs_pte_range() passes the PMD to pmd_folio(), while pagemap_pmd_range_thp() uses pmd_page() followed by page_folio(). Both paths assume the PFN has a struct page and dereference bad address. Use vm_normal_folio_pmd() and vm_normal_page_pmd() so clear_refs skips folio operations and pagemap skips folio-derived flags for special PMDs. This matches the PTE paths, which already use vm_normal_folio() and vm_normal_page(). User-facing notes: - Pagemap still reports PM_PRESENT and the PFN to a CAP_SYS_ADMIN reader. - For a special PMD backed by a valid memmap entry and for the huge zero PMD, PM_FILE changes from set to clear. - PM_MMAP_EXCLUSIVE is already clear in both cases. - The output for an anonymous THP is unchanged. Tested in QEMU with the test driver on broken and fixed kernels. Both proc operations oops before the fix and return 0 after it. This is technically reachable via mshv_vtl_low, so I tagged it stable. Link: https://lore.kernel.org/20260912122822.3348978-1-gourry@gourry.net Fixes: 3c8e44c9b369 ("mm: mark special bits for huge pfn mappings when inject") Signed-off-by: Gregory Price Signed-off-by: Andrew Morton Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260912034833.2952750-1-gourry%40gourry.net Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Jann Horn Cc: Jason Gunthorpe Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Pedro Falcato Cc: Peter Xu Cc: Vlastimil Babka Cc: # v6.19+ --- fs/proc/task_mmu.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 565e6446bd3127..052e8dc796bcf8 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -1704,7 +1704,9 @@ static int clear_refs_pte_range(pmd_t *pmd, unsigned long addr, if (!pmd_present(*pmd)) goto out; - folio = pmd_folio(*pmd); + folio = vm_normal_folio_pmd(vma, addr, *pmd); + if (!folio) + goto out; /* Clear accessed and referenced bits. */ pmdp_test_and_clear_young(vma, addr, pmd); @@ -2024,7 +2026,7 @@ static int pagemap_pmd_range_thp(pmd_t *pmdp, unsigned long addr, goto populate_pagemap; if (pmd_present(pmd)) { - page = pmd_page(pmd); + page = vm_normal_page_pmd(vma, addr, pmd); flags |= PM_PRESENT; if (pmd_soft_dirty(pmd)) From 1533d83157d2a095383ece2d1b272fa217d9687d Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Tue, 15 Sep 2026 10:43:30 +0800 Subject: [PATCH 0700/1012] mm/vmalloc: group xa_init with vbq field initializations Patch series "mm/vmalloc: minor cleanups", v2. Small cleanup series for mm/vmalloc.c, no functional changes: - Group xa_init() with the other vmap_block_queue field initializations in vmalloc_init(), instead of after the unrelated vfree_deferred setup. - Extract vmap_insert_free_area() helper to deduplicate the allocate- and-insert pattern that appeared both inside the loop body and after the loop in vmap_init_free_space(). - Extract show_busy_info() from vmalloc_info_show(), mirroring the existing show_purge_info() pattern, so the top-level show function only orchestrates the two data sources. This patch (of 3): Move xa_init() next to the other vbq field initializations instead of after the unrelated vfree_deferred setup. Link: https://lore.kernel.org/20260915-vmalloc_study-v2-0-cc4dfe635e22@linux.dev Link: https://lore.kernel.org/20260915-vmalloc_study-v2-1-cc4dfe635e22@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) --- mm/vmalloc.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index aed70e4f8e4e4e..ca4aa340fc3c71 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -5574,10 +5574,11 @@ void __init vmalloc_init(void) vbq = &per_cpu(vmap_block_queue, i); spin_lock_init(&vbq->lock); INIT_LIST_HEAD(&vbq->free); + xa_init(&vbq->vmap_blocks); + p = &per_cpu(vfree_deferred, i); init_llist_head(&p->list); INIT_WORK(&p->wq, delayed_vfree_work); - xa_init(&vbq->vmap_blocks); } /* From 944ea0d1884cadeb5b0b60136722acf23af28431 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Tue, 15 Sep 2026 10:43:31 +0800 Subject: [PATCH 0701/1012] mm/vmalloc: extract vmap_insert_free_area helper The allocation and insertion of a free vmap_area is duplicated between the loop body and the tail of vmap_init_free_space. Factor it into a small helper so the main function only deals with computing the free gaps between busy regions. Link: https://lore.kernel.org/20260915-vmalloc_study-v2-2-cc4dfe635e22@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Reviewed-by: Dev Jain --- mm/vmalloc.c | 41 ++++++++++++++++++----------------------- 1 file changed, 18 insertions(+), 23 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index ca4aa340fc3c71..f0a1cc07c337b5 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -5432,11 +5432,23 @@ module_init(proc_vmalloc_init); #endif +static void __init vmap_insert_free_area(unsigned long start, unsigned long end) +{ + struct vmap_area *free = kmem_cache_zalloc(vmap_area_cachep, GFP_NOWAIT); + + if (!WARN_ON_ONCE(!free)) { + free->va_start = start; + free->va_end = end; + insert_vmap_area_augment(free, NULL, + &free_vmap_area_root, + &free_vmap_area_list); + } +} + static void __init vmap_init_free_space(void) { unsigned long vmap_start = 1; const unsigned long vmap_end = ULONG_MAX; - struct vmap_area *free; struct vm_struct *busy; /* @@ -5446,32 +5458,15 @@ static void __init vmap_init_free_space(void) * |<--------------------------------->| */ for (busy = vmlist; busy; busy = busy->next) { - if ((unsigned long) busy->addr - vmap_start > 0) { - free = kmem_cache_zalloc(vmap_area_cachep, GFP_NOWAIT); - if (!WARN_ON_ONCE(!free)) { - free->va_start = vmap_start; - free->va_end = (unsigned long) busy->addr; - - insert_vmap_area_augment(free, NULL, - &free_vmap_area_root, - &free_vmap_area_list); - } - } + if ((unsigned long) busy->addr - vmap_start > 0) + vmap_insert_free_area(vmap_start, + (unsigned long) busy->addr); vmap_start = (unsigned long) busy->addr + busy->size; } - if (vmap_end - vmap_start > 0) { - free = kmem_cache_zalloc(vmap_area_cachep, GFP_NOWAIT); - if (!WARN_ON_ONCE(!free)) { - free->va_start = vmap_start; - free->va_end = vmap_end; - - insert_vmap_area_augment(free, NULL, - &free_vmap_area_root, - &free_vmap_area_list); - } - } + if (vmap_end - vmap_start > 0) + vmap_insert_free_area(vmap_start, vmap_end); } static void vmap_init_nodes(void) From aff77f5e396b4cbb523504a3bb4700d19c0102c8 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Tue, 15 Sep 2026 10:43:32 +0800 Subject: [PATCH 0702/1012] mm/vmalloc: extract show_busy_info from vmalloc_info_show Extract the busy vmap area iteration into show_busy_info, mirroring the existing show_purge_info pattern. Link: https://lore.kernel.org/20260915-vmalloc_study-v2-3-cc4dfe635e22@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) --- mm/vmalloc.c | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index f0a1cc07c337b5..2d29b08f263ad3 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -5344,7 +5344,7 @@ static void show_purge_info(struct seq_file *m) } } -static int vmalloc_info_show(struct seq_file *m, void *p) +static void show_busy_info(struct seq_file *m) { struct vmap_node *vn; struct vmap_area *va; @@ -5414,12 +5414,18 @@ static int vmalloc_info_show(struct seq_file *m, void *p) spin_unlock(&vn->busy.lock); } + if (IS_ENABLED(CONFIG_NUMA)) + kfree(counters); +} + +static int vmalloc_info_show(struct seq_file *m, void *p) +{ + show_busy_info(m); + /* * As a final step, dump "unpurged" areas. */ show_purge_info(m); - if (IS_ENABLED(CONFIG_NUMA)) - kfree(counters); return 0; } From 78e68bfc54b69f935e1b9d66a13e7bbd3ed0e1aa Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 12 Sep 2026 13:48:59 +0200 Subject: [PATCH 0703/1012] mm/secretmem: fix the enable parameter description The module parameter is enable, but its MODULE_PARM_DESC() names the variable behind it, secretmem_enable. secretmem is built in, so the description only reaches modules.builtin.modinfo, where it is filed under secretmem_enable while the parmtype line and /sys/module/secretmem/parameters/ say enable. Use the parameter name in the description. Link: https://lore.kernel.org/20260912114859.88957-1-kmehltretter@gmail.com Fixes: 1507f51255c9 ("mm: introduce memfd_secret system call to create "secret" memory areas") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Assisted-by: LLM --- mm/secretmem.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/secretmem.c b/mm/secretmem.c index 6cbb8efc994a4d..7287a2866897e4 100644 --- a/mm/secretmem.c +++ b/mm/secretmem.c @@ -39,7 +39,7 @@ static bool secretmem_enable __ro_after_init = 1; module_param_named(enable, secretmem_enable, bool, 0400); -MODULE_PARM_DESC(secretmem_enable, +MODULE_PARM_DESC(enable, "Enable secretmem and memfd_secret(2) system call"); static atomic_t secretmem_users; From 1e0bf803be75d4ade312638bab490eb1f8918c29 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Sat, 12 Sep 2026 07:08:32 -0400 Subject: [PATCH 0704/1012] mm/madvise: reclaim isolated folios if PTE restart fails MADV_PAGEOUT collects isolated folios on a local list before reclaiming them after the PTE walk. The reschedule path drops the PTE lock and then restarts the mapping with pte_offset_map_lock(). A concurrent operation can remove or replace the PTE table while the lock is dropped, causing pte_offset_map_lock() to return NULL. Returning directly in that case bypasses reclaim_pages(), leaving the collected folios off the LRU with elevated references. This results in a permanent memory leak. nr_isolated_anon increases reliably and does not decrease when the process dies. Route the failure through the existing cleanup path so any isolated folios are reclaimed or put back. Simplest userland pseudo-code reproducer: p = mmap(PMD_SIZE, ANONYMOUS); touch_every_page(p, PMD_SIZE); parallel { while (1) madvise(p, PMD_SIZE, MADV_PAGEOUT); while (1) { madvise(p, PMD_SIZE, MADV_DONTNEED); touch_every_page(p, PMD_SIZE); } } Reproduced in qemu trivially with some explicit widening of the race window. Link: https://lore.kernel.org/20260912110832.3203902-1-gourry@gourry.net Fixes: b2f557a21bc8 ("mm/madvise: add cond_resched() in madvise_cold_or_pageout_pte_range()") Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260821150912.183976-1-gourry@gourry.net Reviewed-by: Lorenzo Stoakes (ARM) Assisted-by: LLM Cc: David Hildenbrand Cc: Jann Horn Cc: Jiexun Wang Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: --- mm/madvise.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/madvise.c b/mm/madvise.c index 80ea991ce350a3..8cd08fc153e6d6 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -463,7 +463,7 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd, restart: start_pte = pte = pte_offset_map_lock(vma->vm_mm, pmd, addr, &ptl); if (!start_pte) - return 0; + goto out; flush_tlb_batched_pending(mm); lazy_mmu_mode_enable(); for (; addr < end; pte += nr, addr += nr * PAGE_SIZE) { @@ -567,6 +567,7 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd, folio_deactivate(folio); } +out: if (start_pte) { lazy_mmu_mode_disable(); pte_unmap_unlock(start_pte, ptl); From bd2e6a28a56a4bf324cc48157c247bc94d66d93d Mon Sep 17 00:00:00 2001 From: "Harry Yoo (Meta)" Date: Mon, 14 Sep 2026 21:36:39 +0100 Subject: [PATCH 0705/1012] selftests/cgroup: ignore memory.reclaim -EAGAIN for zswap writeback test MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The zswap_writeback_enabled test fails when a write to memory.reclaim returns -EAGAIN, which means less than the requested amount was reclaimed. attempt_writeback() propagates the -EAGAIN to the caller, and the test case is marked as failed even when zswap writeback did happen. This heavily depends on the performance of the backing swap device. Reclaim does not wait for writeback (on cgroup v2), does not count pages that are under writeback as reclaimed, and memory.reclaim gives up after MAX_RECLAIM_RETRIES passes without making progress. On a slow device where reclaim does not make any progress before writeback completes, a write to memory.reclaim fails. On a VM with zswap enabled, where IO delay was injected via dm-delay, the success rate of the zswap writeback test drops dramatically once the delay reaches 11 ms: 7% failures at 10 ms and 79% failures at 11 ms, n = 100. When zswap writeback is enabled, ignore -EAGAIN from memory.reclaim and determine pass/fail based on the zswpwb counter because that is what zswap_writeback_enabled actually wants to test. With this change, the test reliably passes even on a slow swap device (tested up to 1000 ms delay). This makes the test resilient against the performance of the swap device. Link: https://lore.kernel.org/20260914-test-zswap-wb-ignore-eagain-v1-1-6fb715c22cd8@kernel.org Fixes: 158863e5d7cc ("selftests: cgroup: add tests to verify the zswap writeback path") Signed-off-by: Harry Yoo (Meta) Signed-off-by: Andrew Morton Reviewed-by: SJ Park Acked-by: Nhat Pham Assisted-by: LLM Cc: Chengming Zhou Cc: Johannes Weiner Cc: Joshua Hahn Cc: Kiryl Shutsemau Cc: Michal Koutný Cc: Shuah Khan Cc: Tejun Heo Cc: Usama Arif --- tools/testing/selftests/cgroup/test_zswap.c | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/cgroup/test_zswap.c b/tools/testing/selftests/cgroup/test_zswap.c index 1ac77907277570..d50acc1b83ac83 100644 --- a/tools/testing/selftests/cgroup/test_zswap.c +++ b/tools/testing/selftests/cgroup/test_zswap.c @@ -346,7 +346,16 @@ static int attempt_writeback(const char *cgroup, void *arg) * it can't writeback to swap. */ ret = cg_write_numeric(cgroup, "memory.reclaim", memsize); - if (!wb_enabled) + + /* + * When writeback is enabled, memory.reclaim may still fail to reclaim + * the requested amount of memory due to a slow swap device. + * Ignore -EAGAIN here. The caller determines pass/fail based on the + * zswap writeback counter. + */ + if (wb_enabled && ret == -EAGAIN) + ret = 0; + else if (!wb_enabled) ret = (ret == -EAGAIN) ? 0 : -1; out: From 7dd01a31736382d6844d2cfab93a1dde2c3a0584 Mon Sep 17 00:00:00 2001 From: Dan Carpenter Date: Tue, 15 Sep 2026 19:37:27 +0300 Subject: [PATCH 0706/1012] mm/hmm/test: reject overflowing page counts The number of pages comes directly from userspace. Shifting a value larger than ULONG_MAX >> PAGE_SHIFT discards its high bits before the existing end-address check. A request for a huge number of pages can therefore be accepted and processed as a much smaller request. Reject page counts that cannot be represented as a byte size before performing the shift. So far as ChatGPT and I can tell this doesn't cause an issue in practice but preventing this integer overflow is the correct thing to do. Link: https://lore.kernel.org/6f0d39089e938d4f53dc72632eb84aca34de7015.1789457465.git.error27@gmail.com Fixes: b2ef9f5a5cb3 ("mm/hmm/test: add selftest driver for HMM") Signed-off-by: Dan Carpenter Signed-off-by: Andrew Morton Reviewed-by: Alistair Popple Assisted-by: ChatGPT:gpt-5 Cc: Jason Gunthorpe Cc: Jerome Glisse Cc: Leon Romanovsky Cc: Ralph Campbell Cc: Wei Yongjun --- lib/test_hmm.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/lib/test_hmm.c b/lib/test_hmm.c index 6911daa9f8543c..b409c42bfe6fc4 100644 --- a/lib/test_hmm.c +++ b/lib/test_hmm.c @@ -1624,6 +1624,8 @@ static long dmirror_fops_unlocked_ioctl(struct file *filp, if (cmd.addr & ~PAGE_MASK) return -EINVAL; + if (cmd.npages > ULONG_MAX >> PAGE_SHIFT) + return -EINVAL; if (cmd.addr >= (cmd.addr + (cmd.npages << PAGE_SHIFT))) return -EINVAL; From aeb3318e1818adeabb4286066c01a8d08583c3f5 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:50 -0700 Subject: [PATCH 0707/1012] mm/damon/api: introduce DAMON_FILTER_TYPE_HUGEPAGE_SIZE Patch series "mm/damon: introduce hugepage_size probe filter". Knowing whether a given memory is backed by a hugepage of specific size is useful for efficient utilization of hugepages. For easy monitoring of the information, introduce a new data attribute probe filter type, hugepage_size. It works similar to the DAMOS filter of the same name. It works for memory that is backed by a hugepage of a given size range. Patch 1 introduces the new probe filter type to DAMON API and extends related data structures. Patch 2 updates probe filter commit logic to handle the size range. Patch 3 Updates the filtering logic to support the new type. Patch 4 adds new DAMON sysfs files for the size range. Patch 5 updates DAMON sysfs interface to fully support the new filter type. Patches 6-8 updates design, usage and ABI documents for the new feature. This patch (of 8): Introduce a new data attribute probe filter type, hugepage_size. It will work for memory that is backed by a hugepage of a given size range. Add a new damon_filter_type enum DAMON_FILTER_TYPE_HUGEPAGE_SIZE to identify the type. Add two new fields in the damon_filter struct for saving the size range. Link: https://lore.kernel.org/20260915143359.91472-1-sj@kernel.org Link: https://lore.kernel.org/20260915143359.91472-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/include/linux/damon.h b/include/linux/damon.h index 4be7d1df8e71fa..bbb190b4740150 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -783,12 +783,14 @@ struct damon_prep { * @DAMON_FILTER_TYPE_MEMCG: Specific memcg's pages. * @DAMON_FILTER_TYPE_PGIDLE_UNSET: Pgidle is unset. * @DAMON_FILTER_TYPE_PGIDLE_SET: Pgidle is set. + * @DAMON_FILTER_TYPE_HUGEPAGE_SIZE: Page is part of a hugepage. */ enum damon_filter_type { DAMON_FILTER_TYPE_ANON, DAMON_FILTER_TYPE_MEMCG, DAMON_FILTER_TYPE_PGIDLE_UNSET, DAMON_FILTER_TYPE_PGIDLE_SET, + DAMON_FILTER_TYPE_HUGEPAGE_SIZE, }; /** @@ -798,6 +800,8 @@ enum damon_filter_type { * @matching: Whether this filter is for the type-matching ones. * @allow: Whether the @type-@matching ones should pass this filter. * @memcg_id: Memcg id of the question if @type is DAMON_FILTER_MEMCG. + * @range_min: Minimum value of range arguments. + * @range_max: Maximum value of range arguments. */ struct damon_filter { enum damon_filter_type type; @@ -805,6 +809,10 @@ struct damon_filter { bool allow; union { u64 memcg_id; + struct { + unsigned long range_min; + unsigned long range_max; + }; }; /* private: */ /* Siblings list. */ From 3d515d6a8dba4d9de7e4c40fbd60c29a09b733dd Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:51 -0700 Subject: [PATCH 0708/1012] mm/damon/core: commit hugepage_size type damon filter Extend data attribute probe filters commit logic for the new hugepage_size filter type. Since it needs to carry the size range of the hugepage, update the logic to update the size range fields of the commit destination filter struct. While doing that, validate the given range and propagate an error if it is invalid. Add the error handling in the callers, too. Link: https://lore.kernel.org/20260915143359.91472-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 28 +++++++++++++++++++++++----- 1 file changed, 23 insertions(+), 5 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 5fdacb9dcee5f6..e1b49c3d72b865 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -1818,7 +1818,7 @@ static int damon_commit_preps(struct damon_probe *dst, struct damon_probe *src) return 0; } -static void damon_commit_filter(struct damon_filter *dst, +static int damon_commit_filter(struct damon_filter *dst, struct damon_filter *src) { dst->type = src->type; @@ -1828,23 +1828,33 @@ static void damon_commit_filter(struct damon_filter *dst, case DAMON_FILTER_TYPE_MEMCG: dst->memcg_id = src->memcg_id; break; + case DAMON_FILTER_TYPE_HUGEPAGE_SIZE: + if (src->range_max < src->range_min) + return -EINVAL; + dst->range_min = src->range_min; + dst->range_max = src->range_max; + break; default: break; } + return 0; } static int damon_commit_filters(struct damon_probe *dst, struct damon_probe *src) { struct damon_filter *dst_filter, *next, *src_filter, *new_filter; - int i = 0, j = 0; + int i = 0, j = 0, err; damon_for_each_filter_safe(dst_filter, next, dst) { src_filter = damon_nth_filter(i++, src); - if (src_filter) - damon_commit_filter(dst_filter, src_filter); - else + if (src_filter) { + err = damon_commit_filter(dst_filter, src_filter); + if (err) + return err; + } else { damon_destroy_filter(dst_filter); + } } damon_for_each_filter_safe(src_filter, next, src) { @@ -1859,6 +1869,14 @@ static int damon_commit_filters(struct damon_probe *dst, case DAMON_FILTER_TYPE_MEMCG: new_filter->memcg_id = src_filter->memcg_id; break; + case DAMON_FILTER_TYPE_HUGEPAGE_SIZE: + if (src_filter->range_max < src_filter->range_min) { + damon_destroy_filter(new_filter); + return -EINVAL; + } + new_filter->range_min = src_filter->range_min; + new_filter->range_max = src_filter->range_max; + break; default: break; } From be954504dca1c7263cb78049d3b2b89120895936 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:52 -0700 Subject: [PATCH 0709/1012] mm/damon/ops-common: support hugepage_size damon filter matching Update ops-common data attribute filter matching logic to support hugepage_size filter type. Link: https://lore.kernel.org/20260915143359.91472-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/ops-common.c | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/mm/damon/ops-common.c b/mm/damon/ops-common.c index c36cc39cd2c707..77366f42b3e5bf 100644 --- a/mm/damon/ops-common.c +++ b/mm/damon/ops-common.c @@ -536,6 +536,7 @@ bool damon_ops_filter_match(struct damon_filter *filter, struct folio *folio) { bool matched = false; struct mem_cgroup *memcg; + size_t folio_sz; switch (filter->type) { case DAMON_FILTER_TYPE_ANON: @@ -558,6 +559,15 @@ bool damon_ops_filter_match(struct damon_filter *filter, struct folio *folio) matched = filter->memcg_id == mem_cgroup_id(memcg); rcu_read_unlock(); break; + case DAMON_FILTER_TYPE_HUGEPAGE_SIZE: + if (!folio) { + matched = false; + break; + } + folio_sz = folio_size(folio); + matched = filter->range_min <= folio_sz && + folio_sz <= filter->range_max; + break; default: break; } From eef5349965a17f94ef39ac6366ee3e71c9d14b15 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:53 -0700 Subject: [PATCH 0710/1012] mm/damon/sysfs: add min,max files under probe filter directory In future, DAMON sysfs interface will support data attribute probe filter types that have range arguments like the newly added hugepage_size type filter. To prepare such supports, add two new DAMON sysfs files, min and max, under the probe filter directory. Link: https://lore.kernel.org/20260915143359.91472-5-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 48 ++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 51fa506c879b02..8e8d89b8ed981e 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -975,6 +975,8 @@ struct damon_sysfs_filter { bool matching; bool allow; char *path; + unsigned long range_min; + unsigned long range_max; }; static struct damon_sysfs_filter *damon_sysfs_filter_alloc(void) @@ -1127,6 +1129,44 @@ static ssize_t path_store(struct kobject *kobj, return count; } +static ssize_t min_show(struct kobject *kobj, + struct kobj_attribute *attr, char *buf) +{ + struct damon_sysfs_filter *filter = container_of(kobj, + struct damon_sysfs_filter, kobj); + + return sysfs_emit(buf, "%lu\n", filter->range_min); +} + +static ssize_t min_store(struct kobject *kobj, + struct kobj_attribute *attr, const char *buf, size_t count) +{ + struct damon_sysfs_filter *filter = container_of(kobj, + struct damon_sysfs_filter, kobj); + int err = kstrtoul(buf, 0, &filter->range_min); + + return err ? err : count; +} + +static ssize_t max_show(struct kobject *kobj, + struct kobj_attribute *attr, char *buf) +{ + struct damon_sysfs_filter *filter = container_of(kobj, + struct damon_sysfs_filter, kobj); + + return sysfs_emit(buf, "%lu\n", filter->range_max); +} + +static ssize_t max_store(struct kobject *kobj, + struct kobj_attribute *attr, const char *buf, size_t count) +{ + struct damon_sysfs_filter *filter = container_of(kobj, + struct damon_sysfs_filter, kobj); + int err = kstrtoul(buf, 0, &filter->range_max); + + return err ? err : count; +} + static void damon_sysfs_filter_release(struct kobject *kobj) { struct damon_sysfs_filter *filter = container_of(kobj, @@ -1148,11 +1188,19 @@ static struct kobj_attribute damon_sysfs_filter_allow_attr = static struct kobj_attribute damon_sysfs_filter_path_attr = __ATTR_RW_MODE(path, 0600); +static struct kobj_attribute damon_sysfs_filter_min_attr = + __ATTR_RW_MODE(min, 0600); + +static struct kobj_attribute damon_sysfs_filter_max_attr = + __ATTR_RW_MODE(max, 0600); + static struct attribute *damon_sysfs_filter_attrs[] = { &damon_sysfs_filter_type_attr.attr, &damon_sysfs_filter_matching_attr.attr, &damon_sysfs_filter_allow_attr.attr, &damon_sysfs_filter_path_attr.attr, + &damon_sysfs_filter_min_attr.attr, + &damon_sysfs_filter_max_attr.attr, NULL, }; ATTRIBUTE_GROUPS(damon_sysfs_filter); From 707ae91fb36d33dfd6d87dc7884aedb6df3a0ac2 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:54 -0700 Subject: [PATCH 0711/1012] mm/damon/sysfs: support hugepage_size probe filter Extend DAMON sysfs interface to support hugepage_size probe filter. Allows hugepage_size user string input to the filter type file. Pass the size range argument that users set via min/max files under the probe filter directory to the DAMON core. Link: https://lore.kernel.org/20260915143359.91472-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/sysfs.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 8e8d89b8ed981e..43519afb9eb7f3 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -1007,6 +1007,10 @@ damon_sysfs_filter_type_names[] = { .type = DAMON_FILTER_TYPE_PGIDLE_SET, .name = "pgidle_set", }, + { + .type = DAMON_FILTER_TYPE_HUGEPAGE_SIZE, + .name = "hugepage_size", + }, }; static ssize_t type_show(struct kobject *kobj, @@ -2273,6 +2277,9 @@ static int damon_sysfs_set_filters(struct damon_probe *probe, damon_destroy_filter(filter); return err; } + } else if (filter->type == DAMON_FILTER_TYPE_HUGEPAGE_SIZE) { + filter->range_min = sys_filter->range_min; + filter->range_max = sys_filter->range_max; } damon_add_filter(probe, filter); } From fe9cd900b9ae28990bef0ff237d00792324e8694 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:55 -0700 Subject: [PATCH 0712/1012] Docs/mm/damon/design: update for hugepage_size probe filter Update DAMON design document for the newly added hugepage_size data attribute probe filter type. Link: https://lore.kernel.org/20260915143359.91472-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 2 ++ 1 file changed, 2 insertions(+) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 707170bcd1b33c..0a86792f90a184 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -300,6 +300,8 @@ filter types. Currently below filter types are supported. - ``pgidle_unset``: Matches if the page for the memory is marked as not access-idle. - ``pgidle_set``: Matches if the page for the memory is marked as access-idle. +- ``hugepage_size``: Matches if the page for the memory is a part of a hugepage + of a given size range. If such probes are registered, DAMON executes the probes for each region's sampling memory when it does the access :ref:`sampling From 8d5cf174963e1bbe6492976fbcf4c78dd3233cff Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:56 -0700 Subject: [PATCH 0713/1012] Docs/admin-guide/mm/damon/usage: update for hugepage_size Update DAMON usage document for the newly added hugepage_size data attribute filter type. Link: https://lore.kernel.org/20260915143359.91472-8-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/admin-guide/mm/damon/usage.rst | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/Documentation/admin-guide/mm/damon/usage.rst b/Documentation/admin-guide/mm/damon/usage.rst index d3e37400367bdd..6b80bce5d678de 100644 --- a/Documentation/admin-guide/mm/damon/usage.rst +++ b/Documentation/admin-guide/mm/damon/usage.rst @@ -78,7 +78,7 @@ comma (","). │ │ │ │ │ │ │ │ │ 0/prep_action │ │ │ │ │ │ │ │ │ ... │ │ │ │ │ │ │ │ filters/nr_filters - │ │ │ │ │ │ │ │ │ 0/type,matching,allow,path + │ │ │ │ │ │ │ │ │ 0/type,matching,allow,path,min,max │ │ │ │ │ │ │ │ │ ... │ │ │ │ │ │ │ ... │ │ │ │ │ :ref:`targets `/nr_targets @@ -308,6 +308,8 @@ Writing a number (``N``) to the file creates the number of child directories named ``0`` to ``N-1``. Each directory represents each filter and works in a way similar to that for :ref:`DAMOS filter `. When the filter ``type`` is ``memcg``, ``path`` file acts as ``memcg_path`` for :ref:`DAMOS +filter `. When the filter ``type`` is ``hugepage_size``, +``min`` and ``max`` files acts as files of the same names for :ref:`DAMOS filter `. .. _sysfs_targets: From a77e99ab87782744e9f604c3e85c9b47f0e56b1c Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Tue, 15 Sep 2026 16:33:09 -0700 Subject: [PATCH 0714/1012] docs-admin-guide-mm-damon-usage-update-for-hugepage_size-fix s/files acts/file acts/, per SJ Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: SJ Park Cc: Suren Baghdasaryan Cc: Vlastimil Babka Signed-off-by: Andrew Morton --- Documentation/admin-guide/mm/damon/usage.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Documentation/admin-guide/mm/damon/usage.rst b/Documentation/admin-guide/mm/damon/usage.rst index 6b80bce5d678de..2e191a3dff18a6 100644 --- a/Documentation/admin-guide/mm/damon/usage.rst +++ b/Documentation/admin-guide/mm/damon/usage.rst @@ -309,7 +309,7 @@ named ``0`` to ``N-1``. Each directory represents each filter and works in a way similar to that for :ref:`DAMOS filter `. When the filter ``type`` is ``memcg``, ``path`` file acts as ``memcg_path`` for :ref:`DAMOS filter `. When the filter ``type`` is ``hugepage_size``, -``min`` and ``max`` files acts as files of the same names for :ref:`DAMOS +``min`` and ``max`` file acts as files of the same names for :ref:`DAMOS filter `. .. _sysfs_targets: From b7da9528f676cfd7adba7379afed4d24d019f9bd Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Wed, 16 Sep 2026 21:18:41 -0700 Subject: [PATCH 0715/1012] docs-admin-guide-mm-damon-usage-update-for-hugepage_size-fix-fix fix the fix Cc: SJ Park Signed-off-by: Andrew Morton --- Documentation/admin-guide/mm/damon/usage.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/Documentation/admin-guide/mm/damon/usage.rst b/Documentation/admin-guide/mm/damon/usage.rst index 2e191a3dff18a6..ba47255448564b 100644 --- a/Documentation/admin-guide/mm/damon/usage.rst +++ b/Documentation/admin-guide/mm/damon/usage.rst @@ -309,7 +309,7 @@ named ``0`` to ``N-1``. Each directory represents each filter and works in a way similar to that for :ref:`DAMOS filter `. When the filter ``type`` is ``memcg``, ``path`` file acts as ``memcg_path`` for :ref:`DAMOS filter `. When the filter ``type`` is ``hugepage_size``, -``min`` and ``max`` file acts as files of the same names for :ref:`DAMOS +``min`` and ``max`` files act as files of the same names for :ref:`DAMOS filter `. .. _sysfs_targets: From 265908aecbe394e0fcaf6b40045e12fad79560fc Mon Sep 17 00:00:00 2001 From: SJ Park Date: Tue, 15 Sep 2026 07:33:57 -0700 Subject: [PATCH 0716/1012] Docs/ABI/damon: update for hugepage_size probe filter For the newly added hugepage_size data attribute probe filter, two new sysfs files are added for the size range. Update DAMON ABI document for the new files. Link: https://lore.kernel.org/20260915143359.91472-9-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/ABI/testing/sysfs-kernel-mm-damon | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/Documentation/ABI/testing/sysfs-kernel-mm-damon b/Documentation/ABI/testing/sysfs-kernel-mm-damon index ad21f58f3c9126..55df688ea596ff 100644 --- a/Documentation/ABI/testing/sysfs-kernel-mm-damon +++ b/Documentation/ABI/testing/sysfs-kernel-mm-damon @@ -206,6 +206,20 @@ Description: If 'memcg' is written to the 'type' file, writing to and reading from this file sets and gets the path to the memory cgroup of the interest. +What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/filters//min +Date: Sep 2026 +Contact: SJ Park +Description: If 'hugepage_size' is written to the 'type' file, writing to and + reading from this file sets and gets the minimum size of the + huge page of the interest. + +What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/filters//max +Date: Sep 2026 +Contact: SJ Park +Description: If 'hugepage_size' is written to the 'type' file, writing to and + reading from this file sets and gets the maximum size of the + huge page of the interest. + What: /sys/kernel/mm/damon/admin/kdamonds//contexts//monitoring_attrs/probes/

/filters//matching Date: May 2026 Contact: SJ Park From d450b11940e83bd3af5568e290df8fb58d6bbc15 Mon Sep 17 00:00:00 2001 From: Lance Yang Date: Thu, 17 Sep 2026 13:40:15 +0800 Subject: [PATCH 0717/1012] mm/huge_memory: simplify pgtable deposit detection Whether a PMD has a deposited PTE page table only depends on whether the architecture requires deposits or the VMA is anonymous. Implement this rule directly in vma_has_deposited_pgtable(), avoiding mistaking a non-anonymous raw PFN PMD for one with a deposit and attempting to withdraw a page table that was never deposited. There is no known in-tree workload that triggers this. Mostly a defensive/simplifying change. Link: https://lore.kernel.org/20260917054015.23553-1-lance.yang@linux.dev Signed-off-by: Lance Yang Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Acked-by: Zi Yan Reviewed-by: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Liam R. Howlett Cc: Ryan Roberts --- mm/huge_memory.c | 22 +++++----------------- 1 file changed, 5 insertions(+), 17 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index ec37a63b8a2ec7..d2e990da86b0d9 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -2515,25 +2515,13 @@ static struct folio *normal_or_softleaf_folio_pmd(struct vm_area_struct *vma, return pmd_to_softleaf_folio(pmdval); } -static bool has_deposited_pgtable(struct vm_area_struct *vma, pmd_t pmdval, - struct folio *folio) +static bool vma_has_deposited_pgtable(struct vm_area_struct *vma) { - /* Some architectures require unconditional depositing. */ - if (arch_needs_pgtable_deposit()) - return true; - - /* - * Huge zero always deposited except for DAX which handles itself, see - * set_huge_zero_folio(). - */ - if (is_huge_zero_pmd(pmdval)) - return !vma_is_dax(vma); - /* - * Otherwise, only anonymous folios are deposited, see - * __do_huge_pmd_anonymous_page(). + * PMDs in anonymous VMAs always have a deposited page table. PMDs in + * other VMAs only have one when required by the architecture. */ - return folio && folio_test_anon(folio); + return arch_needs_pgtable_deposit() || vma_is_anonymous(vma); } /** @@ -2573,7 +2561,7 @@ bool zap_huge_pmd(struct mmu_gather *tlb, struct vm_area_struct *vma, is_present = pmd_present(orig_pmd); folio = normal_or_softleaf_folio_pmd(vma, addr, orig_pmd, is_present); - has_deposit = has_deposited_pgtable(vma, orig_pmd, folio); + has_deposit = vma_has_deposited_pgtable(vma); if (folio) zap_huge_pmd_folio(mm, vma, orig_pmd, folio, is_present); if (has_deposit) From bfe990ef011ac0089fd4528089fcf1e839c28af8 Mon Sep 17 00:00:00 2001 From: Sebastian Andrzej Siewior Date: Fri, 18 Sep 2026 12:50:13 +0200 Subject: [PATCH 0718/1012] mm/vmalloc: use %p for pointer formatting Commit 45ec16908e84e ("mm: use %pK for /proc/vmallocinfo") introduced the %pK in order not to leak kernel pointers. Since commit ad67b74d2469d ("printk: hash addresses printed with %p") pointers are hashed by default and the behaviour can be controlled by `hash_pointers' boot argument. The policy on %p is to not introduce new ones. Removing %pK makes it possible to remove its handling from the library. Looking at the output, having the pointer in the output makes it possible to distinguish the individual entries. Use %p instead %pK. /proc/vmallocinfo has mode 0400, so this change doesn't make kernel pointer information available to unprivileged userspace. Link: https://lore.kernel.org/20260918105013.UpdykT6j@linutronix.de Signed-off-by: Sebastian Andrzej Siewior Signed-off-by: Andrew Morton Reviewed-by: SJ Park Reviewed-by: Uladzislau Rezki (Sony) --- mm/vmalloc.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 2d29b08f263ad3..fad918765f054e 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -5336,7 +5336,7 @@ static void show_purge_info(struct seq_file *m) for_each_vmap_node(vn) { spin_lock(&vn->lazy.lock); list_for_each_entry(va, &vn->lazy.head, list) { - seq_printf(m, "0x%pK-0x%pK %7ld unpurged vm_area\n", + seq_printf(m, "0x%p-0x%p %7ld unpurged vm_area\n", (void *)va->va_start, (void *)va->va_end, va_size(va)); } @@ -5359,7 +5359,7 @@ static void show_busy_info(struct seq_file *m) list_for_each_entry(va, &vn->busy.head, list) { if (!va->vm) { if (va->flags & VMAP_RAM) - seq_printf(m, "0x%pK-0x%pK %7ld vm_map_ram\n", + seq_printf(m, "0x%p-0x%p %7ld vm_map_ram\n", (void *)va->va_start, (void *)va->va_end, va_size(va)); @@ -5373,7 +5373,7 @@ static void show_busy_info(struct seq_file *m) /* Pair with smp_wmb() in clear_vm_uninitialized_flag() */ smp_rmb(); - seq_printf(m, "0x%pK-0x%pK %7ld", + seq_printf(m, "0x%p-0x%p %7ld", v->addr, v->addr + v->size, v->size); if (v->caller) From 3c0500d983a7a91d1077711160707b0b3d53ccf8 Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Tue, 15 Sep 2026 15:55:24 +0530 Subject: [PATCH 0719/1012] mm/gup_test: safely calculate GUP batch size __gup_test_ioctl() calculates the end of a GUP batch using: next = addr + nr * PAGE_SIZE; If nr is too large, it can cause next to overflow and wrap around. If it wraps, the next > end check is bypassed and a large value of nr is passed to the GUP call, even though the pages array was allocated according to gup->size. This can lead to out of bounds writes. Also, when fewer than PAGE_SIZE bytes remain, the calculated batch contains zero pages. The code still calls GUP functions with pages + i, which can point past the allocated array. Calculate nr by taking the minimum of the number of pages per call and the pages remaining in the address range. Reject zero sized and non page aligned gup->size values and zero nr_pages_per_call value. Link: https://lore.kernel.org/20260915102524.125758-1-sarthak.sharma@arm.com Fixes: 64c349f4ae78 ("mm: add infrastructure for get_user_pages_fast() benchmarking") Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Cc: Jason Gunthorpe Cc: John Hubbard Cc: Peter Xu --- mm/gup_test.c | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/mm/gup_test.c b/mm/gup_test.c index 185ba3bb8ed10b..ba74bf3f410486 100644 --- a/mm/gup_test.c +++ b/mm/gup_test.c @@ -116,7 +116,8 @@ static int __gup_test_ioctl(unsigned int cmd, bool needs_mmap_lock = cmd != GUP_FAST_BENCHMARK && cmd != PIN_FAST_BENCHMARK; - if (gup->addr > ULONG_MAX || gup->size > ULONG_MAX) + if (gup->addr > ULONG_MAX || gup->size > ULONG_MAX || !gup->size || + !gup->nr_pages_per_call || !PAGE_ALIGNED(gup->size)) return -EINVAL; if (check_add_overflow((unsigned long)gup->addr, (unsigned long)gup->size, &end)) @@ -139,11 +140,9 @@ static int __gup_test_ioctl(unsigned int cmd, if (nr != gup->nr_pages_per_call) break; + nr = min_t(unsigned long, nr, (end - addr) / PAGE_SIZE); + next = addr + nr * PAGE_SIZE; - if (next > end) { - next = end; - nr = (next - addr) / PAGE_SIZE; - } switch (cmd) { case GUP_FAST_BENCHMARK: From aaedd1378908839141078844af061f8ddfc705ca Mon Sep 17 00:00:00 2001 From: "Barry Song (Xiaomi)" Date: Tue, 15 Sep 2026 18:15:56 +0800 Subject: [PATCH 0720/1012] mm/mglru: restore accidentally removed seq < max_seq check Since commit 798c0330c2ca ("mm/mglru: rework aging feedback"), the following sanity check was accidentally removed: if (seq < max_seq) return 0; That means we can perform aging for any value less than or equal to max_gen_nr. This has been inconsistent with Documentation/admin-guide/mm/multigen_lru.rst, which states: Users can write the following command to ``lru_gen`` to create a new generation ``max_gen_nr+1``: ``+ memcg_id node_id max_gen_nr [can_swap [force_scan]]`` The correct semantics are that writing a value smaller than max_gen_nr should return 0, since the requested generation already exists. Link: https://lore.kernel.org/20260915101556.50467-1-baohua@kernel.org Fixes: 798c0330c2ca ("mm/mglru: rework aging feedback") Signed-off-by: Barry Song (Xiaomi) Signed-off-by: Andrew Morton Reported-by: Chuanhua Han Reviewed-by: Kairui Song Reviewed-by: Baolin Wang Cc: Axel Rasmussen Cc: Baoquan He Cc: David Hildenbrand Cc: Johannes Weiner Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Shakeel Butt Cc: Wei Xu Cc: Yuanchu Xie Cc: Yu Zhao --- mm/vmscan.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/mm/vmscan.c b/mm/vmscan.c index 836f50814ffae5..f2e641e9cf7de9 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -5840,6 +5840,9 @@ static int run_aging(struct lruvec *lruvec, unsigned long seq, { DEFINE_MAX_SEQ(lruvec); + if (seq < max_seq) + return 0; + if (seq > max_seq) return -EINVAL; From 44491231f31c99c4d7435b372a1cb152aa688db2 Mon Sep 17 00:00:00 2001 From: Zhenghui Hao Date: Tue, 15 Sep 2026 16:20:29 +0800 Subject: [PATCH 0721/1012] mm/hugetlb: fix misspelled parameter names in comment The comment above hugepages_setup() refers to "hugepagsz" and "default_hugepagsz", but the parameters parsed by this code are "hugepagesz" and "default_hugepagesz". Fix the spelling and also drop the redundant spaces so the comment reads normally. No functional change. Link: https://lore.kernel.org/tencent_034C6FC23D4817C40657E5F17F64E260A009@qq.com Signed-off-by: Zhenghui Hao Signed-off-by: Andrew Morton Acked-by: Muchun Song Reviewed-by: SJ Park Cc: David Hildenbrand Cc: Oscar Salvador --- mm/hugetlb.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 1b53ba991d360a..9d05fecf21326d 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -4330,9 +4330,9 @@ static __init void hugetlb_parse_params(void) /* * hugepages command line processing - * hugepages normally follows a valid hugepagsz or default_hugepagsz - * specification. If not, ignore the hugepages value. hugepages can also - * be the first huge page command line option in which case it implicitly + * hugepages normally follows a valid hugepagesz or default_hugepagesz + * specification. If not, ignore the hugepages value. hugepages can also + * be the first huge page command line option in which case it implicitly * specifies the number of huge pages for the default size. */ static int __init hugepages_setup(char *s) From 733b5e4cc84c9d5545678aac0db35cdb4d0925df Mon Sep 17 00:00:00 2001 From: Qiqi Liu Date: Tue, 15 Sep 2026 15:49:28 +0800 Subject: [PATCH 0722/1012] mm/page_alloc: apply per-task GFP context in bulk allocator alloc_pages_bulk_noprof() does not call current_gfp_context(), so per-task scoped allocation constraints (PF_MEMALLOC_NOIO, PF_MEMALLOC_NOFS, PF_MEMALLOC_PIN) are not applied on the bulk fast path. By ignoring PF_MEMALLOC_NOIO and PF_MEMALLOC_NOFS, the allocation can theoretically result in a deadlock. Ignoring PF_MEMALLOC_PIN also has consequences: without clearing __GFP_MOVABLE, prepare_alloc_pages() selects MIGRATE_MOVABLE for the PCP list, and a task with PF_MEMALLOC_PIN set receives movable pages from the bulk allocator. Once pinned, these pages can no longer be migrated but remain in MOVABLE pageblocks, violating the mobility contract. This can increase fragmentation and interfere with compaction or contiguous-memory allocations, eventually surfacing as higher allocation latency or allocation failures under memory pressure. Found via review of the bulk allocation tracepoint hooks [1]. Link: https://sashiko.dev/#/patchset/20260907120949.418450-1-liuqiqi%40kylinos.cn Link: https://lore.kernel.org/all/20260907120949.418450-1-liuqiqi@kylinos.cn/ [1] Link: https://lore.kernel.org/20260915074928.327471-1-liuqiqi@kylinos.cn Fixes: 387ba26fb1cb ("mm/page_alloc: add a bulk page allocator") Signed-off-by: Qiqi Liu Signed-off-by: Andrew Morton Reviewed-by: Vlastimil Babka (SUSE) Cc: Brendan Jackman Cc: Johannes Weiner Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Zi Yan --- mm/page_alloc.c | 1 + 1 file changed, 1 insertion(+) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 306e3d34adf45a..25f0bdf26ce839 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -5228,6 +5228,7 @@ unsigned long alloc_pages_bulk_noprof(gfp_t gfp, int preferred_nid, /* May set ALLOC_NOFRAGMENT, fragmentation will return 1 page. */ gfp &= gfp_allowed_mask; + gfp = current_gfp_context(gfp); if (!prepare_alloc_pages(gfp, 0, preferred_nid, nodemask, &ac, &gfp, &alloc_flags)) goto out; From c780dd3e1dd8306e297a7011790297e322786206 Mon Sep 17 00:00:00 2001 From: Gregory Price Date: Sat, 12 Sep 2026 07:05:40 -0400 Subject: [PATCH 0723/1012] mm/madvise: use folio_trylock() in the cold/pageout PMD split MADV_COLD or MADV_PAGEOUT over part of a PMD splits the THP in madvise_cold_or_pageout_pte_range(). Two threads doing that to the same THP create spurious failures. CPU0 CPU1 ---- ---- folio_get() spin_unlock(ptl) folio_lock() folio_get() spin_unlock(ptl) folio_lock() <- blocks, keeps its ref split_folio() folio_expected_ref_count(folio) != folio_ref_count(folio) - 1 -EAGAIN CPU1 cannot drop its reference until it gets the lock CPU0 holds, so CPU0's split always fails. folio_trylock() makes CPU1 leave without ever taking a reference. The PTE branch of this same function already does this, as do madvise_free_pte_range() and madvise_free_huge_pmd(). Reproducer: 400 rounds of eight threads calling MADV_COLD on half of each of eight THPs, re-formed with MADV_COLLAPSE between rounds. From /proc/vmstat: thp_split_page thp_split_page_failed before 3186 860 after 3200 0 The short before count is rounds where every thread failed and the advice was dropped for that THP entirely. On failure the walker returns 0 and nothing retries. The PMD path becomes best effort when the folio lock is held elsewhere - same as the PTE path. Link: https://lore.kernel.org/20260912110540.3203010-1-gourry@gourry.net Signed-off-by: Gregory Price (Meta) Signed-off-by: Andrew Morton Reported-by: sashiko-bot Closes: https://sashiko.dev/#/patchset/20260817220810.1175596-1-gourry%40gourry.net Acked-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Assisted-by: LLM Cc: Jann Horn Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: --- mm/madvise.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/madvise.c b/mm/madvise.c index 8cd08fc153e6d6..32a28b9bb6880a 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -418,9 +418,10 @@ static int madvise_cold_or_pageout_pte_range(pmd_t *pmd, if (next - addr != HPAGE_PMD_SIZE) { int err; + if (!folio_trylock(folio)) + goto huge_unlock; folio_get(folio); spin_unlock(ptl); - folio_lock(folio); err = split_folio(folio); folio_unlock(folio); folio_put(folio); From a92a8687e588368470abc2b10b3560406cc4f387 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:05 -0400 Subject: [PATCH 0724/1012] mm: memcontrol: take a const folio in folio_memcg() and friends Patch series "mm: memcontrol: constify the read side of the read side of the memcg API", v3. The memcg accessors, lruvec helpers, and stat readers only read from the memcg, folio, or lruvec they are given, but take non-const pointers. Constify them, along with the page_counter readers and the swap I/O blkg helpers along the way. This patch (of 11): The folio_memcg() family only reads from the folio, and everything it calls already takes a const folio. Constify it, along with the page wrappers built on top of it and get_obj_cgroup_from_folio(), which only reads the folio's objcg. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-0-c239a6010b58@columbia.edu Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-1-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 50 +++++++++++++++++++------------------- mm/memcontrol.c | 8 +++--- 2 files changed, 29 insertions(+), 29 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 46bf724cae7af9..5a5ca6814782a3 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -410,7 +410,7 @@ static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) * or NULL. This function assumes that the folio is known to have a * proper object cgroup pointer. */ -static inline struct obj_cgroup *folio_objcg(struct folio *folio) +static inline struct obj_cgroup *folio_objcg(const struct folio *folio) { unsigned long memcg_data = folio->memcg_data; @@ -448,7 +448,7 @@ static inline struct obj_cgroup *folio_objcg(struct folio *folio) * Note: The caller should hold an rcu read lock or cgroup_mutex to protect * memcg associated with a folio from being released. */ -static inline struct mem_cgroup *folio_memcg(struct folio *folio) +static inline struct mem_cgroup *folio_memcg(const struct folio *folio) { struct obj_cgroup *objcg = folio_objcg(folio); @@ -461,7 +461,7 @@ static inline struct mem_cgroup *folio_memcg(struct folio *folio) * * Returns true if folio is charged to a memory cgroup, otherwise returns false. */ -static inline bool folio_memcg_charged(struct folio *folio) +static inline bool folio_memcg_charged(const struct folio *folio) { return folio->memcg_data != 0; } @@ -481,7 +481,7 @@ static inline bool folio_memcg_charged(struct folio *folio) * A caller should hold an rcu read lock to protect memcg associated with a * page from being released. */ -static inline struct mem_cgroup *folio_memcg_check(struct folio *folio) +static inline struct mem_cgroup *folio_memcg_check(const struct folio *folio) { /* * Because folio->memcg_data might be changed asynchronously @@ -498,11 +498,11 @@ static inline struct mem_cgroup *folio_memcg_check(struct folio *folio) return obj_cgroup_memcg(objcg); } -static inline struct mem_cgroup *page_memcg_check(struct page *page) +static inline struct mem_cgroup *page_memcg_check(const struct page *page) { if (PageTail(page)) return NULL; - return folio_memcg_check((struct folio *)page); + return folio_memcg_check((const struct folio *)page); } static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *objcg) @@ -527,14 +527,14 @@ static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *ob * that the folio has an associated memory cgroup. It's not safe to call * this function against some types of folios, e.g. slab folios. */ -static inline bool folio_memcg_kmem(struct folio *folio) +static inline bool folio_memcg_kmem(const struct folio *folio) { VM_BUG_ON_PGFLAGS(PageTail(&folio->page), &folio->page); VM_BUG_ON_FOLIO(folio->memcg_data & MEMCG_DATA_OBJEXTS, folio); return folio->memcg_data & MEMCG_DATA_KMEM; } -static inline bool PageMemcgKmem(struct page *page) +static inline bool PageMemcgKmem(const struct page *page) { return folio_memcg_kmem(page_folio(page)); } @@ -777,7 +777,7 @@ struct mem_cgroup *get_mem_cgroup_from_mm(struct mm_struct *mm); struct mem_cgroup *get_mem_cgroup_from_current(void); -struct mem_cgroup *get_mem_cgroup_from_folio(struct folio *folio); +struct mem_cgroup *get_mem_cgroup_from_folio(const struct folio *folio); struct lruvec *folio_lruvec_lock(struct folio *folio); struct lruvec *folio_lruvec_lock_irq(struct folio *folio); @@ -905,8 +905,8 @@ static inline bool mm_match_cgroup(struct mm_struct *mm, return match; } -struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(struct folio *folio); -ino_t page_cgroup_ino(struct page *page); +struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(const struct folio *folio); +ino_t page_cgroup_ino(const struct page *page); static inline bool mem_cgroup_online(struct mem_cgroup *memcg) { @@ -956,7 +956,7 @@ void mem_cgroup_print_oom_group(struct mem_cgroup *memcg); void mod_memcg_state(struct mem_cgroup *memcg, enum memcg_stat_item idx, int val); -static inline void mod_memcg_page_state(struct page *page, +static inline void mod_memcg_page_state(const struct page *page, enum memcg_stat_item idx, int val) { struct mem_cgroup *memcg; @@ -990,7 +990,7 @@ void mod_lruvec_kmem_state(void *p, enum node_stat_item idx, int val); void count_memcg_events(struct mem_cgroup *memcg, enum vm_event_item idx, unsigned long count); -static inline void count_memcg_folio_events(struct folio *folio, +static inline void count_memcg_folio_events(const struct folio *folio, enum vm_event_item idx, unsigned long nr) { struct mem_cgroup *memcg; @@ -1084,22 +1084,22 @@ static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) #define root_mem_cgroup (NULL) -static inline struct mem_cgroup *folio_memcg(struct folio *folio) +static inline struct mem_cgroup *folio_memcg(const struct folio *folio) { return NULL; } -static inline bool folio_memcg_charged(struct folio *folio) +static inline bool folio_memcg_charged(const struct folio *folio) { return false; } -static inline struct mem_cgroup *folio_memcg_check(struct folio *folio) +static inline struct mem_cgroup *folio_memcg_check(const struct folio *folio) { return NULL; } -static inline struct mem_cgroup *page_memcg_check(struct page *page) +static inline struct mem_cgroup *page_memcg_check(const struct page *page) { return NULL; } @@ -1109,12 +1109,12 @@ static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *ob return NULL; } -static inline bool folio_memcg_kmem(struct folio *folio) +static inline bool folio_memcg_kmem(const struct folio *folio) { return false; } -static inline bool PageMemcgKmem(struct page *page) +static inline bool PageMemcgKmem(const struct page *page) { return false; } @@ -1248,7 +1248,7 @@ static inline struct mem_cgroup *get_mem_cgroup_from_current(void) return NULL; } -static inline struct mem_cgroup *get_mem_cgroup_from_folio(struct folio *folio) +static inline struct mem_cgroup *get_mem_cgroup_from_folio(const struct folio *folio) { return NULL; } @@ -1406,7 +1406,7 @@ static inline void mod_memcg_state(struct mem_cgroup *memcg, { } -static inline void mod_memcg_page_state(struct page *page, +static inline void mod_memcg_page_state(const struct page *page, enum memcg_stat_item idx, int val) { } @@ -1471,7 +1471,7 @@ static inline void count_memcg_events(struct mem_cgroup *memcg, { } -static inline void count_memcg_folio_events(struct folio *folio, +static inline void count_memcg_folio_events(const struct folio *folio, enum vm_event_item idx, unsigned long nr) { } @@ -1767,7 +1767,7 @@ void __memcg_kmem_uncharge_page(struct page *page, int order); * needs to be used outside of the local scope. */ struct obj_cgroup *current_obj_cgroup(void); -struct obj_cgroup *get_obj_cgroup_from_folio(struct folio *folio); +struct obj_cgroup *get_obj_cgroup_from_folio(const struct folio *folio); static inline struct obj_cgroup *get_obj_cgroup_from_current(void) { @@ -1870,7 +1870,7 @@ static inline void __memcg_kmem_uncharge_page(struct page *page, int order) { } -static inline struct obj_cgroup *get_obj_cgroup_from_folio(struct folio *folio) +static inline struct obj_cgroup *get_obj_cgroup_from_folio(const struct folio *folio) { return NULL; } @@ -1901,7 +1901,7 @@ static inline void count_objcg_events(struct obj_cgroup *objcg, { } -static inline ino_t page_cgroup_ino(struct page *page) +static inline ino_t page_cgroup_ino(const struct page *page) { return 0; } diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 791e536efaebe8..72c0e9358cfce2 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -348,7 +348,7 @@ EXPORT_SYMBOL(memcg_bpf_enabled_key); * If memcg is bound to a traditional hierarchy, the css of root_mem_cgroup * is returned. */ -struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(struct folio *folio) +struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(const struct folio *folio) { struct mem_cgroup *memcg; @@ -373,7 +373,7 @@ struct cgroup_subsys_state *get_mem_cgroup_css_from_folio(struct folio *folio) * after page_cgroup_ino() returns, so it only should be used by callers that * do not care (such as procfs interfaces). */ -ino_t page_cgroup_ino(struct page *page) +ino_t page_cgroup_ino(const struct page *page) { struct mem_cgroup *memcg; unsigned long ino = 0; @@ -1256,7 +1256,7 @@ struct mem_cgroup *get_mem_cgroup_from_current(void) * * See folio_memcg() for folio->objcg/memcg binding rules. */ -struct mem_cgroup *get_mem_cgroup_from_folio(struct folio *folio) +struct mem_cgroup *get_mem_cgroup_from_folio(const struct folio *folio) { struct mem_cgroup *memcg; @@ -3154,7 +3154,7 @@ __always_inline struct obj_cgroup *current_obj_cgroup(void) return rcu_dereference_check(root_mem_cgroup->nodeinfo[nid]->objcg, 1); } -struct obj_cgroup *get_obj_cgroup_from_folio(struct folio *folio) +struct obj_cgroup *get_obj_cgroup_from_folio(const struct folio *folio) { struct obj_cgroup *objcg; From 7e1a69441942b3bac818395bcdce129c70ecbf1f Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:06 -0400 Subject: [PATCH 0725/1012] mm: memcontrol: constify obj_cgroup_memcg() and friends obj_cgroup_memcg() and get_mem_cgroup_from_objcg() only read from the objcg. Constify them. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-2-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 5a5ca6814782a3..65cb45f4be3204 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -396,7 +396,7 @@ enum objext_flags { * * The caller must ensure that the returned memcg won't be released. */ -static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) +static inline struct mem_cgroup *obj_cgroup_memcg(const struct obj_cgroup *objcg) { lockdep_assert_once(rcu_read_lock_held() || lockdep_is_held(&cgroup_mutex)); return objcg ? READ_ONCE(objcg->memcg) : NULL; @@ -505,7 +505,7 @@ static inline struct mem_cgroup *page_memcg_check(const struct page *page) return folio_memcg_check((const struct folio *)page); } -static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *objcg) +static inline struct mem_cgroup *get_mem_cgroup_from_objcg(const struct obj_cgroup *objcg) { struct mem_cgroup *memcg; @@ -1075,7 +1075,7 @@ void mem_cgroup_flush_workqueue(void); extern int mem_cgroup_init(void); #else /* CONFIG_MEMCG */ -static inline struct mem_cgroup *obj_cgroup_memcg(struct obj_cgroup *objcg) +static inline struct mem_cgroup *obj_cgroup_memcg(const struct obj_cgroup *objcg) { return NULL; } @@ -1104,7 +1104,7 @@ static inline struct mem_cgroup *page_memcg_check(const struct page *page) return NULL; } -static inline struct mem_cgroup *get_mem_cgroup_from_objcg(struct obj_cgroup *objcg) +static inline struct mem_cgroup *get_mem_cgroup_from_objcg(const struct obj_cgroup *objcg) { return NULL; } From 977ffa9d367da481a82c150b47b302393757924a Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:07 -0400 Subject: [PATCH 0726/1012] mm: memcontrol: constify the lruvec helpers The lruvec lookup helpers only read from the memcg, folio, or lruvec they are given. Constify them, along with lruvec_pgdat() and the folio argument of the folio_lruvec_relock_irq() helpers. mem_cgroup_lruvec() may update lruvec->pgdat for a newly onlined node, but that lives in the per-node structure, not the memcg. Similarly, lruvec_pgdat() still returns a non-const pgdat and does not use container_of_const(). Use container_of_const() in lruvec_memcg() and mem_cgroup_get_zone_lru_size(), so the const isn't silently cast away. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-3-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 46 +++++++++++++++++++------------------- include/linux/mmzone.h | 2 +- mm/memcontrol.c | 6 ++--- 3 files changed, 27 insertions(+), 27 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 65cb45f4be3204..436934030d796b 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -722,7 +722,7 @@ void mem_cgroup_migrate(struct folio *old, struct folio *new); * @pgdat combination. This can be the node lruvec, if the memory * controller is disabled. */ -static inline struct lruvec *mem_cgroup_lruvec(struct mem_cgroup *memcg, +static inline struct lruvec *mem_cgroup_lruvec(const struct mem_cgroup *memcg, struct pglist_data *pgdat) { struct mem_cgroup_per_node *mz; @@ -763,7 +763,7 @@ static inline struct lruvec *mem_cgroup_lruvec(struct mem_cgroup *memcg, * their binding is stable if the returned lruvec matches the one the caller has * locked. Useful for lock batching. */ -static inline struct lruvec *folio_lruvec(struct folio *folio) +static inline struct lruvec *folio_lruvec(const struct folio *folio) { struct mem_cgroup *memcg = folio_memcg(folio); @@ -779,9 +779,9 @@ struct mem_cgroup *get_mem_cgroup_from_current(void); struct mem_cgroup *get_mem_cgroup_from_folio(const struct folio *folio); -struct lruvec *folio_lruvec_lock(struct folio *folio); -struct lruvec *folio_lruvec_lock_irq(struct folio *folio); -struct lruvec *folio_lruvec_lock_irqsave(struct folio *folio, +struct lruvec *folio_lruvec_lock(const struct folio *folio); +struct lruvec *folio_lruvec_lock_irq(const struct folio *folio); +struct lruvec *folio_lruvec_lock_irqsave(const struct folio *folio, unsigned long *flags); static inline @@ -861,14 +861,14 @@ static inline struct mem_cgroup *mem_cgroup_from_seq(struct seq_file *m) return mem_cgroup_from_css(seq_css(m)); } -static inline struct mem_cgroup *lruvec_memcg(struct lruvec *lruvec) +static inline struct mem_cgroup *lruvec_memcg(const struct lruvec *lruvec) { - struct mem_cgroup_per_node *mz; + const struct mem_cgroup_per_node *mz; if (mem_cgroup_disabled()) return NULL; - mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec); + mz = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec); return mz->memcg; } @@ -919,13 +919,13 @@ void mem_cgroup_update_lru_size(struct lruvec *lruvec, enum lru_list lru, int zid, long nr_pages); static inline -unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec, +unsigned long mem_cgroup_get_zone_lru_size(const struct lruvec *lruvec, enum lru_list lru, int zone_idx) { long val; - struct mem_cgroup_per_node *mz; + const struct mem_cgroup_per_node *mz; - mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec); + mz = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec); val = READ_ONCE(mz->lru_zone_size[zone_idx][lru]); if (WARN_ON_ONCE(val < 0)) return 0; @@ -1215,13 +1215,13 @@ static inline void mem_cgroup_migrate(struct folio *old, struct folio *new) { } -static inline struct lruvec *mem_cgroup_lruvec(struct mem_cgroup *memcg, +static inline struct lruvec *mem_cgroup_lruvec(const struct mem_cgroup *memcg, struct pglist_data *pgdat) { return &pgdat->__lruvec; } -static inline struct lruvec *folio_lruvec(struct folio *folio) +static inline struct lruvec *folio_lruvec(const struct folio *folio) { struct pglist_data *pgdat = folio_pgdat(folio); return &pgdat->__lruvec; @@ -1281,7 +1281,7 @@ static inline void mem_cgroup_put(struct mem_cgroup *memcg) { } -static inline struct lruvec *folio_lruvec_lock(struct folio *folio) +static inline struct lruvec *folio_lruvec_lock(const struct folio *folio) { struct pglist_data *pgdat = folio_pgdat(folio); @@ -1290,7 +1290,7 @@ static inline struct lruvec *folio_lruvec_lock(struct folio *folio) return &pgdat->__lruvec; } -static inline struct lruvec *folio_lruvec_lock_irq(struct folio *folio) +static inline struct lruvec *folio_lruvec_lock_irq(const struct folio *folio) { struct pglist_data *pgdat = folio_pgdat(folio); @@ -1299,7 +1299,7 @@ static inline struct lruvec *folio_lruvec_lock_irq(struct folio *folio) return &pgdat->__lruvec; } -static inline struct lruvec *folio_lruvec_lock_irqsave(struct folio *folio, +static inline struct lruvec *folio_lruvec_lock_irqsave(const struct folio *folio, unsigned long *flagsp) { struct pglist_data *pgdat = folio_pgdat(folio); @@ -1354,7 +1354,7 @@ static inline struct mem_cgroup *mem_cgroup_from_seq(struct seq_file *m) return NULL; } -static inline struct mem_cgroup *lruvec_memcg(struct lruvec *lruvec) +static inline struct mem_cgroup *lruvec_memcg(const struct lruvec *lruvec) { return NULL; } @@ -1365,7 +1365,7 @@ static inline bool mem_cgroup_online(struct mem_cgroup *memcg) } static inline -unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec, +unsigned long mem_cgroup_get_zone_lru_size(const struct lruvec *lruvec, enum lru_list lru, int zone_idx) { return 0; @@ -1505,7 +1505,7 @@ static inline void mem_cgroup_flush_workqueue(void) { } static inline int mem_cgroup_init(void) { return 0; } #endif /* CONFIG_MEMCG */ -static inline struct lruvec *parent_lruvec(struct lruvec *lruvec) +static inline struct lruvec *parent_lruvec(const struct lruvec *lruvec) { struct mem_cgroup *memcg; @@ -1568,15 +1568,15 @@ static inline void lruvec_unlock_irqrestore(struct lruvec *lruvec, unsigned long } /* Test requires a stable folio->memcg binding, see folio_memcg() */ -static inline bool folio_matches_lruvec(struct folio *folio, - struct lruvec *lruvec) +static inline bool folio_matches_lruvec(const struct folio *folio, + const struct lruvec *lruvec) { return lruvec_pgdat(lruvec) == folio_pgdat(folio) && lruvec_memcg(lruvec) == folio_memcg(folio); } /* Don't lock again iff page's lruvec locked */ -static inline struct lruvec *folio_lruvec_relock_irq(struct folio *folio, +static inline struct lruvec *folio_lruvec_relock_irq(const struct folio *folio, struct lruvec *locked_lruvec) { if (locked_lruvec) { @@ -1590,7 +1590,7 @@ static inline struct lruvec *folio_lruvec_relock_irq(struct folio *folio, } /* Don't lock again iff folio's lruvec locked */ -static inline void folio_lruvec_relock_irqsave(struct folio *folio, +static inline void folio_lruvec_relock_irqsave(const struct folio *folio, struct lruvec **lruvecp, unsigned long *flags) { if (*lruvecp) { diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 65de3bb13eb3ae..8e4e0bda3b586b 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -1638,7 +1638,7 @@ extern void init_currently_empty_zone(struct zone *zone, unsigned long start_pfn extern void lruvec_init(struct lruvec *lruvec); -static inline struct pglist_data *lruvec_pgdat(struct lruvec *lruvec) +static inline struct pglist_data *lruvec_pgdat(const struct lruvec *lruvec) { #ifdef CONFIG_MEMCG return lruvec->pgdat; diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 72c0e9358cfce2..294827ef54a54c 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -1477,7 +1477,7 @@ void mem_cgroup_scan_tasks(struct mem_cgroup *memcg, * * Return: The lruvec this folio is on with its lock held and rcu read lock held. */ -struct lruvec *folio_lruvec_lock(struct folio *folio) +struct lruvec *folio_lruvec_lock(const struct folio *folio) { struct lruvec *lruvec; @@ -1505,7 +1505,7 @@ struct lruvec *folio_lruvec_lock(struct folio *folio) * Return: The lruvec this folio is on with its lock held and interrupts * disabled and rcu read lock held. */ -struct lruvec *folio_lruvec_lock_irq(struct folio *folio) +struct lruvec *folio_lruvec_lock_irq(const struct folio *folio) { struct lruvec *lruvec; @@ -1534,7 +1534,7 @@ struct lruvec *folio_lruvec_lock_irq(struct folio *folio) * Return: The lruvec this folio is on with its lock held and interrupts * disabled and rcu read lock held. */ -struct lruvec *folio_lruvec_lock_irqsave(struct folio *folio, +struct lruvec *folio_lruvec_lock_irqsave(const struct folio *folio, unsigned long *flags) { struct lruvec *lruvec; From 2766b799ca510b01034a41602fb9c5dd0cd34190 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:08 -0400 Subject: [PATCH 0727/1012] mm/page_io: take a const folio in bio_associate_blkg_from_folio() bio_associate_blkg_from_folio() and its helpers only read from the folio. Constify their folio arguments. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-4-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- mm/page_io.c | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index 1da4ff484f0971..5f7756e370f7a4 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -256,12 +256,13 @@ int swap_writeout(struct swap_io_ctx *ctx, struct folio *folio) } #if defined(CONFIG_MEMCG) && defined(CONFIG_BLK_CGROUP) -static struct cgroup_subsys_state *folio_memcg_blkg_css(struct folio *folio) +static struct cgroup_subsys_state *folio_memcg_blkg_css(const struct folio *folio) { return cgroup_e_css(folio_memcg(folio)->css.cgroup, &io_cgrp_subsys); } -static bool folio_blkg_can_merge(struct folio *folio, struct folio *prev_folio) +static bool folio_blkg_can_merge(const struct folio *folio, + const struct folio *prev_folio) { bool can_merge = true; @@ -277,7 +278,8 @@ static bool folio_blkg_can_merge(struct folio *folio, struct folio *prev_folio) return can_merge; } -static void bio_associate_blkg_from_folio(struct bio *bio, struct folio *folio) +static void bio_associate_blkg_from_folio(struct bio *bio, + const struct folio *folio) { struct cgroup_subsys_state *css; @@ -294,11 +296,13 @@ static void bio_associate_blkg_from_folio(struct bio *bio, struct folio *folio) css_put(css); } #else -static bool folio_blkg_can_merge(struct folio *folio, struct folio *prev_folio) +static bool folio_blkg_can_merge(const struct folio *folio, + const struct folio *prev_folio) { return true; } -static void bio_associate_blkg_from_folio(struct bio *bio, struct folio *folio) +static void bio_associate_blkg_from_folio(struct bio *bio, + const struct folio *folio) { } #endif /* CONFIG_MEMCG && CONFIG_BLK_CGROUP */ From 10c3b489502a271300670fd6cb129cc3d643709d Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:09 -0400 Subject: [PATCH 0728/1012] mm: memcontrol: constify the mem_cgroup accessors mem_cgroup_id(), parent_mem_cgroup(), mem_cgroup_is_root(), mem_cgroup_is_descendant(), memcg_kmem_id(), and the other memcg accessors only read from the memcg. Constify them, along with mem_cgroup_shrink_is_root()'s shrink_control. mem_cgroup_print_oom_context()'s task stays non-const, as task_cgroup() takes a non-const task. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-5-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 41 +++++++++++++++++++------------------- mm/memcontrol.c | 3 ++- 2 files changed, 23 insertions(+), 21 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 436934030d796b..01f74413fc79e9 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -539,7 +539,7 @@ static inline bool PageMemcgKmem(const struct page *page) return folio_memcg_kmem(page_folio(page)); } -static inline bool mem_cgroup_is_root(struct mem_cgroup *memcg) +static inline bool mem_cgroup_is_root(const struct mem_cgroup *memcg) { return (memcg == root_mem_cgroup); } @@ -555,7 +555,7 @@ static inline bool mem_cgroup_is_root(struct mem_cgroup *memcg) * and do not honour sc->memcg can use this to early-return 0 in per-memcg * contexts. */ -static inline bool mem_cgroup_shrink_is_root(struct shrink_control *sc) +static inline bool mem_cgroup_shrink_is_root(const struct shrink_control *sc) { return !sc->memcg || mem_cgroup_is_root(sc->memcg); } @@ -840,7 +840,7 @@ void mem_cgroup_iter_break(struct mem_cgroup *, struct mem_cgroup *); void mem_cgroup_scan_tasks(struct mem_cgroup *memcg, int (*)(struct task_struct *, void *), void *arg); -static inline unsigned short mem_cgroup_private_id(struct mem_cgroup *memcg) +static inline unsigned short mem_cgroup_private_id(const struct mem_cgroup *memcg) { if (mem_cgroup_disabled()) return 0; @@ -849,7 +849,7 @@ static inline unsigned short mem_cgroup_private_id(struct mem_cgroup *memcg) } struct mem_cgroup *mem_cgroup_from_private_id(unsigned short id); -static inline u64 mem_cgroup_id(struct mem_cgroup *memcg) +static inline u64 mem_cgroup_id(const struct mem_cgroup *memcg) { return memcg ? cgroup_id(memcg->css.cgroup) : 0; } @@ -878,13 +878,13 @@ static inline struct mem_cgroup *lruvec_memcg(const struct lruvec *lruvec) * * Returns the parent memcg, or NULL if this is the root. */ -static inline struct mem_cgroup *parent_mem_cgroup(struct mem_cgroup *memcg) +static inline struct mem_cgroup *parent_mem_cgroup(const struct mem_cgroup *memcg) { return mem_cgroup_from_css(memcg->css.parent); } -static inline bool mem_cgroup_is_descendant(struct mem_cgroup *memcg, - struct mem_cgroup *root) +static inline bool mem_cgroup_is_descendant(const struct mem_cgroup *memcg, + const struct mem_cgroup *root) { if (root == memcg) return true; @@ -892,7 +892,7 @@ static inline bool mem_cgroup_is_descendant(struct mem_cgroup *memcg, } static inline bool mm_match_cgroup(struct mm_struct *mm, - struct mem_cgroup *memcg) + const struct mem_cgroup *memcg) { struct mem_cgroup *task_memcg; bool match = false; @@ -943,7 +943,7 @@ static inline void mem_cgroup_handle_over_high(gfp_t gfp_mask) unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg); -void mem_cgroup_print_oom_context(struct mem_cgroup *memcg, +void mem_cgroup_print_oom_context(const struct mem_cgroup *memcg, struct task_struct *p); void mem_cgroup_print_oom_meminfo(struct mem_cgroup *memcg); @@ -1119,12 +1119,12 @@ static inline bool PageMemcgKmem(const struct page *page) return false; } -static inline bool mem_cgroup_is_root(struct mem_cgroup *memcg) +static inline bool mem_cgroup_is_root(const struct mem_cgroup *memcg) { return true; } -static inline bool mem_cgroup_shrink_is_root(struct shrink_control *sc) +static inline bool mem_cgroup_shrink_is_root(const struct shrink_control *sc) { return true; } @@ -1227,13 +1227,13 @@ static inline struct lruvec *folio_lruvec(const struct folio *folio) return &pgdat->__lruvec; } -static inline struct mem_cgroup *parent_mem_cgroup(struct mem_cgroup *memcg) +static inline struct mem_cgroup *parent_mem_cgroup(const struct mem_cgroup *memcg) { return NULL; } static inline bool mm_match_cgroup(struct mm_struct *mm, - struct mem_cgroup *memcg) + const struct mem_cgroup *memcg) { return true; } @@ -1327,7 +1327,7 @@ static inline void mem_cgroup_scan_tasks(struct mem_cgroup *memcg, { } -static inline unsigned short mem_cgroup_private_id(struct mem_cgroup *memcg) +static inline unsigned short mem_cgroup_private_id(const struct mem_cgroup *memcg) { return 0; } @@ -1339,7 +1339,7 @@ static inline struct mem_cgroup *mem_cgroup_from_private_id(unsigned short id) return NULL; } -static inline u64 mem_cgroup_id(struct mem_cgroup *memcg) +static inline u64 mem_cgroup_id(const struct mem_cgroup *memcg) { return 0; } @@ -1377,7 +1377,8 @@ static inline unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg) } static inline void -mem_cgroup_print_oom_context(struct mem_cgroup *memcg, struct task_struct *p) +mem_cgroup_print_oom_context(const struct mem_cgroup *memcg, + struct task_struct *p) { } @@ -1813,7 +1814,7 @@ static inline void memcg_kmem_uncharge_page(struct page *page, int order) * A helper for accessing memcg's kmem_id, used for getting * corresponding LRU lists. */ -static inline int memcg_kmem_id(struct mem_cgroup *memcg) +static inline int memcg_kmem_id(const struct mem_cgroup *memcg) { return memcg ? memcg->kmemcg_id : -1; } @@ -1885,7 +1886,7 @@ static inline bool memcg_kmem_online(void) return false; } -static inline int memcg_kmem_id(struct mem_cgroup *memcg) +static inline int memcg_kmem_id(const struct mem_cgroup *memcg) { return -1; } @@ -1962,7 +1963,7 @@ static inline bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg) #ifdef CONFIG_MEMCG_V1 bool mem_cgroup_oom_synchronize(bool wait); -static inline bool task_in_memcg_oom(struct task_struct *p) +static inline bool task_in_memcg_oom(const struct task_struct *p) { return p->memcg_in_oom; } @@ -1980,7 +1981,7 @@ static inline void mem_cgroup_exit_user_fault(void) } #else /* CONFIG_MEMCG_V1 */ -static inline bool task_in_memcg_oom(struct task_struct *p) +static inline bool task_in_memcg_oom(const struct task_struct *p) { return false; } diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 294827ef54a54c..adc93d28dbe72b 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -1832,7 +1832,8 @@ static void memory_stat_format(struct mem_cgroup *memcg, struct seq_buf *s) * NOTE: @memcg and @p's mem_cgroup can be different when hierarchy is * enabled */ -void mem_cgroup_print_oom_context(struct mem_cgroup *memcg, struct task_struct *p) +void mem_cgroup_print_oom_context(const struct mem_cgroup *memcg, + struct task_struct *p) { rcu_read_lock(); From 4a3af80d5d1acaef6fe58899ce8c2e948f147391 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:10 -0400 Subject: [PATCH 0729/1012] mm: page_counter: constify page_counter_read() and page_counter_margin() Both only read the counter. Constify them so that users can read counters from const memcgs. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-6-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Reviewed-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/page_counter.h | 4 ++-- mm/page_counter.c | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/include/linux/page_counter.h b/include/linux/page_counter.h index 07b7cb12249c7c..2baf7a2b29b2e1 100644 --- a/include/linux/page_counter.h +++ b/include/linux/page_counter.h @@ -63,12 +63,12 @@ static inline void page_counter_init(struct page_counter *counter, counter->track_failcnt = false; } -static inline unsigned long page_counter_read(struct page_counter *counter) +static inline unsigned long page_counter_read(const struct page_counter *counter) { return atomic_long_read(&counter->usage); } -long page_counter_margin(struct page_counter *counter); +long page_counter_margin(const struct page_counter *counter); void page_counter_cancel(struct page_counter *counter, unsigned long nr_pages); void page_counter_charge(struct page_counter *counter, unsigned long nr_pages); bool page_counter_try_charge(struct page_counter *counter, diff --git a/mm/page_counter.c b/mm/page_counter.c index 98322803941a70..9167ffd1380c82 100644 --- a/mm/page_counter.c +++ b/mm/page_counter.c @@ -54,7 +54,7 @@ static void propagate_protected_usage(struct page_counter *c, * Return: The minimum value of max minus usage across @counter and all of * its ancestors. The value may be negative during a concurrent charge. */ -long page_counter_margin(struct page_counter *counter) +long page_counter_margin(const struct page_counter *counter) { long margin = PAGE_COUNTER_MAX; From 2abb2506cd03e19dfff441bdcdedefdf4a5b8cdc Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:11 -0400 Subject: [PATCH 0730/1012] mm: memcontrol: constify the reclaim protection helpers mem_cgroup_protection(), mem_cgroup_unprotected(), mem_cgroup_below_low(), and mem_cgroup_below_min() only read the protection state. Constify them. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-7-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 32 ++++++++++++++++---------------- 1 file changed, 16 insertions(+), 16 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 01f74413fc79e9..22067899eb6c34 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -570,8 +570,8 @@ static inline bool mem_cgroup_disabled(void) return !cgroup_subsys_enabled(memory_cgrp_subsys); } -static inline void mem_cgroup_protection(struct mem_cgroup *root, - struct mem_cgroup *memcg, +static inline void mem_cgroup_protection(const struct mem_cgroup *root, + const struct mem_cgroup *memcg, unsigned long *min, unsigned long *low, unsigned long *usage) @@ -625,8 +625,8 @@ static inline void mem_cgroup_protection(struct mem_cgroup *root, void mem_cgroup_calculate_protection(struct mem_cgroup *root, struct mem_cgroup *memcg); -static inline bool mem_cgroup_unprotected(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_unprotected(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { /* * The root memcg doesn't account charges, and doesn't support @@ -637,8 +637,8 @@ static inline bool mem_cgroup_unprotected(struct mem_cgroup *target, memcg == target; } -static inline bool mem_cgroup_below_low(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_below_low(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { if (mem_cgroup_unprotected(target, memcg)) return false; @@ -647,8 +647,8 @@ static inline bool mem_cgroup_below_low(struct mem_cgroup *target, page_counter_read(&memcg->memory); } -static inline bool mem_cgroup_below_min(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_below_min(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { if (mem_cgroup_unprotected(target, memcg)) return false; @@ -1149,8 +1149,8 @@ static inline void memcg_memory_event_mm(struct mm_struct *mm, { } -static inline void mem_cgroup_protection(struct mem_cgroup *root, - struct mem_cgroup *memcg, +static inline void mem_cgroup_protection(const struct mem_cgroup *root, + const struct mem_cgroup *memcg, unsigned long *min, unsigned long *low, unsigned long *usage) @@ -1163,19 +1163,19 @@ static inline void mem_cgroup_calculate_protection(struct mem_cgroup *root, { } -static inline bool mem_cgroup_unprotected(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_unprotected(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { return true; } -static inline bool mem_cgroup_below_low(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_below_low(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { return false; } -static inline bool mem_cgroup_below_min(struct mem_cgroup *target, - struct mem_cgroup *memcg) +static inline bool mem_cgroup_below_min(const struct mem_cgroup *target, + const struct mem_cgroup *memcg) { return false; } From 5e9919064e5a2759e1b38e1b348bd0d90c3c354d Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:12 -0400 Subject: [PATCH 0731/1012] mm: memcontrol: constify the memcg and lruvec stat readers The memcg_page_state(), memcg_events(), and lruvec_page_state() families only read counters. Constify them. Use container_of_const() in the lruvec_page_state() family while at it, so the const isn't silently cast away. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-8-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 23 ++++++++++++----------- mm/memcontrol-v1.h | 6 +++--- mm/memcontrol.c | 30 +++++++++++++++--------------- 3 files changed, 30 insertions(+), 29 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 22067899eb6c34..9beb065c087935 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -971,15 +971,16 @@ static inline void mod_memcg_page_state(const struct page *page, rcu_read_unlock(); } -unsigned long memcg_events(struct mem_cgroup *memcg, int event); -unsigned long memcg_page_state(struct mem_cgroup *memcg, int idx); -unsigned long memcg_page_state_output(struct mem_cgroup *memcg, int item); +unsigned long memcg_events(const struct mem_cgroup *memcg, int event); +unsigned long memcg_page_state(const struct mem_cgroup *memcg, int idx); +unsigned long memcg_page_state_output(const struct mem_cgroup *memcg, int item); bool memcg_stat_item_valid(int idx); bool memcg_vm_event_item_valid(enum vm_event_item idx); -unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx); -unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec, +unsigned long lruvec_page_state(const struct lruvec *lruvec, + enum node_stat_item idx); +unsigned long lruvec_page_state_monotonic(const struct lruvec *lruvec, enum node_stat_item idx); -unsigned long lruvec_page_state_local(struct lruvec *lruvec, +unsigned long lruvec_page_state_local(const struct lruvec *lruvec, enum node_stat_item idx); void mem_cgroup_flush_stats(struct mem_cgroup *memcg); @@ -1412,12 +1413,12 @@ static inline void mod_memcg_page_state(const struct page *page, { } -static inline unsigned long memcg_page_state(struct mem_cgroup *memcg, int idx) +static inline unsigned long memcg_page_state(const struct mem_cgroup *memcg, int idx) { return 0; } -static inline unsigned long memcg_page_state_output(struct mem_cgroup *memcg, int item) +static inline unsigned long memcg_page_state_output(const struct mem_cgroup *memcg, int item) { return 0; } @@ -1432,19 +1433,19 @@ static inline bool memcg_vm_event_item_valid(enum vm_event_item idx) return false; } -static inline unsigned long lruvec_page_state(struct lruvec *lruvec, +static inline unsigned long lruvec_page_state(const struct lruvec *lruvec, enum node_stat_item idx) { return node_page_state(lruvec_pgdat(lruvec), idx); } -static inline unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec, +static inline unsigned long lruvec_page_state_monotonic(const struct lruvec *lruvec, enum node_stat_item idx) { return node_page_state_monotonic(lruvec_pgdat(lruvec), idx); } -static inline unsigned long lruvec_page_state_local(struct lruvec *lruvec, +static inline unsigned long lruvec_page_state_local(const struct lruvec *lruvec, enum node_stat_item idx) { return node_page_state(lruvec_pgdat(lruvec), idx); diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index 2cd37e1792d79e..0952b2a783e524 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -37,9 +37,9 @@ static inline bool do_memsw_account(void) return !cgroup_subsys_on_dfl(memory_cgrp_subsys); } -unsigned long memcg_events_local(struct mem_cgroup *memcg, int event); -unsigned long memcg_page_state_local(struct mem_cgroup *memcg, int idx); -unsigned long memcg_page_state_local_output(struct mem_cgroup *memcg, int item); +unsigned long memcg_events_local(const struct mem_cgroup *memcg, int event); +unsigned long memcg_page_state_local(const struct mem_cgroup *memcg, int idx); +unsigned long memcg_page_state_local_output(const struct mem_cgroup *memcg, int item); bool memcg1_alloc_events(struct mem_cgroup *memcg); void memcg1_free_events(struct mem_cgroup *memcg); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index adc93d28dbe72b..1b0e511a18638b 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -506,9 +506,9 @@ struct lruvec_stats { long state_pending[NR_MEMCG_NODE_STAT_ITEMS]; }; -unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx) +unsigned long lruvec_page_state(const struct lruvec *lruvec, enum node_stat_item idx) { - struct mem_cgroup_per_node *pn; + const struct mem_cgroup_per_node *pn; long x; int i; @@ -519,7 +519,7 @@ unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx) if (WARN_ONCE(BAD_STAT_IDX(i), "%s: missing stat item %d\n", __func__, idx)) return 0; - pn = container_of(lruvec, struct mem_cgroup_per_node, lruvec); + pn = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec); x = READ_ONCE(pn->lruvec_stats->state[i]); #ifdef CONFIG_SMP if (x < 0) @@ -547,10 +547,10 @@ unsigned long lruvec_page_state(struct lruvec *lruvec, enum node_stat_item idx) * monotonically-incremented event counters are stored in * enum node_stat_item. */ -unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec, +unsigned long lruvec_page_state_monotonic(const struct lruvec *lruvec, enum node_stat_item idx) { - struct mem_cgroup_per_node *pn; + const struct mem_cgroup_per_node *pn; int i; if (mem_cgroup_disabled()) @@ -560,14 +560,14 @@ unsigned long lruvec_page_state_monotonic(struct lruvec *lruvec, if (WARN_ONCE(BAD_STAT_IDX(i), "%s: missing stat item %d\n", __func__, idx)) return 0; - pn = container_of(lruvec, struct mem_cgroup_per_node, lruvec); + pn = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec); return (unsigned long)READ_ONCE(pn->lruvec_stats->state[i]); } -unsigned long lruvec_page_state_local(struct lruvec *lruvec, +unsigned long lruvec_page_state_local(const struct lruvec *lruvec, enum node_stat_item idx) { - struct mem_cgroup_per_node *pn; + const struct mem_cgroup_per_node *pn; long x; int i; @@ -578,7 +578,7 @@ unsigned long lruvec_page_state_local(struct lruvec *lruvec, if (WARN_ONCE(BAD_STAT_IDX(i), "%s: missing stat item %d\n", __func__, idx)) return 0; - pn = container_of(lruvec, struct mem_cgroup_per_node, lruvec); + pn = container_of_const(lruvec, struct mem_cgroup_per_node, lruvec); x = READ_ONCE(pn->lruvec_stats->state_local[i]); #ifdef CONFIG_SMP if (x < 0) @@ -841,7 +841,7 @@ static void flush_memcg_stats_dwork(struct work_struct *w) queue_delayed_work(system_dfl_wq, &stats_flush_dwork, FLUSH_TIME); } -unsigned long memcg_page_state(struct mem_cgroup *memcg, int idx) +unsigned long memcg_page_state(const struct mem_cgroup *memcg, int idx) { long x; int i = memcg_stats_index(idx); @@ -964,7 +964,7 @@ void mod_memcg_state(struct mem_cgroup *memcg, enum memcg_stat_item idx, #ifdef CONFIG_MEMCG_V1 /* idx can be of type enum memcg_stat_item or node_stat_item. */ -unsigned long memcg_page_state_local(struct mem_cgroup *memcg, int idx) +unsigned long memcg_page_state_local(const struct mem_cgroup *memcg, int idx) { long x; int i = memcg_stats_index(idx); @@ -1127,7 +1127,7 @@ void count_memcg_events(struct mem_cgroup *memcg, enum vm_event_item idx, put_cpu(); } -unsigned long memcg_events(struct mem_cgroup *memcg, int event) +unsigned long memcg_events(const struct mem_cgroup *memcg, int event) { int i = memcg_events_index(event); @@ -1146,7 +1146,7 @@ bool memcg_vm_event_item_valid(enum vm_event_item idx) } #ifdef CONFIG_MEMCG_V1 -unsigned long memcg_events_local(struct mem_cgroup *memcg, int event) +unsigned long memcg_events_local(const struct mem_cgroup *memcg, int event) { int i = memcg_events_index(event); @@ -1729,14 +1729,14 @@ static int memcg_page_state_output_unit(int item) } } -unsigned long memcg_page_state_output(struct mem_cgroup *memcg, int item) +unsigned long memcg_page_state_output(const struct mem_cgroup *memcg, int item) { return memcg_page_state(memcg, item) * memcg_page_state_output_unit(item); } #ifdef CONFIG_MEMCG_V1 -unsigned long memcg_page_state_local_output(struct mem_cgroup *memcg, int item) +unsigned long memcg_page_state_local_output(const struct mem_cgroup *memcg, int item) { return memcg_page_state_local(memcg, item) * memcg_page_state_output_unit(item); From 87696b2968e3fb268b82d6405b1c90603bcec6ab Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:13 -0400 Subject: [PATCH 0732/1012] mm: memcontrol: constify the swap accounting helpers mem_cgroup_get_nr_swap_pages(), mem_cgroup_get_folio_swap_margin(), and mem_cgroup_swap_full() only read swap counters and limits. Constify them. Remove externs from function declarations while at it. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-9-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/swap.h | 12 ++++++------ mm/memcontrol.c | 6 +++--- 2 files changed, 9 insertions(+), 9 deletions(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index 61005501888c53..82f0bfec611fc7 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -527,9 +527,9 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_p __mem_cgroup_uncharge_swap(id, nr_pages); } -long mem_cgroup_get_folio_swap_margin(struct folio *folio); -extern long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg); -extern bool mem_cgroup_swap_full(struct folio *folio); +long mem_cgroup_get_folio_swap_margin(const struct folio *folio); +long mem_cgroup_get_nr_swap_pages(const struct mem_cgroup *memcg); +bool mem_cgroup_swap_full(const struct folio *folio); #else static inline int mem_cgroup_try_charge_swap(struct folio *folio) { @@ -541,17 +541,17 @@ static inline void mem_cgroup_uncharge_swap(unsigned short id, { } -static inline long mem_cgroup_get_folio_swap_margin(struct folio *folio) +static inline long mem_cgroup_get_folio_swap_margin(const struct folio *folio) { return PAGE_COUNTER_MAX; } -static inline long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) +static inline long mem_cgroup_get_nr_swap_pages(const struct mem_cgroup *memcg) { return get_nr_swap_pages(); } -static inline bool mem_cgroup_swap_full(struct folio *folio) +static inline bool mem_cgroup_swap_full(const struct folio *folio) { return vm_swap_full(); } diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 1b0e511a18638b..d954ffb43a4323 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -6017,7 +6017,7 @@ void __mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_pages) rcu_read_unlock(); } -long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) +long mem_cgroup_get_nr_swap_pages(const struct mem_cgroup *memcg) { long nr_swap_pages = get_nr_swap_pages(); @@ -6033,7 +6033,7 @@ long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) * * Return: Remaining chargeable pages in the folio's memcg hierarchy. */ -long mem_cgroup_get_folio_swap_margin(struct folio *folio) +long mem_cgroup_get_folio_swap_margin(const struct folio *folio) { struct mem_cgroup *memcg; long margin; @@ -6050,7 +6050,7 @@ long mem_cgroup_get_folio_swap_margin(struct folio *folio) return margin; } -bool mem_cgroup_swap_full(struct folio *folio) +bool mem_cgroup_swap_full(const struct folio *folio) { struct mem_cgroup *memcg; bool ret = false; From f537c2aee5135852a067a4d5ec1969d38884565b Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:14 -0400 Subject: [PATCH 0733/1012] mm: memcontrol: constify mem_cgroup_swappiness() and mem_cgroup_get_max() Both only read swappiness and the memory limits. Constify them. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-10-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 4 ++-- mm/memcontrol.c | 2 +- mm/swap.h | 2 +- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 9beb065c087935..f118990854734e 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -941,7 +941,7 @@ static inline void mem_cgroup_handle_over_high(gfp_t gfp_mask) __mem_cgroup_handle_over_high(gfp_mask); } -unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg); +unsigned long mem_cgroup_get_max(const struct mem_cgroup *memcg); void mem_cgroup_print_oom_context(const struct mem_cgroup *memcg, struct task_struct *p); @@ -1372,7 +1372,7 @@ unsigned long mem_cgroup_get_zone_lru_size(const struct lruvec *lruvec, return 0; } -static inline unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg) +static inline unsigned long mem_cgroup_get_max(const struct mem_cgroup *memcg) { return 0; } diff --git a/mm/memcontrol.c b/mm/memcontrol.c index d954ffb43a4323..36101129679fd0 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -1898,7 +1898,7 @@ void mem_cgroup_print_oom_meminfo(struct mem_cgroup *memcg) /* * Return the memory (and swap, if configured) limit for a memcg. */ -unsigned long mem_cgroup_get_max(struct mem_cgroup *memcg) +unsigned long mem_cgroup_get_max(const struct mem_cgroup *memcg) { unsigned long max = READ_ONCE(memcg->memory.max); diff --git a/mm/swap.h b/mm/swap.h index b3b54c28929a19..1957960dc60d63 100644 --- a/mm/swap.h +++ b/mm/swap.h @@ -82,7 +82,7 @@ enum swap_cluster_flags { extern int vm_swappiness; -static inline int mem_cgroup_swappiness(struct mem_cgroup *memcg) +static inline int mem_cgroup_swappiness(const struct mem_cgroup *memcg) { #ifdef CONFIG_MEMCG_V1 if (!cgroup_subsys_on_dfl(memory_cgrp_subsys) && From 62388e537086811932d1c23e42cb9d642f61c157 Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:20:15 -0400 Subject: [PATCH 0734/1012] mm: memcontrol: constify the zswap and socket pressure helpers mem_cgroup_zswap_writeback_enabled() only reads the zswap_writeback flags, and mem_cgroup_get_socket_pressure() only reads the socket pressure timestamp. Constify them. Link: https://lore.kernel.org/20260915-folio_memcg-const-v3-11-c239a6010b58@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Muchun Song Acked-by: Shakeel Butt Acked-by: Lorenzo Stoakes (ARM) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Chris Li Cc: David Hildenbrand Cc: Johannes Weiner Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Mike Rapoport Cc: Nhat Pham Cc: Roman Gushchin Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/memcontrol.h | 8 ++++---- mm/memcontrol.c | 2 +- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index f118990854734e..64c183be8cbfe7 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -1684,7 +1684,7 @@ static inline void mem_cgroup_set_socket_pressure(struct mem_cgroup *memcg) write_sequnlock_irqrestore(&memcg->socket_pressure_seqlock, flags); } -static inline u64 mem_cgroup_get_socket_pressure(struct mem_cgroup *memcg) +static inline u64 mem_cgroup_get_socket_pressure(const struct mem_cgroup *memcg) { unsigned int seq; u64 val; @@ -1702,7 +1702,7 @@ static inline void mem_cgroup_set_socket_pressure(struct mem_cgroup *memcg) WRITE_ONCE(memcg->socket_pressure, jiffies + HZ); } -static inline u64 mem_cgroup_get_socket_pressure(struct mem_cgroup *memcg) +static inline u64 mem_cgroup_get_socket_pressure(const struct mem_cgroup *memcg) { return READ_ONCE(memcg->socket_pressure); } @@ -1937,7 +1937,7 @@ static inline void mem_cgroup_calculate_protection_path(struct mem_cgroup *root, bool obj_cgroup_may_zswap(struct obj_cgroup *objcg); void obj_cgroup_charge_zswap(struct obj_cgroup *objcg, size_t size); void obj_cgroup_uncharge_zswap(struct obj_cgroup *objcg, size_t size); -bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg); +bool mem_cgroup_zswap_writeback_enabled(const struct mem_cgroup *memcg); #else static inline bool obj_cgroup_may_zswap(struct obj_cgroup *objcg) { @@ -1951,7 +1951,7 @@ static inline void obj_cgroup_uncharge_zswap(struct obj_cgroup *objcg, size_t size) { } -static inline bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg) +static inline bool mem_cgroup_zswap_writeback_enabled(const struct mem_cgroup *memcg) { /* if zswap is disabled, do not block pages going to the swapping device */ return true; diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 36101129679fd0..4d00748c8a5b87 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -6318,7 +6318,7 @@ void obj_cgroup_uncharge_zswap(struct obj_cgroup *objcg, size_t size) rcu_read_unlock(); } -bool mem_cgroup_zswap_writeback_enabled(struct mem_cgroup *memcg) +bool mem_cgroup_zswap_writeback_enabled(const struct mem_cgroup *memcg) { /* if zswap is disabled, do not block pages going to the swapping device */ if (!zswap_is_enabled()) From 445da827e5e24baac3c5415cc202aaf1d22d6ba1 Mon Sep 17 00:00:00 2001 From: Usama Arif Date: Wed, 16 Sep 2026 05:51:22 -0700 Subject: [PATCH 0735/1012] mm: filemap: move lruvec accounting outside the xarray lock __filemap_add_folio() inserts a folio and updates mapping->nrpages while holding mapping->i_pages.xa_lock with interrupts disabled. The XArray insertion and nrpages update require the lock, but the lruvec statistic updates do not. With CONFIG_MEMCG, those calls also update per-CPU memcg and lruvec counters and notify cgroup rstat, extending the critical section. Move the lruvec accounting after a successful XArray insertion and after xas_unlock_irq(). The page-cache references pin the folio, while the folio lock keeps folio->mapping stable and prevents removal until accounting is complete. This moves one lruvec update for ordinary folios and a second for PMD-mappable folios out of the serialized section. In a 30-second system-wide perf lock contention -ab capture on a production host, the hottest caller-stack record attributed to __filemap_add_folio() had 20,867 contentions and 557.930 ms total wait. That was 14% of the 3.998 seconds of aggregate lock wait in the capture. Moving lruvec accuting outside of critical section should help optimize it. Link: https://lore.kernel.org/20260916125122.2696271-1-usama.arif@linux.dev Signed-off-by: Usama Arif Signed-off-by: Andrew Morton Reviewed-by: Shakeel Butt Acked-by: Muchun Song Reviewed-by: Vishal Moola (Fractile) Reviewed-by: Jan Kara Cc: David Hildenbrand Cc: Johannes Weiner Cc: Matthew Wilcox (Oracle) Cc: Michal Hocko Cc: Rik van Riel Cc: Roman Gushchin --- mm/filemap.c | 15 +++++++-------- 1 file changed, 7 insertions(+), 8 deletions(-) diff --git a/mm/filemap.c b/mm/filemap.c index 00fd89cf6f5509..4720bbfc1a6636 100644 --- a/mm/filemap.c +++ b/mm/filemap.c @@ -918,14 +918,6 @@ noinline int __filemap_add_folio(struct address_space *mapping, mapping->nrpages += nr; - /* hugetlb pages do not participate in page cache accounting */ - if (!huge) { - lruvec_stat_mod_folio(folio, NR_FILE_PAGES, nr); - if (folio_test_pmd_mappable(folio)) - lruvec_stat_mod_folio(folio, - NR_FILE_THPS, nr); - } - unlock: xas_unlock_irq(&xas); @@ -942,6 +934,13 @@ noinline int __filemap_add_folio(struct address_space *mapping, if (xas_error(&xas)) goto error; + /* hugetlb pages do not participate in page cache accounting */ + if (!huge) { + lruvec_stat_mod_folio(folio, NR_FILE_PAGES, nr); + if (folio_test_pmd_mappable(folio)) + lruvec_stat_mod_folio(folio, NR_FILE_THPS, nr); + } + trace_mm_filemap_add_to_page_cache(folio); return 0; error: From b43e5238c4cb4c5a72eabaeaf3807ebebcc2210b Mon Sep 17 00:00:00 2001 From: Yuanhe Shu Date: Wed, 16 Sep 2026 19:25:45 +0800 Subject: [PATCH 0736/1012] mm/page_alloc: do not boost watermarks in kdump capture kernels A watermark boost is not confined to one watermark: wmark_pages() adds it to min, low and high alike, so every watermark check sees it, including should_reclaim_retry() and the last ditch ALLOC_WMARK_HIGH attempt in __alloc_pages_may_oom(). Once the boost exceeds the memory still free the allocator gives up and invokes the OOM killer, and a capture kernel that is still booting has nothing to kill: the boot panics and the vmcore is lost. Seen on an arm64 machine with 64K pages, CONFIG_PAGE_BLOCK_MAX_ORDER=10 (pageblock = 64M) and crashkernel=512M, running a distribution kernel based on 7.0.14. A high order UNMOVABLE allocation fell back to a MOVABLE pageblock while the capture kernel was still in do_initcalls(): Node 0 DMA free:68096kB boost:65536kB min:68160kB low:68800kB high:69440kB managed:479168kB Out of memory and no killable processes... Kernel panic - not syncing: System is deadlocked on memory The zone was not short of memory. Subtracting the boost gives min:2624kB low:3264kB high:3904kB, so the 68096kB still free sat 17 times above the high watermark and the allocator would not even have entered its slow path. The boost supplied 65536kB of the 68160kB min and by itself put the zone 64kB under water. It is that large because boost_watermark() clamps it with max(pageblock_nr_pages, max_boost); watermark_boost_factor alone would have allowed 5824kB. Commit 14f69140ff9c ("mm: limit boost_watermark on small zones") already tried to protect capture kernels, but it infers them from the zone size and skips the boost only below four pageblocks. arm64 64K pageblocks were 512M then, so the guard reached zones up to 2G; CONFIG_PAGE_BLOCK_MAX_ORDER can cap them at 64M, which shrinks the guard to zones under 256M and lets this 468M zone through. kdump is a property of the kernel, not of the zone, so test for it directly. A capture kernel exits within seconds and never uses the fragmentation avoidance the boost buys. Normal kernels are unaffected: the size based check still covers their genuinely tiny zones. Passing sysctl.vm.watermark_boost_factor=0 to the capture kernel does not cover this window: sysctl.* parameters are written through procfs by do_sysctl_args(), which runs after do_initcalls() where the panic above happened, and watermark_boost_factor has no early_param of its own. Link: https://lore.kernel.org/20260916112545.3707893-1-xiangzao@linux.alibaba.com Fixes: 1c30844d2dfe ("mm: reclaim small amounts of memory when an external fragmentation event occurs") Signed-off-by: Yuanhe Shu Signed-off-by: Andrew Morton Reviewed-by: Vlastimil Babka (SUSE) Reviewed-by: Johannes Weiner Cc: Henry Willard Cc: Brendan Jackman Cc: David Hildenbrand Cc: Mel Gorman Cc: Michal Hocko Cc: Suren Baghdasaryan Cc: Zi Yan Cc: # v5.0 --- mm/page_alloc.c | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/mm/page_alloc.c b/mm/page_alloc.c index 25f0bdf26ce839..641384085e3416 100644 --- a/mm/page_alloc.c +++ b/mm/page_alloc.c @@ -37,6 +37,7 @@ #include #include #include +#include #include #include #include @@ -2160,10 +2161,17 @@ static inline bool boost_watermark(struct zone *zone) if (!watermark_boost_factor) return false; + + /* + * A kdump capture kernel exits before a boost can pay off, while + * the raised watermark can exceed the memory left for the dump. + */ + if (is_kdump_kernel()) + return false; + /* * Don't bother in zones that are unlikely to produce results. - * On small machines, including kdump capture kernels running - * in a small area, boosting the watermark can cause an out of + * On small machines, boosting the watermark can cause an out of * memory situation immediately. */ if ((pageblock_nr_pages * 4) > zone_managed_pages(zone)) From 36c526d5e657c505aeb7b8c53bd41e469cd7d0bd Mon Sep 17 00:00:00 2001 From: Kefeng Wang Date: Wed, 16 Sep 2026 12:31:52 +0800 Subject: [PATCH 0737/1012] mm: mincore: use per-vma lock during page table walk do_mincore() performs a read-only, per-VMA residency query, making it a good candidate for per-VMA locking. Convert it to acquire the per-VMA lock, thereby reducing contention on the per-MM mmap_lock. Link: https://lore.kernel.org/20260916043153.2631696-1-wangkefeng.wang@huawei.com Signed-off-by: Kefeng Wang Signed-off-by: Andrew Morton Reviewed-by: Pedro Falcato Acked-by: David Hildenbrand (Arm) Reviewed-by: Lorenzo Stoakes (ARM) Cc: Jann Horn Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Zi Yan --- mm/mincore.c | 29 +++++++++++++++++------------ 1 file changed, 17 insertions(+), 12 deletions(-) diff --git a/mm/mincore.c b/mm/mincore.c index c086836bc4bcc5..0fe50f8a7e6291 100644 --- a/mm/mincore.c +++ b/mm/mincore.c @@ -235,24 +235,22 @@ static const struct mm_walk_ops mincore_walk_ops = { .pmd_entry = mincore_pte_range, .pte_hole = mincore_unmapped_range, .hugetlb_entry = mincore_hugetlb, - .walk_lock = PGWALK_RDLOCK, + .walk_lock = PGWALK_VMA_RDLOCK_VERIFY, }; /* * Do a chunk of "sys_mincore()". We've already checked - * all the arguments, we hold the mmap semaphore: we should + * all the arguments, we hold the VMA read lock: we should * just return the amount of info we're asked for. */ -static long do_mincore(unsigned long addr, unsigned long pages, unsigned char *vec) +static long do_mincore(struct vm_area_struct *vma, unsigned long addr, + unsigned long pages, unsigned char *vec) { - struct vm_area_struct *vma; - unsigned long end; + unsigned long end = min(vma->vm_end, addr + (pages << PAGE_SHIFT)); int err; - vma = vma_lookup(current->mm, addr); - if (!vma) - return -ENOMEM; - end = min(vma->vm_end, addr + (pages << PAGE_SHIFT)); + vma_assert_locked(vma); + if (!can_do_mincore(vma)) { unsigned long pages = DIV_ROUND_UP(end - addr, PAGE_SIZE); memset(vec, 1, pages); @@ -319,13 +317,20 @@ SYSCALL_DEFINE3(mincore, unsigned long, start, size_t, len, retval = 0; while (pages) { + struct vm_area_struct *vma; + + vma = vma_start_read_unlocked(current->mm, start); + if (!vma) { + retval = -ENOMEM; + break; + } + /* * Do at most PAGE_SIZE entries per iteration, due to * the temporary buffer size. */ - mmap_read_lock(current->mm); - retval = do_mincore(start, min(pages, PAGE_SIZE), tmp); - mmap_read_unlock(current->mm); + retval = do_mincore(vma, start, min(pages, PAGE_SIZE), tmp); + vma_end_read(vma); if (retval <= 0) break; From 64001f944618c109bbb340cb3c7572602bb0635c Mon Sep 17 00:00:00 2001 From: Tal Zussman Date: Tue, 15 Sep 2026 19:42:15 -0400 Subject: [PATCH 0738/1012] vmcore: convert mmap_vmcore_fault() to use folios Use a folio for the page cache page the s390 fault handler reads the old kernel's memory into. This removes four compound_head() calls and one of the last callers of find_or_create_page(). The vmcore mapping only has order-0 folios, so the logic is unchanged. Compile tested for s390. Link: https://lore.kernel.org/20260915-vmcore-fault-folio-v1-1-a0cb6278670f@columbia.edu Signed-off-by: Tal Zussman Signed-off-by: Andrew Morton Acked-by: Pratyush Yadav Cc: Baoquan He Cc: Dave Young Cc: Mike Rapoport Cc: Pasha Tatashin --- fs/proc/vmcore.c | 27 ++++++++++++++------------- 1 file changed, 14 insertions(+), 13 deletions(-) diff --git a/fs/proc/vmcore.c b/fs/proc/vmcore.c index 44d15436439fd9..2dc3803aa11dcd 100644 --- a/fs/proc/vmcore.c +++ b/fs/proc/vmcore.c @@ -474,29 +474,30 @@ static vm_fault_t mmap_vmcore_fault(struct vm_fault *vmf) pgoff_t index = vmf->pgoff; struct iov_iter iter; struct kvec kvec; - struct page *page; + struct folio *folio; loff_t offset; int rc; - page = find_or_create_page(mapping, index, GFP_KERNEL); - if (!page) + folio = __filemap_get_folio(mapping, index, + FGP_LOCK | FGP_ACCESSED | FGP_CREAT, GFP_KERNEL); + if (IS_ERR(folio)) return VM_FAULT_OOM; - if (!PageUptodate(page)) { - offset = (loff_t) index << PAGE_SHIFT; - kvec.iov_base = page_address(page); - kvec.iov_len = PAGE_SIZE; - iov_iter_kvec(&iter, ITER_DEST, &kvec, 1, PAGE_SIZE); + if (!folio_test_uptodate(folio)) { + offset = folio_pos(folio); + kvec.iov_base = folio_address(folio); + kvec.iov_len = folio_size(folio); + iov_iter_kvec(&iter, ITER_DEST, &kvec, 1, folio_size(folio)); rc = __read_vmcore(&iter, &offset); if (rc < 0) { - unlock_page(page); - put_page(page); + folio_unlock(folio); + folio_put(folio); return vmf_error(rc); } - SetPageUptodate(page); + folio_mark_uptodate(folio); } - unlock_page(page); - vmf->page = page; + folio_unlock(folio); + vmf->page = folio_file_page(folio, index); return 0; #else return VM_FAULT_SIGBUS; From 12445795cf4208cc59cce5ae979651f6018f14d0 Mon Sep 17 00:00:00 2001 From: Alexandre Ghiti Date: Mon, 21 Sep 2026 17:13:02 +0200 Subject: [PATCH 0739/1012] mm: swap: move LRU insertion out of the swap cache allocator Patch series "mm: zswap: free cold writeback folios promptly", v6. When zswap writes an entry back, it allocates an order-0 swap cache folio, decompresses into it, and issues the write. The folio is cold by construction, yet today it is left on the LRU for page reclaim to find and free later. That wastes a reclaim scan and keeps cold memory resident longer than necessary. Rather than implement this in zswap, extend the existing dropbehind mechanism to swap cache folios and have zswap opt into it (Yosry). A PG_dropbehind folio is already dropped from its cache once writeback completes instead of being left for reclaim; for a swap cache folio that "drop" is removing it from the swap cache. Patch 1 - move LRU insertion out of the swap cache allocator into its callers, so zswap writeback can allocate off the LRU. Patch 2 - drop dropbehind swap cache folios on writeback completion. Patch 3 - zswap allocates its writeback folio off the LRU and marks it dropbehind, opting into the mechanism above. This patch (of 3): This is a preparatory patch. swap_cache_alloc_folio() adds the new folio to the LRU itself, which leaves its callers no way to act on the folio before it becomes visible to reclaim. Two users need exactly that: - moving the refault evaluation out of the swap cache folio allocation requires it to happen before folio_add_lru(): that consumes PG_active to file the folio on the inactive or the active list, and under MGLRU it also reads PG_workingset to pick the generation. Setting either flag afterwards does not move the folio; - zswap writeback dropbehind needs the buffer folio to stay off the LRU entirely, as the per-CPU LRU batch would hold a reference on it and keep remove_mapping() from freeing it once writeback completes. Defer the LRU insertion to the callers and rename the helper to __swap_cache_alloc_folio(): each caller adds the folio right after the allocation, so there is no functional change intended. Link: https://lore.kernel.org/20260921151306.625134-1-alex@ghiti.fr Link: https://lore.kernel.org/20260921151306.625134-2-alex@ghiti.fr Signed-off-by: Alexandre Ghiti Signed-off-by: Andrew Morton Suggested-by: Kairui Song Reviewed-by: Kairui Song Reviewed-by: Nhat Pham Reviewed-by: Kunwu Chan Acked-by: Usama Arif Reviewed-by: Barry Song Cc: Al Viro Cc: Axel Rasmussen Cc: Baoquan He Cc: Chengming Zhou Cc: Chis Li (Google) Cc: Christian Brauner (Amutable) Cc: "David Hildenbrand (arm)" Cc: Jan Kara Cc: Johannes Weiner Cc: Kemeng Shi Cc: Lorenzo Stoakes (ARM) Cc: Matthew Wilcox Cc: Michal Hocko Cc: Qi Zheng Cc: Shakeel Butt Cc: Tal Zussman Cc: Wei Xu Cc: Yosry Ahmed Cc: Youngjun Park Cc: Yuanchu Xie --- mm/swap.h | 6 +++--- mm/swap_state.c | 20 ++++++++++++-------- mm/swapfile.c | 2 +- mm/zswap.c | 5 +++-- 4 files changed, 19 insertions(+), 14 deletions(-) diff --git a/mm/swap.h b/mm/swap.h index 1957960dc60d63..d5bf21f517dcea 100644 --- a/mm/swap.h +++ b/mm/swap.h @@ -312,9 +312,9 @@ bool swap_cache_has_folio(swp_entry_t entry); struct folio *swap_cache_get_folio(swp_entry_t entry); void *swap_cache_get_shadow(swp_entry_t entry); void swap_cache_del_folio(struct folio *folio); -struct folio *swap_cache_alloc_folio(swp_entry_t target_entry, gfp_t gfp_mask, - unsigned long orders, struct vm_fault *vmf, - struct mempolicy *mpol, pgoff_t ilx); +struct folio *__swap_cache_alloc_folio(swp_entry_t target_entry, gfp_t gfp_mask, + unsigned long orders, struct vm_fault *vmf, + struct mempolicy *mpol, pgoff_t ilx); /* Below helpers require the caller to lock and pass in the swap cluster. */ void __swap_cache_add_folio(struct swap_cluster_info *ci, struct folio *folio, swp_entry_t entry); diff --git a/mm/swap_state.c b/mm/swap_state.c index cef44aadee6158..87790369dd9aa3 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -497,13 +497,11 @@ static struct folio *__swap_cache_alloc(struct swap_cluster_info *ci, node_stat_mod_folio(folio, NR_FILE_PAGES, nr_pages); lruvec_stat_mod_folio(folio, NR_SWAPCACHE, nr_pages); - /* Caller will initiate read into locked new_folio */ - folio_add_lru(folio); return folio; } /** - * swap_cache_alloc_folio - Allocate folio for swapped out slot in swap cache. + * __swap_cache_alloc_folio - Allocate folio for swapped out slot in swap cache. * @targ_entry: swap entry indicating the target slot * @gfp: memory allocation flags * @orders: allocation orders, must be non zero @@ -515,13 +513,17 @@ static struct folio *__swap_cache_alloc(struct swap_cluster_info *ci, * doing IO (e.g. swap in or zswap writeback). The swap slot indicated by * @targ_entry must have a non-zero swap count (swapped out). * + * The returned folio is locked and is NOT on the LRU. The caller must either + * add it to the LRU with folio_add_lru() so page reclaim can find it, or free + * it directly once done; a folio left off the LRU is unreclaimable and leaks. + * * Context: Caller must protect the swap device with reference count or locks. * Return: Returns the folio if allocation succeeded and folio is in the swap * cache. Returns error code if failed due to race, OOM or invalid arguments. */ -struct folio *swap_cache_alloc_folio(swp_entry_t targ_entry, gfp_t gfp, - unsigned long orders, struct vm_fault *vmf, - struct mempolicy *mpol, pgoff_t ilx) +struct folio *__swap_cache_alloc_folio(swp_entry_t targ_entry, gfp_t gfp, + unsigned long orders, struct vm_fault *vmf, + struct mempolicy *mpol, pgoff_t ilx) { int order, err; struct folio *ret; @@ -657,12 +659,13 @@ static struct folio *swap_cache_read_folio(struct swap_io_ctx *ctx, folio = swap_cache_get_folio(entry); if (folio) return folio; - folio = swap_cache_alloc_folio(entry, gfp, BIT(0), NULL, mpol, ilx); + folio = __swap_cache_alloc_folio(entry, gfp, BIT(0), NULL, mpol, ilx); } while (PTR_ERR(folio) == -EEXIST); if (IS_ERR_OR_NULL(folio)) return NULL; + folio_add_lru(folio); swap_read_folio(ctx, folio); if (readahead) { folio_set_readahead(folio); @@ -698,12 +701,13 @@ struct folio *swapin_sync(swp_entry_t entry, gfp_t gfp, unsigned long orders, folio = swap_cache_get_folio(entry); if (folio) return folio; - folio = swap_cache_alloc_folio(entry, gfp, orders, vmf, mpol, ilx); + folio = __swap_cache_alloc_folio(entry, gfp, orders, vmf, mpol, ilx); } while (PTR_ERR(folio) == -EEXIST); if (IS_ERR(folio)) return folio; + folio_add_lru(folio); swap_read_folio(&ctx, folio); swap_read_submit(&ctx); return folio; diff --git a/mm/swapfile.c b/mm/swapfile.c index c1c5fbb3c909d3..0901cb8fa7291c 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1887,7 +1887,7 @@ void folio_put_swap(struct folio *folio, struct page *page) * CPU1 CPU2 * do_swap_page() * ... swapoff+swapon - * swap_cache_alloc_folio() + * __swap_cache_alloc_folio() * // check swap_map * // verify PTE not changed * diff --git a/mm/zswap.c b/mm/zswap.c index 584dd306376943..c79cca61abf933 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1019,8 +1019,8 @@ static int zswap_writeback_entry(struct zswap_entry *entry, return -ENOENT; mpol = get_task_policy(current); - folio = swap_cache_alloc_folio(swpentry, GFP_KERNEL, BIT(0), NULL, mpol, - NO_INTERLEAVE_INDEX); + folio = __swap_cache_alloc_folio(swpentry, GFP_KERNEL, BIT(0), NULL, mpol, + NO_INTERLEAVE_INDEX); put_swap_device(si); /* @@ -1032,6 +1032,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry, */ if (IS_ERR(folio)) return PTR_ERR(folio); + folio_add_lru(folio); /* * folio is locked, and the swapcache is now secured against From 771da70595ad9a20dc047168c46c439f3262bd5d Mon Sep 17 00:00:00 2001 From: Alexandre Ghiti Date: Mon, 21 Sep 2026 17:13:03 +0200 Subject: [PATCH 0740/1012] mm: swap: drop dropbehind swap cache folios on writeback completion A PG_dropbehind folio is dropped from its cache once writeback completes rather than left for reclaim to find later; this is implemented for file folios in folio_end_dropbehind(). Extend it to swap cache folios. The drop blocks on the folio lock, so it cannot run in interrupt context. Set BIO_COMPLETE_IN_TASK on the write, as the file dropbehind paths do, and drop the folio directly from folio_end_writeback(). It has to block rather than trylock: the folio is off the LRU, so skipping it would leave it in the swap cache with nothing able to reclaim it, and it cannot be put back while another thread holds its lock. Link: https://lore.kernel.org/20260921151306.625134-3-alex@ghiti.fr Signed-off-by: Alexandre Ghiti Signed-off-by: Andrew Morton Suggested-by: Yosry Ahmed Suggested-by: Johannes Weiner Suggested-by: Nhat Pham Reviewed-by: Nhat Pham Reviewed-by: Kunwu Chan Reviewed-by: Barry Song Cc: Al Viro Cc: Axel Rasmussen Cc: Baoquan He Cc: Chengming Zhou Cc: Chis Li Cc: Christian Brauner (Amutable) Cc: "David Hildenbrand (arm)" Cc: Jan Kara Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes (ARM) Cc: Matthew Wilcox Cc: Michal Hocko Cc: Qi Zheng Cc: Shakeel Butt Cc: Tal Zussman Cc: Usama Arif Cc: Wei Xu Cc: Youngjun Park Cc: Yuanchu Xie --- include/linux/swap.h | 6 ++++++ mm/filemap.c | 19 +++++++++++++++++ mm/page_io.c | 9 ++++++++ mm/swap_state.c | 42 +++++++++++++++++++++++++++++++++++++ mm/vmscan.c | 49 +++++++++++++++++++++++++++++++++++--------- 5 files changed, 115 insertions(+), 10 deletions(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index 82f0bfec611fc7..cb434cccd653a6 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -336,6 +336,9 @@ static inline bool lru_cache_disabled(void) extern unsigned long shrink_all_memory(unsigned long nr_pages); long remove_mapping(struct address_space *mapping, struct folio *folio); +long remove_mapping_set_shadow(struct address_space *mapping, + struct folio *folio, + struct mem_cgroup *target_memcg); #if defined(CONFIG_SYSFS) && defined(CONFIG_NUMA) extern int reclaim_register_node(struct node *node); @@ -421,6 +424,8 @@ void swap_put_entries_direct(swp_entry_t entry, int nr); */ bool folio_free_swap(struct folio *folio); +void swap_writeback_dropbehind_folio(struct folio *folio); + /* Allocate / free (hibernation) exclusive entries */ swp_entry_t swap_alloc_hibernation_slot(int type); void swap_free_hibernation_slot(swp_entry_t entry); @@ -431,6 +436,7 @@ static inline void put_swap_device(struct swap_info_struct *si) } #else /* CONFIG_SWAP */ +static inline void swap_writeback_dropbehind_folio(struct folio *folio) {} static inline struct swap_info_struct *get_swap_device(swp_entry_t entry) { return NULL; diff --git a/mm/filemap.c b/mm/filemap.c index 4720bbfc1a6636..b74bc1e5015c6f 100644 --- a/mm/filemap.c +++ b/mm/filemap.c @@ -1685,6 +1685,8 @@ EXPORT_SYMBOL_GPL(folio_end_writeback_no_dropbehind); */ void folio_end_writeback(struct folio *folio) { + bool swap_dropbehind; + VM_BUG_ON_FOLIO(!folio_test_writeback(folio), folio); /* @@ -1694,7 +1696,24 @@ void folio_end_writeback(struct folio *folio) * reused before the folio_wake_bit(). */ folio_get(folio); + + /* + * Sample this before folio_end_writeback_no_dropbehind() clears + * PG_writeback: until then a racing swapin cannot remove the folio from + * the swap cache. Afterwards it can, and the drop below then finds a + * non-swapcache folio and puts it back on the LRU instead. The + * reference taken above keeps the folio alive across that window. + */ + swap_dropbehind = folio_test_swapcache(folio) && + folio_test_dropbehind(folio); + folio_end_writeback_no_dropbehind(folio); + + if (swap_dropbehind) { + swap_writeback_dropbehind_folio(folio); + return; + } + folio_end_dropbehind(folio); folio_put(folio); } diff --git a/mm/page_io.c b/mm/page_io.c index 5f7756e370f7a4..0808808f0309ce 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -608,6 +608,15 @@ static void swap_bdev_submit_write(struct swap_io_ctx *ctx) submit_bio_wait(bio); end_swap_bio_write(bio); } else { + int p; + + for (p = 0; p < sio->nr_bvecs; p++) { + if (folio_test_dropbehind(bvec_folio(&sio->bvecs[p]))) { + bio_set_flag(bio, BIO_COMPLETE_IN_TASK); + break; + } + } + bio->bi_end_io = end_swap_bio_write; submit_bio(bio); } diff --git a/mm/swap_state.c b/mm/swap_state.c index 87790369dd9aa3..2475ba29126dca 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -551,6 +551,48 @@ struct folio *__swap_cache_alloc_folio(swp_entry_t targ_entry, gfp_t gfp, return ret; } +/** + * swap_writeback_dropbehind_folio - drop a dropbehind swap cache folio + * @folio: the off-LRU folio whose writeback has completed + * + * Context: task context, with the reference taken by folio_end_writeback() + * donated to us. + */ +void swap_writeback_dropbehind_folio(struct folio *folio) +{ + struct mem_cgroup *memcg; + + folio_lock(folio); + + /* The folio was allocated off the LRU and nothing re-adds it here. */ + VM_WARN_ON_ONCE_FOLIO(folio_test_lru(folio), folio); + + rcu_read_lock(); + memcg = folio_memcg(folio); + if (!mem_cgroup_tryget(memcg)) + memcg = NULL; + rcu_read_unlock(); + + /* + * Gate remove_mapping_set_shadow() on folio_test_swapcache(): a racing + * swapin may have freed the swap slot (folio_free_swap()) and dropped the + * folio from the cache, and it must not run on a non-swapcache folio (it + * would trip __remove_mapping()'s mapping == folio_mapping() check). + */ + if (!folio_test_swapcache(folio) || folio_test_writeback(folio) || + !remove_mapping_set_shadow(swap_address_space(folio->swap), folio, + memcg)) { + /* Raced: the folio is now owned by the swapin; put it back. */ + folio_clear_dropbehind(folio); + folio_add_lru(folio); + } + + mem_cgroup_put(memcg); + + folio_unlock(folio); + folio_put(folio); +} + /* * If we are the only user, then try to free up the swap cache. * diff --git a/mm/vmscan.c b/mm/vmscan.c index f2e641e9cf7de9..e200ce3eb056b1 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -854,6 +854,22 @@ static int __remove_mapping(struct address_space *mapping, struct folio *folio, return 0; } +static long __remove_mapping_unfreeze(struct address_space *mapping, + struct folio *folio, bool reclaimed, + struct mem_cgroup *target_memcg) +{ + if (__remove_mapping(mapping, folio, reclaimed, target_memcg)) { + /* + * Unfreezing the refcount with 1 effectively + * drops the pagecache ref for us without requiring another + * atomic operation. + */ + folio_ref_unfreeze(folio, 1); + return folio_nr_pages(folio); + } + return 0; +} + /** * remove_mapping() - Attempt to remove a folio from its mapping. * @mapping: The address space. @@ -868,16 +884,29 @@ static int __remove_mapping(struct address_space *mapping, struct folio *folio, */ long remove_mapping(struct address_space *mapping, struct folio *folio) { - if (__remove_mapping(mapping, folio, false, NULL)) { - /* - * Unfreezing the refcount with 1 effectively - * drops the pagecache ref for us without requiring another - * atomic operation. - */ - folio_ref_unfreeze(folio, 1); - return folio_nr_pages(folio); - } - return 0; + return __remove_mapping_unfreeze(mapping, folio, false, NULL); +} + +/** + * remove_mapping_set_shadow() - Remove a folio and record an eviction shadow. + * @mapping: The address space. + * @folio: The folio to remove. + * @target_memcg: The memcg to charge the eviction shadow to; the caller must + * keep it alive across the call. + * + * Like remove_mapping(), but stores a workingset eviction shadow the way page + * reclaim does, so that a later refault can be detected and the folio + * re-activated. + * Return: The number of pages removed from the mapping. 0 if the folio + * could not be removed. + * Context: The caller should have a single refcount on the folio and + * hold its lock. + */ +long remove_mapping_set_shadow(struct address_space *mapping, + struct folio *folio, + struct mem_cgroup *target_memcg) +{ + return __remove_mapping_unfreeze(mapping, folio, true, target_memcg); } /** From 88d622902eb1cb6bceddbcac3395ecb6a819a4f5 Mon Sep 17 00:00:00 2001 From: Alexandre Ghiti Date: Mon, 21 Sep 2026 17:13:04 +0200 Subject: [PATCH 0741/1012] mm: zswap: drop cold writeback folios via swap dropbehind zswap writeback decompresses an entry into a fresh swap cache folio and writes it back. The folio is cold by construction, yet it is left on the LRU for reclaim to find and free later, wasting a reclaim scan and keeping cold memory resident longer than necessary. Allocate the folio off the LRU and mark it PG_dropbehind so the swap dropbehind path frees it from the swap cache once writeback completes. __swap_cache_alloc_folio() evaluates a refault on the new folio, and workingset_refault() sets PG_active when it looks recent. Until now folio_add_lru() consumed that flag and __page_cache_release() cleared it once the folio left the LRU. This folio never reaches the LRU, so nothing would clear PG_active and the folio would be freed with a PAGE_FLAGS_CHECK_AT_FREE flag set, tripping bad_page() under CONFIG_DEBUG_VM. Clear it after allocation. That is a workaround: the refault should not be evaluated on a writeback buffer at all. A fix for that is on the mailing list [1]. Link: https://lore.kernel.org/20260921151306.625134-4-alex@ghiti.fr Link: https://lore.kernel.org/linux-mm/20260911092012.92399-1-alex@ghiti.fr/ [1] Signed-off-by: Alexandre Ghiti Signed-off-by: Andrew Morton Suggested-by: Johannes Weiner Suggested-by: Nhat Pham Reviewed-by: Nhat Pham Reviewed-by: Kunwu Chan Cc: Al Viro Cc: Axel Rasmussen Cc: Baoquan He Cc: Barry Song Cc: Chengming Zhou Cc: Chis Li Cc: Christian Brauner (Amutable) Cc: "David Hildenbrand (arm)" Cc: Jan Kara Cc: Kairui Song Cc: Kemeng Shi Cc: Lorenzo Stoakes (ARM) Cc: Matthew Wilcox Cc: Michal Hocko Cc: Qi Zheng Cc: Shakeel Butt Cc: Tal Zussman Cc: Usama Arif Cc: Wei Xu Cc: Yosry Ahmed Cc: Youngjun Park Cc: Yuanchu Xie --- mm/zswap.c | 33 +++++++++++++++++++++++---------- 1 file changed, 23 insertions(+), 10 deletions(-) diff --git a/mm/zswap.c b/mm/zswap.c index c79cca61abf933..ae19e301fced7e 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1032,7 +1032,8 @@ static int zswap_writeback_entry(struct zswap_entry *entry, */ if (IS_ERR(folio)) return PTR_ERR(folio); - folio_add_lru(folio); + + folio_clear_active(folio); /* * folio is locked, and the swapcache is now secured against @@ -1046,12 +1047,12 @@ static int zswap_writeback_entry(struct zswap_entry *entry, tree = swap_zswap_tree(swpentry); if (entry != xa_load(tree, offset)) { ret = -ENOMEM; - goto out; + goto err; } if (!zswap_decompress(entry, folio)) { ret = -EIO; - goto out; + goto err; } xa_erase(tree, offset); @@ -1065,18 +1066,30 @@ static int zswap_writeback_entry(struct zswap_entry *entry, /* folio is up to date */ folio_mark_uptodate(folio); - /* move it to the tail of the inactive list after end_writeback */ - folio_set_reclaim(folio); + folio_set_dropbehind(folio); + + /* + * Drop our reference before starting writeback so the swap cache holds + * the only one: the drop in folio_end_writeback() needs that for + * remove_mapping_set_shadow() to succeed, otherwise the folio is + * handed back to reclaim instead. + * + * Nothing can free the folio in the meantime: we hold the folio lock + * until writeback starts, PG_writeback then blocks swap cache removal, + * and folio_end_writeback() takes its own reference before clearing + * PG_writeback and donates it to the drop. + */ + folio_put(folio); /* start writeback */ __swap_writeout(&ctx, folio); swap_write_submit(&ctx); -out: - if (ret) { - swap_cache_del_folio(folio); - folio_unlock(folio); - } + return 0; + +err: + swap_cache_del_folio(folio); + folio_unlock(folio); folio_put(folio); return ret; } From cb99344b7229d6c6871b049a64eb67cd2bdeebe8 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 20:47:41 +0100 Subject: [PATCH 0742/1012] mm/vma: const-ify vma_assert_stabilised() and associated functions Patch series "mm: implement and use vma_has_anon_rmap(), silence KCSAN". Provide a function to abstract the common task of checking whether a VMA has an anonymous reverse mapping associated with it. In the first patch, const-ify vma_assert_stabilised() and related functions so that vma_has_anon_rmap() can reference a const vma pointer. In the second patch, introduce vma_has_anon_rmap(). Finally in the third patch, update comments referencing anon_vma to instead reference the anon rmap to abstract this conceptually to avoid confusion. There are still other functions which reference anon_vma directly, those can be addressed in a follow up. This patch (of 3): The vma pointers are not modified in any of these functions so make it official by const-ify them. Propagate this throughout the call stack. No functional change intended. Link: https://lore.kernel.org/20260917-vma-is-faulted-v3-0-5c22314a72e7@kernel.org Link: https://lore.kernel.org/20260917-vma-is-faulted-v3-1-5c22314a72e7@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reviewed-by: Zi Yan Acked-by: David Hildenbrand (Arm) Reviewed-by: Pedro Falcato Reviewed-by: Kiryl Shutsemau (Meta) Reviewed-by: Lance Yang Cc: Alistair Popple Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Byungchul Park Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Gregory Price Cc: Guilherme Giacomo Simoes Cc: Harry Yoo Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Brost Cc: Michal Hocko Cc: Mike Rapoport Cc: Muchun Song Cc: Nhat Pham Cc: Oscar Salvador Cc: Peter Xu Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Shakeel Butt Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: xu xin --- include/linux/mmap_lock.h | 12 ++++++------ tools/testing/vma/include/dup.h | 2 +- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/include/linux/mmap_lock.h b/include/linux/mmap_lock.h index 28e3696ce9ff37..e5553f4a414cd1 100644 --- a/include/linux/mmap_lock.h +++ b/include/linux/mmap_lock.h @@ -273,7 +273,7 @@ static inline void vma_end_read(struct vm_area_struct *vma) vma_refcount_put(vma); } -static inline unsigned int __vma_raw_mm_seqnum(struct vm_area_struct *vma) +static inline unsigned int __vma_raw_mm_seqnum(const struct vm_area_struct *vma) { const struct mm_struct *mm = vma->vm_mm; @@ -288,7 +288,7 @@ static inline unsigned int __vma_raw_mm_seqnum(struct vm_area_struct *vma) * * Returns true if write-locked, otherwise false. */ -static inline bool __is_vma_write_locked(struct vm_area_struct *vma) +static inline bool __is_vma_write_locked(const struct vm_area_struct *vma) { /* * current task is holding mmap_write_lock, both vma->vm_lock_seq and @@ -344,7 +344,7 @@ int vma_start_write_killable(struct vm_area_struct *vma) * vma_assert_write_locked() - assert that @vma holds a VMA write lock. * @vma: The VMA to assert. */ -static inline void vma_assert_write_locked(struct vm_area_struct *vma) +static inline void vma_assert_write_locked(const struct vm_area_struct *vma) { if (!IS_ENABLED(CONFIG_MMU)) { mmap_assert_write_locked(vma->vm_mm); @@ -359,7 +359,7 @@ static inline void vma_assert_write_locked(struct vm_area_struct *vma) * lock and is not detached. * @vma: The VMA to assert. */ -static inline void vma_assert_locked(struct vm_area_struct *vma) +static inline void vma_assert_locked(const struct vm_area_struct *vma) { unsigned int refcnt; @@ -410,7 +410,7 @@ static inline void vma_assert_locked(struct vm_area_struct *vma) * With lockdep disabled we may sometimes race with other threads acquiring the * mmap read lock simultaneous with our VMA read lock. */ -static inline void vma_assert_stabilised(struct vm_area_struct *vma) +static inline void vma_assert_stabilised(const struct vm_area_struct *vma) { /* * If another thread owns an mmap lock, it may go away at any time, and @@ -445,7 +445,7 @@ static inline void vma_assert_stabilised(struct vm_area_struct *vma) vma_assert_locked(vma); } -static inline bool vma_is_attached(struct vm_area_struct *vma) +static inline bool vma_is_attached(const struct vm_area_struct *vma) { return refcount_read(&vma->vm_refcnt); } diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h index c21f67decab58d..a4e3d30b2fc171 100644 --- a/tools/testing/vma/include/dup.h +++ b/tools/testing/vma/include/dup.h @@ -1178,7 +1178,7 @@ static inline struct vm_area_struct *vma_next(struct vma_iterator *vmi) return mas_find(&vmi->mas, ULONG_MAX); } -static inline bool vma_is_attached(struct vm_area_struct *vma) +static inline bool vma_is_attached(const struct vm_area_struct *vma) { return refcount_read(&vma->vm_refcnt); } From 99354e29c4f0ab96db1c2765b29df84a52b62222 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 20:47:42 +0100 Subject: [PATCH 0743/1012] mm: implement and use vma_has_anon_rmap(), silence KCSAN Provide a function to abstract the common task of checking whether a VMA has an anonymous reverse mapping associated with it. If the VMA is attached, a VMA or mmap lock must be held when calling this function. For an attached, anonymous, VMA: Transition | VMA/mmap Lock state -----------------------------|------------------------------------------- No anon rmap to anon rmap | Write lock/read lock + mm->page_table_lock Anon rmap to no anon rmap | Write lock vma_has_anon_rmap() never provides a false positive (the lock precludes it), but if only a read lock is held, a negative result must be re-checked with mm->page_table_lock held. A VMA obtains an anonymous reverse mapping when first faulted or forked and it is removed when it is freed. Detached VMAs cannot be concurrently manipulated as they are removed from the maple tree so require no guarantees. Use data_race() to silence KCSAN about non-existent data races between concurrent vma->anon_vma read/write on optimistic fault tests. Update the core VMA merge/split, rmap, mremap, KSM, fork, khugepaged and fault preparation callers which test vma->anon_vma directly to use vma_has_anon_rmap() instead. Finally, update comments that reference anon_vma to reference the anon rmap instead. Since the lockless read in reusable_anon_vma() is doing more than checking whether the VMA has anon rmap - it is returning the anon_vma to be used on fault - do not alter it. There is one odd one out - file_backed_vma_is_retractable() - which holds neither a VMA nor mmap lock and is stabilised by the file rmap lock only, so simply add a comment to explain why it's necessary. Link: https://lore.kernel.org/20260917-vma-is-faulted-v3-2-5c22314a72e7@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Reported-by: Guilherme Giacomo Simoes Closes: https://lore.kernel.org/all/20260829100034.423064-1-trintaeoitogc@gmail.com/ Closes: https://lore.kernel.org/all/20260909115723.528501-1-trintaeoitogc@gmail.com/ Reviewed-by: Pedro Falcato Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Reviewed-by: Lance Yang Cc: Alistair Popple Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Byungchul Park Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Gregory Price Cc: Harry Yoo Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Kairui Song Cc: Kemeng Shi Cc: Liam R. Howlett Cc: Matthew Brost Cc: Michal Hocko Cc: Mike Rapoport Cc: Muchun Song Cc: Nhat Pham Cc: Oscar Salvador Cc: Peter Xu Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Shakeel Butt Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: xu xin Cc: Zi Yan --- mm/huge_memory.c | 4 ++-- mm/internal.h | 2 +- mm/khugepaged.c | 5 ++++- mm/ksm.c | 10 +++++----- mm/madvise.c | 4 ++-- mm/memory.c | 4 ++-- mm/mprotect.c | 2 +- mm/mremap.c | 4 ++-- mm/rmap.c | 22 +++++++++++----------- mm/swapfile.c | 2 +- mm/userfaultfd.c | 2 +- mm/vma.c | 29 +++++++++++++++-------------- mm/vma.h | 29 ++++++++++++++++++++++++++++- tools/testing/vma/include/stubs.h | 4 ++++ 14 files changed, 79 insertions(+), 44 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index d2e990da86b0d9..e0e252cfbf9580 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -259,7 +259,7 @@ unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma, * Allow page fault since anon_vma may be not initialized until * the first page fault. */ - if (!vma->anon_vma) + if (!vma_has_anon_rmap(vma)) return (smaps || in_pf) ? orders : 0; return orders; @@ -2171,7 +2171,7 @@ vm_fault_t do_huge_pmd_wp_page(struct vm_fault *vmf) pmd_t orig_pmd = vmf->orig_pmd; vmf->ptl = pmd_lockptr(vma->vm_mm, vmf->pmd); - VM_BUG_ON_VMA(!vma->anon_vma, vma); + VM_BUG_ON_VMA(!vma_has_anon_rmap(vma), vma); if (is_huge_zero_pmd(orig_pmd)) { vm_fault_t ret = do_huge_zero_wp_pmd(vmf); diff --git a/mm/internal.h b/mm/internal.h index 3b9fdb826162df..0434dfcfc36f14 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -339,7 +339,7 @@ void unlink_anon_vmas(struct vm_area_struct *vma); static inline int anon_vma_prepare(struct vm_area_struct *vma) { - if (likely(vma->anon_vma)) + if (likely(vma_has_anon_rmap(vma))) return 0; return __anon_vma_prepare(vma); diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 8a5c7f38096ef7..3e8dbd2dd46d8d 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1045,7 +1045,7 @@ enum scan_result collapse_vma_revalidate(struct mm_struct *mm, unsigned long add * thp_vma_allowable_orders() may return true for qualified file * vmas. */ - if (expect_anon && (!(*vmap)->anon_vma || !vma_is_anonymous(*vmap))) + if (expect_anon && (!vma_has_anon_rmap(vma) || !vma_is_anonymous(vma))) return SCAN_PAGE_ANON; return SCAN_SUCCEED; } @@ -2079,6 +2079,9 @@ static bool file_backed_vma_is_retractable(struct vm_area_struct *vma) * Check vma->anon_vma to exclude MAP_PRIVATE mappings that * got written to. These VMAs are likely not worth removing * page tables from, as PMD-mapping is likely to be split later. + * + * Can't use vma_has_anon_rmap() here as the VMA may be stabilised + * by the file rmap lock. */ if (READ_ONCE(vma->anon_vma)) return false; diff --git a/mm/ksm.c b/mm/ksm.c index f80372bfd4b2fc..bcf5799bfe371f 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -775,7 +775,7 @@ static struct vm_area_struct *find_mergeable_vma(struct mm_struct *mm, if (ksm_test_exit(mm)) return NULL; vma = vma_lookup(mm, addr); - if (!vma || !(vma->vm_flags & VM_MERGEABLE) || !vma->anon_vma) + if (!vma || !(vma->vm_flags & VM_MERGEABLE) || !vma_has_anon_rmap(vma)) return NULL; return vma; } @@ -1239,7 +1239,7 @@ static int unmerge_and_remove_all_rmap_items(void) goto mm_exiting; for_each_vma(vmi, vma) { - if (!(vma->vm_flags & VM_MERGEABLE) || !vma->anon_vma) + if (!(vma->vm_flags & VM_MERGEABLE) || !vma_has_anon_rmap(vma)) continue; err = break_ksm(vma, vma->vm_start, vma->vm_end, false); if (err) @@ -2689,7 +2689,7 @@ static struct ksm_rmap_item *scan_get_next_rmap_item(struct page **page) continue; if (ksm_scan.address < vma->vm_start) ksm_scan.address = vma->vm_start; - if (!vma->anon_vma) + if (!vma_has_anon_rmap(vma)) ksm_scan.address = vma->vm_end; while (ksm_scan.address < vma->vm_end) { @@ -2880,7 +2880,7 @@ static int __ksm_del_vma(struct vm_area_struct *vma) if (!(vma->vm_flags & VM_MERGEABLE)) return 0; - if (vma->anon_vma) { + if (vma_has_anon_rmap(vma)) { err = break_ksm(vma, vma->vm_start, vma->vm_end, true); if (err) return err; @@ -3032,7 +3032,7 @@ int ksm_madvise(struct vm_area_struct *vma, unsigned long start, if (!(*vm_flags & VM_MERGEABLE)) return 0; /* just ignore the advice */ - if (vma->anon_vma) { + if (vma_has_anon_rmap(vma)) { err = break_ksm(vma, start, end, true); if (err) return err; diff --git a/mm/madvise.c b/mm/madvise.c index 32a28b9bb6880a..010ad3d47f5bb3 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -1331,7 +1331,7 @@ static long madvise_guard_install(struct madvise_behavior *madv_behavior) * as part of the VMA lock logic. */ if (vma_is_anonymous(vma)) { - VM_WARN_ON_ONCE(!vma->anon_vma && + VM_WARN_ON_ONCE(!vma_has_anon_rmap(vma) && madv_behavior->lock_mode != MADVISE_MMAP_READ_LOCK); err = anon_vma_prepare(vma); @@ -1793,7 +1793,7 @@ static bool is_vma_lock_sufficient(struct vm_area_struct *vma, * check overly paranoid which is safe. */ if (vma_is_anonymous(vma) && - prepares_anon_vma(madv_behavior->behavior) && !vma->anon_vma) + prepares_anon_vma(madv_behavior->behavior) && !vma_has_anon_rmap(vma)) return false; return true; diff --git a/mm/memory.c b/mm/memory.c index 338fce99e71197..544a9e5068f09d 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -1536,7 +1536,7 @@ vma_needs_copy(struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma) * The presence of an anon_vma indicates an anonymous VMA has page * tables which naturally cannot be reconstituted on page fault. */ - if (src_vma->anon_vma) + if (vma_has_anon_rmap(src_vma)) return true; /* @@ -4011,7 +4011,7 @@ vm_fault_t __vmf_anon_prepare(struct vm_fault *vmf) struct vm_area_struct *vma = vmf->vma; vm_fault_t ret = 0; - if (likely(vma->anon_vma)) + if (likely(vma_has_anon_rmap(vma))) return 0; if (vmf->flags & FAULT_FLAG_VMA_LOCK) { if (!mmap_read_trylock(vma->vm_mm)) diff --git a/mm/mprotect.c b/mm/mprotect.c index a1b6d29bf03908..4b1296f0d502a2 100644 --- a/mm/mprotect.c +++ b/mm/mprotect.c @@ -816,7 +816,7 @@ mprotect_fixup(struct vma_iterator *vmi, struct mmu_gather *tlb, vma_flags_set(&new_vma_flags, VMA_ACCOUNT_BIT); } } else if (vma_flags_test(&old_vma_flags, VMA_ACCOUNT_BIT) && - vma_is_anonymous(vma) && !vma->anon_vma) { + vma_is_anonymous(vma) && !vma_has_anon_rmap(vma)) { vma_flags_clear(&new_vma_flags, VMA_ACCOUNT_BIT); } diff --git a/mm/mremap.c b/mm/mremap.c index 49dc25d8a34dc2..6be5867bb45566 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -144,13 +144,13 @@ static void take_rmap_locks(struct vm_area_struct *vma) { if (vma->vm_file) i_mmap_lock_write(vma->vm_file->f_mapping); - if (vma->anon_vma) + if (vma_has_anon_rmap(vma)) anon_vma_lock_write(vma->anon_vma); } static void drop_rmap_locks(struct vm_area_struct *vma) { - if (vma->anon_vma) + if (vma_has_anon_rmap(vma)) anon_vma_unlock_write(vma->anon_vma); if (vma->vm_file) i_mmap_unlock_write(vma->vm_file->f_mapping); diff --git a/mm/rmap.c b/mm/rmap.c index 6661bc11ce658b..fbd66a2823b7b3 100644 --- a/mm/rmap.c +++ b/mm/rmap.c @@ -208,7 +208,7 @@ int __anon_vma_prepare(struct vm_area_struct *vma) anon_vma_lock_write(anon_vma); /* page_table_lock to protect against threads */ spin_lock(&mm->page_table_lock); - if (likely(!vma->anon_vma)) { + if (likely(!vma_has_anon_rmap(vma))) { /* * Make anon_vma fields visible before anon_vma is published. * Paired with an address dependency in reusable_anon_vma(). @@ -246,21 +246,21 @@ static void check_anon_vma_clone(struct vm_area_struct *dst, VM_WARN_ON_ONCE(operation != VMA_OP_FORK && dst->vm_mm != src->vm_mm); /* If we have anything to do src->anon_vma must be provided. */ - VM_WARN_ON_ONCE(!src->anon_vma && !list_empty(&src->anon_vma_chain)); - VM_WARN_ON_ONCE(!src->anon_vma && dst->anon_vma); + VM_WARN_ON_ONCE(!vma_has_anon_rmap(src) && !list_empty(&src->anon_vma_chain)); + VM_WARN_ON_ONCE(!vma_has_anon_rmap(src) && vma_has_anon_rmap(dst)); /* We are establishing a new anon_vma_chain. */ VM_WARN_ON_ONCE(!list_empty(&dst->anon_vma_chain)); /* * On fork, dst->anon_vma is set NULL (temporarily). Otherwise, anon_vma * must be the same across dst and src. */ - VM_WARN_ON_ONCE(dst->anon_vma && dst->anon_vma != src->anon_vma); + VM_WARN_ON_ONCE(vma_has_anon_rmap(dst) && dst->anon_vma != src->anon_vma); /* * Essentially equivalent to above - if not a no-op, we should expect * dst->anon_vma to be set for everything except a fork. */ - VM_WARN_ON_ONCE(operation != VMA_OP_FORK && src->anon_vma && - !dst->anon_vma); + VM_WARN_ON_ONCE(operation != VMA_OP_FORK && vma_has_anon_rmap(src) && + !vma_has_anon_rmap(dst)); /* For the anon_vma to be compatible, it can only be singular. */ VM_WARN_ON_ONCE(operation == VMA_OP_MERGE_UNFAULTED && !list_is_singular(&src->anon_vma_chain)); @@ -273,7 +273,7 @@ static void maybe_reuse_anon_vma(struct vm_area_struct *dst, struct anon_vma *anon_vma) { /* If already populated, nothing to do.*/ - if (dst->anon_vma) + if (vma_has_anon_rmap(dst)) return; /* @@ -327,7 +327,7 @@ int anon_vma_clone(struct vm_area_struct *dst, struct vm_area_struct *src, check_anon_vma_clone(dst, src, operation); - if (!active_anon_vma) + if (!vma_has_anon_rmap(src)) return 0; /* @@ -384,7 +384,7 @@ int anon_vma_fork(struct vm_area_struct *vma, struct vm_area_struct *pvma) int rc; /* Don't bother if the parent process has no anon_vma here. */ - if (!pvma->anon_vma) + if (!vma_has_anon_rmap(pvma)) return 0; /* Drop inherited anon_vma, we'll reuse existing or allocate new. */ @@ -405,7 +405,7 @@ int anon_vma_fork(struct vm_area_struct *vma, struct vm_area_struct *pvma) */ rc = anon_vma_clone(vma, pvma, VMA_OP_FORK); /* An error arose or an existing anon_vma was reused, all done then. */ - if (rc || vma->anon_vma) { + if (rc || vma_has_anon_rmap(vma)) { put_anon_vma(anon_vma); anon_vma_chain_free(avc); return rc; @@ -864,7 +864,7 @@ unsigned long page_address_in_vma(const struct folio *folio, * Note: swapoff's unuse_vma() is more efficient with this * check, and needs it to match anon_vma when KSM is active. */ - if (!vma->anon_vma || !anon_vma || + if (!vma_has_anon_rmap(vma) || !anon_vma || vma->anon_vma->root != anon_vma->root) return -EFAULT; /* KSM folios don't reach here because of the !anon_vma check */ diff --git a/mm/swapfile.c b/mm/swapfile.c index 0901cb8fa7291c..6187c02ec5ec73 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -2707,7 +2707,7 @@ static int unuse_mm(struct mm_struct *mm, unsigned int type) if (check_stable_address_space(mm)) goto unlock; for_each_vma(vmi, vma) { - if (vma->anon_vma && !vma_is_hugetlb(vma)) { + if (vma_has_anon_rmap(vma) && !vma_is_hugetlb(vma)) { ret = unuse_vma(vma, type); if (ret) break; diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index 17ecbb0ceddf11..666cc18902639b 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -145,7 +145,7 @@ static struct vm_area_struct *uffd_lock_vma(struct mm_struct *mm, * We know we're going to need to use anon_vma, so check * that early. */ - if (!(vma->vm_flags & VM_SHARED) && unlikely(!vma->anon_vma)) + if (!(vma->vm_flags & VM_SHARED) && unlikely(!vma_has_anon_rmap(vma))) vma_end_read(vma); else return vma; diff --git a/mm/vma.c b/mm/vma.c index 0862d0861d0128..be3840d9ba4d8a 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -100,7 +100,8 @@ static bool vma_is_fork_child(struct vm_area_struct *vma) * parents. This can improve scalability caused by the anon_vma root * lock. */ - return vma && vma->anon_vma && !list_is_singular(&vma->anon_vma_chain); + return vma && vma_has_anon_rmap(vma) && + !list_is_singular(&vma->anon_vma_chain); } static inline bool is_mergeable_vma(struct vma_merge_struct *vmg, bool merge_next) @@ -140,7 +141,7 @@ static bool is_mergeable_anon_vma(struct vma_merge_struct *vmg, bool merge_next) VM_WARN_ON(src && src_anon != src->anon_vma); /* Case 1 - we will dup_anon_vma() from src into tgt. */ - if (!tgt_anon && src_anon) { + if (!vma_has_anon_rmap(tgt) && src_anon) { struct vm_area_struct *copied_from = vmg->copied_from; if (vma_is_fork_child(src)) @@ -151,7 +152,7 @@ static bool is_mergeable_anon_vma(struct vma_merge_struct *vmg, bool merge_next) return true; } /* Case 2 - we will simply use tgt's anon_vma. */ - if (tgt_anon && !src_anon) + if (vma_has_anon_rmap(tgt) && !src_anon) return !vma_is_fork_child(tgt); /* Case 3 - the anon_vma's are already shared. */ return src_anon == tgt_anon; @@ -190,10 +191,10 @@ static void init_multi_vma_prep(struct vma_prepare *vp, adjust = NULL; vp->adj_next = adjust; - if (!vp->anon_vma && adjust) + if (!vma_has_anon_rmap(vma) && adjust) vp->anon_vma = adjust->anon_vma; - VM_WARN_ON(vp->anon_vma && adjust && adjust->anon_vma && + VM_WARN_ON(vma_has_anon_rmap(vma) && adjust && vma_has_anon_rmap(adjust) && vp->anon_vma != adjust->anon_vma); vp->file = vma->vm_file; @@ -430,7 +431,7 @@ static void vma_complete(struct vma_prepare *vp, struct vma_iterator *vmi, vp->remove->vm_end); fput(vp->file); } - if (vp->remove->anon_vma) + if (vma_has_anon_rmap(vp->remove)) unlink_anon_vmas(vp->remove); mm->map_count--; mpol_put(vma_policy(vp->remove)); @@ -500,7 +501,7 @@ static bool can_vma_merge_right(struct vma_merge_struct *vmg, * We therefore check this in addition to mergeability to either side. */ prev = vmg->prev; - return !prev->anon_vma || !next->anon_vma || + return !vma_has_anon_rmap(prev) || !vma_has_anon_rmap(next) || prev->anon_vma == next->anon_vma; } @@ -670,7 +671,7 @@ static int dup_anon_vma(struct vm_area_struct *dst, * that is it is unfaulted, we need to ensure that the newly merged * range is referenced by the anon_vma's of the source. */ - if (src->anon_vma && !dst->anon_vma) { + if (vma_has_anon_rmap(src) && !vma_has_anon_rmap(dst)) { int ret; vma_assert_write_locked(dst); @@ -720,7 +721,7 @@ void validate_mm(struct mm_struct *mm) } #ifdef CONFIG_DEBUG_VM_RB - if (anon_vma) { + if (vma_has_anon_rmap(vma)) { anon_vma_lock_read(anon_vma); list_for_each_entry(avc, &vma->anon_vma_chain, same_vma) anon_rmap_tree_verify(avc); @@ -1020,7 +1021,7 @@ static __must_check struct vm_area_struct *vma_merge_existing_range( * simply a case of, if prev has no anon_vma object, which of * next or middle contains the anon_vma we must duplicate. */ - err = dup_anon_vma(prev, next->anon_vma ? next : middle, + err = dup_anon_vma(prev, vma_has_anon_rmap(next) ? next : middle, &anon_dup); } else if (merge_left) { /* @@ -1960,7 +1961,7 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, * If a vma has not yet been faulted, update its anonymous pgoff to * match the new location to increase its chance of merging. */ - if (!vma->anon_vma) { + if (!vma_has_anon_rmap(vma)) { anon_pgoff = addr >> PAGE_SHIFT; if (vma_is_anonymous(vma)) { @@ -2387,7 +2388,7 @@ int mm_take_all_locks(struct mm_struct *mm) for_each_vma(vmi, vma) { if (signal_pending(current)) goto out_unlock; - if (vma->anon_vma) + if (vma_has_anon_rmap(vma)) list_for_each_entry(avc, &vma->anon_vma_chain, same_vma) vm_lock_anon_vma(mm, avc->anon_vma); } @@ -2449,7 +2450,7 @@ void mm_drop_all_locks(struct mm_struct *mm) BUG_ON(!mutex_is_locked(&mm_all_locks_mutex)); for_each_vma(vmi, vma) { - if (vma->anon_vma) + if (vma_has_anon_rmap(vma)) list_for_each_entry(avc, &vma->anon_vma_chain, same_vma) vm_unlock_anon_vma(avc->anon_vma); if (vma->vm_file && vma->vm_file->f_mapping) @@ -3597,7 +3598,7 @@ int insert_vm_struct(struct mm_struct *mm, struct vm_area_struct *vma) * Similarly in do_mmap and in do_brk_flags. */ if (vma_is_anonymous(vma)) { - WARN_ON_ONCE(vma->anon_vma); + WARN_ON_ONCE(vma_has_anon_rmap(vma)); vma_set_pgoff(vma, vma->vm_start >> PAGE_SHIFT); } vma_set_anon_pgoff(vma, vma->vm_start >> PAGE_SHIFT); diff --git a/mm/vma.h b/mm/vma.h index b9b99fa02a861d..7a683272c0a82a 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -255,6 +255,33 @@ static inline pgoff_t vmg_end_pgoff(const struct vma_merge_struct *vmg) return vmg_start_pgoff(vmg) + vmg_pages(vmg); } +/** + * vma_has_anon_rmap() - does @vma possess an anonymous reverse mapping? + * @vma: The VMA to be checked. + * + * If the VMA is attached, a VMA or mmap lock must be held. + * + * This state is only possible for CoW mappings, see the comment for + * vma_flags_is_cow_mapping() for details. + * + * Importantly, a VMA which possesses an anonymous rmap may map anonymous + * folios. + * + * This function will not result in a false positive. + * + * However, if only a read lock is held, it may give a false negative, in which + * case it should be re-checked with mm->page_table_lock held. + * + * Returns: true if @vma has an anonymous reverse mapping, otherwise false. + */ +static inline bool vma_has_anon_rmap(const struct vm_area_struct *vma) +{ + if (vma_is_attached(vma)) + vma_assert_stabilised(vma); + /* KCSAN gets confused about the optimistic check. Silence it. */ + return data_race(vma->anon_vma); +} + static inline void assert_sane_pgoff(struct vm_area_struct *vma, pgoff_t pgoff) { /* nommu doesn't set a virtual pgoff for anon VMAs. */ @@ -268,7 +295,7 @@ static inline void assert_sane_pgoff(struct vm_area_struct *vma, pgoff_t pgoff) if (!vma_is_anonymous(vma)) return; /* If faulted in, could have been remapped. */ - if (vma->anon_vma) + if (vma_has_anon_rmap(vma)) return; /* OK this is really an anon VMA - expect virtual page offset. */ VM_WARN_ON_ONCE(pgoff != vma->vm_start >> PAGE_SHIFT); diff --git a/tools/testing/vma/include/stubs.h b/tools/testing/vma/include/stubs.h index 48d1dc53df42cb..e4acc6f1fe7bab 100644 --- a/tools/testing/vma/include/stubs.h +++ b/tools/testing/vma/include/stubs.h @@ -302,6 +302,10 @@ static inline void vma_assert_write_locked(struct vm_area_struct *vma) { } +static inline void vma_assert_stabilised(const struct vm_area_struct *vma) +{ +} + static inline void ksm_add_vma(struct vm_area_struct *vma) { } From 9f68b90757943e064982b4a3bb6bbd54994c720b Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Thu, 17 Sep 2026 20:47:43 +0100 Subject: [PATCH 0744/1012] mm: update comments to refer to anon rmap rather than anon_vma Now that vma_has_anon_rmap() abstracts whether a VMA has an anonymous reverse mapping, remove references to anon_vma and instead reference the anon rmap. The anon_vma is an implementation detail and should be treated as such. Do not update mm/rmap.c which implements the anon_vma mechanism as it is reasonable to directly reference it there. No functional change intended. Link: https://lore.kernel.org/20260917-vma-is-faulted-v3-3-5c22314a72e7@kernel.org Signed-off-by: Lorenzo Stoakes (ARM) Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Pedro Falcato Reviewed-by: Lance Yang Cc: Alistair Popple Cc: Baolin Wang Cc: Baoquan He Cc: Barry Song Cc: Byungchul Park Cc: Chengming Zhou Cc: Chris Li Cc: Dev Jain Cc: Gregory Price Cc: Guilherme Giacomo Simoes Cc: Harry Yoo Cc: "Huang, Ying" Cc: Jann Horn Cc: Joshua Hahn Cc: Kairui Song Cc: Kemeng Shi Cc: Kiryl Shutsemau Cc: Liam R. Howlett Cc: Matthew Brost Cc: Michal Hocko Cc: Mike Rapoport Cc: Muchun Song Cc: Nhat Pham Cc: Oscar Salvador Cc: Peter Xu Cc: Rakie Kim Cc: Rik van Riel Cc: Ryan Roberts Cc: Shakeel Butt Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: xu xin Cc: Zi Yan --- mm/huge_memory.c | 8 ++-- mm/hugetlb.c | 2 +- mm/khugepaged.c | 12 +++--- mm/ksm.c | 6 +-- mm/madvise.c | 6 +-- mm/memory.c | 12 +++--- mm/migrate.c | 12 +++--- mm/mmap.c | 6 +-- mm/mprotect.c | 4 +- mm/mremap.c | 6 +-- mm/pgtable-generic.c | 2 +- mm/userfaultfd.c | 10 ++--- mm/vma.c | 98 ++++++++++++++++++++++---------------------- 13 files changed, 92 insertions(+), 92 deletions(-) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index e0e252cfbf9580..ba5e20bbfd3bc3 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -254,10 +254,10 @@ unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma, /* * THPeligible bit of smaps should show 1 for proper VMAs even - * though anon_vma is not initialized yet. + * though they don't have an anon rmap yet. * - * Allow page fault since anon_vma may be not initialized until - * the first page fault. + * Allow page fault since the VMA may not have an anon rmap until the + * first page fault. */ if (!vma_has_anon_rmap(vma)) return (smaps || in_pf) ? orders : 0; @@ -4370,7 +4370,7 @@ static int __folio_split(struct folio *folio, unsigned int new_order, * THP pages in the middle of migration, due to allocation issues on either * side. * - * anon_vma_lock is not required to be held, mmap_read_lock() or + * The anon rmap lock is not required to be held, mmap_read_lock() or * mmap_write_lock() should be held. @folio is expected to be locked by the * caller. device-private and non device-private folios are supported along * with folios that are in the swapcache. @folio should also be unmapped and diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 9d05fecf21326d..da980377d35339 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -5677,7 +5677,7 @@ static vm_fault_t hugetlb_wp(struct vm_fault *vmf) /* * When the original hugepage is shared one, it does not have - * anon_vma prepared. + * an anon rmap prepared. */ ret = __vmf_anon_prepare(vmf); if (unlikely(ret)) diff --git a/mm/khugepaged.c b/mm/khugepaged.c index 3e8dbd2dd46d8d..913086eaf17ba0 100644 --- a/mm/khugepaged.c +++ b/mm/khugepaged.c @@ -1315,7 +1315,7 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, /* * Prevent all access to pagetables with the exception of * gup_fast later handled by the pmdp_collapse_flush() and the VM - * handled by the anon_vma lock + folio lock. + * handled by the anon rmap lock + folio lock. * * UFFDIO_MOVE is prevented to race as well thanks to the * mmap_lock. @@ -1382,8 +1382,8 @@ static enum scan_result collapse_huge_page(struct mm_struct *mm, } /* - * For PMD collapse all pages are isolated and locked so anon_vma - * rmap can't run anymore. For mTHP collapse the PMD entry has been + * For PMD collapse all pages are isolated and locked so the anon + * rmap walk can't run anymore. For mTHP collapse the PMD entry has been * removed and not all pages are isolated and locked, so we must hold * the lock to prevent neighboring folios from attempting to access * this PMD until its reinstalled. @@ -2168,9 +2168,9 @@ static void retract_page_tables(struct address_space *mapping, pgoff_t pgoff) /* * Huge page lock is still held, so normally the page table must - * remain empty; and we have already skipped anon_vma and - * userfaultfd_wp() vmas. But since the mmap_lock is not held, - * it is still possible for a racing userfaultfd_ioctl() or + * remain empty; and we have already skipped vmas with an anon + * rmap and userfaultfd_wp() vmas. But since the mmap_lock is not + * held, it is still possible for a racing userfaultfd_ioctl() or * madvise() to have inserted ptes or markers. Now that we hold * ptlock, repeating the retractable checks protects us from * races against the prior checks. diff --git a/mm/ksm.c b/mm/ksm.c index bcf5799bfe371f..fbeae63ced2043 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -793,7 +793,7 @@ static void break_cow(struct ksm_rmap_item *rmap_item) /* * It is not an accident that whenever we want to break COW - * to undo, we also need to drop a reference to the anon_vma. + * to undo, we also need to drop a reference to the anon rmap. */ put_anon_vma(rmap_item->anon_vma); /* @@ -1413,7 +1413,7 @@ static int replace_page(struct vm_area_struct *vma, struct page *page, goto out; /* * Some THP functions use the sequence pmdp_huge_clear_flush(), set_pmd_at() - * without holding anon_vma lock for write. So when looking for a + * without holding the anon rmap lock for write. So when looking for a * genuine pmde (in which to find pte), test present and !THP together. */ pmde = pmdp_get_lockless(pmd); @@ -1617,7 +1617,7 @@ static int try_to_merge_with_ksm_page(struct ksm_rmap_item *rmap_item, /* * We can consider the VMA only while still holding the mmap lock, - * so lock, so reference the anon_vma and calculate the linear + * so lock, so reference the anon rmap and calculate the linear * page index early, before stable_tree_append(). If anything goes * wrong that prevents the rmap_item from being added to the * stable_tree, break_cow() will clean it up. diff --git a/mm/madvise.c b/mm/madvise.c index 010ad3d47f5bb3..20135275cb5550 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -1325,7 +1325,7 @@ static long madvise_guard_install(struct madvise_behavior *madv_behavior) /* * If anonymous and we are establishing page tables the VMA ought to - * have an anon_vma associated with it. + * have an anon rmap associated with it. * * We will hold an mmap read lock if this is necessary, this is checked * as part of the VMA lock logic. @@ -1789,8 +1789,8 @@ static bool is_vma_lock_sufficient(struct vm_area_struct *vma, * anon_vma_prepare() explicitly requires an mmap lock for * serialisation, so we cannot use a VMA lock in this case. * - * Note we might race with anon_vma being set, however this makes this - * check overly paranoid which is safe. + * Note we might race with the anon rmap being assigned, however this + * makes this check overly paranoid which is safe. */ if (vma_is_anonymous(vma) && prepares_anon_vma(madv_behavior->behavior) && !vma_has_anon_rmap(vma)) diff --git a/mm/memory.c b/mm/memory.c index 544a9e5068f09d..6349ef676549a8 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -1291,8 +1291,8 @@ copy_pte_range(struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma, * copy_pmd_range()'s prior pmd_none_or_clear_bad(src_pmd), and the * error handling here, assume that exclusive mmap_lock on dst and src * protects anon from unexpected THP transitions; with shmem and file - * protected by mmap_lock-less collapse skipping areas with anon_vma - * (whereas vma_needs_copy() skips areas without anon_vma). A rework + * protected by mmap_lock-less collapse skipping areas with an anon rmap + * (whereas vma_needs_copy() skips areas without one). A rework * can remove such assumptions later, but this is good enough for now. */ dst_pte = pte_alloc_map_lock(dst_mm, dst_pmd, addr, &dst_ptl); @@ -1533,8 +1533,8 @@ vma_needs_copy(struct vm_area_struct *dst_vma, struct vm_area_struct *src_vma) if (dst_vma->vm_flags & VM_COPY_ON_FORK) return true; /* - * The presence of an anon_vma indicates an anonymous VMA has page - * tables which naturally cannot be reconstituted on page fault. + * The presence of an anon rmap indicates the VMA may map anonymous + * folios which naturally cannot be reconstituted on page fault. */ if (vma_has_anon_rmap(src_vma)) return true; @@ -3997,10 +3997,10 @@ static inline vm_fault_t vmf_can_call_fault(const struct vm_fault *vmf) * * When preparing to insert an anonymous page into a VMA from a * fault handler, call this function rather than anon_vma_prepare(). - * If this vma does not already have an associated anon_vma and we are + * If this vma does not already have an anon rmap and we are * only protected by the per-VMA lock, the caller must retry with the * mmap_lock held. __anon_vma_prepare() will look at adjacent VMAs to - * determine if this VMA can share its anon_vma, and that's not safe to + * determine if this VMA can share its anon rmap, and that's not safe to * do with only the per-VMA lock held for this VMA. * * Return: 0 if fault handling can proceed. Any other value should be diff --git a/mm/migrate.c b/mm/migrate.c index 7e3a81f0697442..7bdcdb57652f8f 100644 --- a/mm/migrate.c +++ b/mm/migrate.c @@ -1174,7 +1174,7 @@ static void migrate_folio_undo_src(struct folio *src, int was_mapped, { if (was_mapped) remove_migration_ptes(src, src, 0); - /* Drop an anon_vma reference if we took one */ + /* Drop an anon rmap reference if we took one */ if (anon_vma) put_anon_vma(anon_vma); if (locked) @@ -1281,15 +1281,15 @@ static int migrate_folio_unmap(new_folio_t get_new_folio, /* * By try_to_migrate(), src->mapcount goes down to 0 here. In this case, - * we cannot notice that anon_vma is freed while we migrate a page. - * This get_anon_vma() delays freeing anon_vma pointer until the end + * we cannot notice that the anon rmap is freed while we migrate a page. + * This get_anon_vma() delays freeing the anon rmap until the end * of migration. File cache pages are no problem because of page_lock() * File Caches may use write_page() or lock_page() in migration, then, * just care Anon page here. * * Only folio_get_anon_vma() understands the subtleties of - * getting a hold on an anon_vma from outside one of its mms. - * But if we cannot get anon_vma, then we won't need it anyway, + * getting a hold on an anon rmap from outside one of its mms. + * But if we cannot get the anon rmap, then we won't need it anyway, * because that implies that the anon page is no longer mapped * (and cannot be remapped so long as we hold the page lock). */ @@ -1432,7 +1432,7 @@ static int migrate_folio_move(free_folio_t put_new_folio, unsigned long private, * and will be freed. */ list_del(&src->lru); - /* Drop an anon_vma reference if we took one */ + /* Drop an anon rmap reference if we took one */ if (anon_vma) put_anon_vma(anon_vma); folio_unlock(src); diff --git a/mm/mmap.c b/mm/mmap.c index 98449f364af1c4..148ba01c03739f 100644 --- a/mm/mmap.c +++ b/mm/mmap.c @@ -547,7 +547,7 @@ unsigned long do_mmap(struct file *file, unsigned long addr, } case MAP_PRIVATE: /* - * Set pgoff according to addr for anon_vma. + * Set pgoff according to addr for the anon rmap. */ pgoff = addr >> PAGE_SHIFT; break; @@ -1774,8 +1774,8 @@ __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm) if (vma_test(tmp, VMA_WIPEONFORK_BIT)) { /* * VMA_WIPEONFORK_BIT gets a clean slate in the child. - * Don't prepare anon_vma until fault since we don't - * copy page for current vma. + * Don't prepare the anon rmap until fault since we + * don't copy pages for the current vma. */ tmp->anon_vma = NULL; } else if (anon_vma_fork(tmp, mpnt)) diff --git a/mm/mprotect.c b/mm/mprotect.c index 4b1296f0d502a2..e59c69cb5a2388 100644 --- a/mm/mprotect.c +++ b/mm/mprotect.c @@ -796,8 +796,8 @@ mprotect_fixup(struct vma_iterator *vmi, struct mmu_gather *tlb, /* * If we make a private mapping writable we increase our commit; * but (without finer accounting) cannot reduce our commit if we - * make it unwritable again except in the anonymous case where no - * anon_vma has yet to be assigned. + * make it unwritable again except in the anonymous case where the + * VMA's anon rmap has yet to be assigned. * * hugetlb mapping were accounted for even if read-only so there is * no need to account for them here. diff --git a/mm/mremap.c b/mm/mremap.c index 6be5867bb45566..1b8d06195bf37f 100644 --- a/mm/mremap.c +++ b/mm/mremap.c @@ -214,7 +214,7 @@ static int move_ptes(struct pagetable_move_control *pmc, int err = 0; /* - * When need_rmap_locks is true, we take the i_mmap_rwsem and anon_vma + * When need_rmap_locks is true, we take the i_mmap_rwsem and anon rmap * locks to ensure that rmap will always observe either the old or the * new ptes. This is the easiest way to avoid races with * truncate_pagecache(), page migration, etc... @@ -1366,8 +1366,8 @@ static void dontunmap_complete(struct vma_remap_struct *vrm, vma_clear_flags_mask(vma, VMA_LOCKED_MASK); /* - * anon_vma links of the old vma is no longer needed after its page - * table has been moved. + * The anon rmap links of the old vma are no longer needed after its + * page table has been moved. */ unlink_anon_vmas(vma); /* diff --git a/mm/pgtable-generic.c b/mm/pgtable-generic.c index 26643d76bfb00e..6e83ce3801b8af 100644 --- a/mm/pgtable-generic.c +++ b/mm/pgtable-generic.c @@ -349,7 +349,7 @@ pte_t *pte_offset_map_rw_nolock(struct mm_struct *mm, pmd_t *pmd, * pte_offset_map_lock(mm, pmd, addr, ptlp) is usually called with the pmd * pointer for addr, reached by walking down the mm's pgd, p4d, pud for addr: * either while holding mmap_lock or vma lock for read or for write; or in - * truncate or rmap context, while holding file's i_mmap_lock or anon_vma lock + * truncate or rmap context, while holding file's i_mmap_lock or anon rmap lock * for read (or for write). In a few cases, it may be used with pmd pointing to * a pmd_t already copied to or constructed on the stack. * diff --git a/mm/userfaultfd.c b/mm/userfaultfd.c index 666cc18902639b..3c7fd39deb1376 100644 --- a/mm/userfaultfd.c +++ b/mm/userfaultfd.c @@ -130,7 +130,7 @@ struct vm_area_struct *find_vma_and_prepare_anon(struct mm_struct *mm, * Should be called without holding mmap_lock. * * Return: A locked vma containing @address, -ENOENT if no vma is found, - * -ENOMEM if anon_vma couldn't be allocated, or -EAGAIN if vma refcount + * -ENOMEM if the anon rmap couldn't be allocated, or -EAGAIN if vma refcount * overflow happened due to high number of readers and the caller should * retry later. */ @@ -142,8 +142,8 @@ static struct vm_area_struct *uffd_lock_vma(struct mm_struct *mm, vma = lock_vma_under_rcu(mm, address); if (vma) { /* - * We know we're going to need to use anon_vma, so check - * that early. + * We know we're going to need an anon rmap, so check that + * early. */ if (!(vma->vm_flags & VM_SHARED) && unlikely(!vma_has_anon_rmap(vma))) vma_end_read(vma); @@ -1681,7 +1681,7 @@ static long move_pages_ptes(struct mm_struct *mm, pmd_t *dst_pmd, pmd_t *src_pmd /* * Verify the existence of the swapcache. If present, the folio's * index and mapping must be updated even when the PTE is a swap - * entry. The anon_vma lock is not taken during this process since + * entry. The anon rmap lock is not taken during this process since * the folio has already been unmapped, and the swap entry is * exclusive, preventing rmap walks. * @@ -1918,7 +1918,7 @@ static void uffd_move_unlock(struct vm_area_struct *dst_vma, * * move_pages() remaps arbitrary anonymous pages atomically in zero * copy. It only works on non shared anonymous pages because those can - * be relocated without generating non linear anon_vmas in the rmap + * be relocated without generating non linear anon rmaps in the rmap * code. * * It provides a zero copy mechanism to handle userspace page faults. diff --git a/mm/vma.c b/mm/vma.c index be3840d9ba4d8a..ea4dc3032657a1 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -97,7 +97,7 @@ static bool vma_is_fork_child(struct vm_area_struct *vma) { /* * The list_is_singular() test is to avoid merging VMA cloned from - * parents. This can improve scalability caused by the anon_vma root + * parents. This can improve scalability caused by the anon rmap root * lock. */ return vma && vma_has_anon_rmap(vma) && @@ -135,7 +135,7 @@ static bool is_mergeable_anon_vma(struct vma_merge_struct *vmg, bool merge_next) /* * We _can_ have !src, vmg->anon_vma via copy_vma(). In this instance we - * will remove the existing VMA's anon_vma's so there's no scalability + * will remove the existing VMA's anon rmap so there's no scalability * concerns. */ VM_WARN_ON(src && src_anon != src->anon_vma); @@ -151,10 +151,10 @@ static bool is_mergeable_anon_vma(struct vma_merge_struct *vmg, bool merge_next) return true; } - /* Case 2 - we will simply use tgt's anon_vma. */ + /* Case 2 - we will simply use tgt's anon rmap. */ if (vma_has_anon_rmap(tgt) && !src_anon) return !vma_is_fork_child(tgt); - /* Case 3 - the anon_vma's are already shared. */ + /* Case 3 - src and tgt already share an anon rmap. */ return src_anon == tgt_anon; } @@ -228,8 +228,8 @@ static bool needs_adjacent_anon_pgoff(const struct vma_merge_struct *vmg) * Return true if we can merge this (vma_flags,anon_vma,file,vm_pgoff) * in front of (at a lower virtual address and file offset than) the vma. * - * We cannot merge two vmas if they have differently assigned (non-NULL) - * anon_vmas, nor if same anon_vma is assigned but offsets incompatible. + * We cannot merge two vmas if they have differently assigned anon rmaps, + * nor if the same anon rmap is assigned but offsets incompatible. * * We don't check here for the merged mmap wrapping around the end of pagecache * indices (16TB on ia32) because do_mmap() does not permit mmap's which @@ -255,8 +255,8 @@ static bool can_vma_merge_before(struct vma_merge_struct *vmg) * Return true if we can merge this (vma_flags,anon_vma,file,vm_pgoff) * beyond (at a higher virtual address and file offset than) the vma. * - * We cannot merge two vmas if they have differently assigned (non-NULL) - * anon_vmas, nor if same anon_vma is assigned but offsets incompatible. + * We cannot merge two vmas if they have differently assigned anon rmaps, + * nor if the same anon rmap is assigned but offsets incompatible. * * We assume that vma is not removed as part of the merge. */ @@ -300,18 +300,18 @@ static void __remove_shared_vm_struct(struct vm_area_struct *vma, } /* - * vma has some anon_vma assigned, and is already inserted on that - * anon_vma's interval trees. + * vma has an anon rmap assigned, and is already inserted on its interval + * trees. * * Before updating the vma's vm_start / vm_end / vm_pgoff fields, the - * vma must be removed from the anon_vma's interval trees using + * vma must be removed from the anon rmap's interval trees using * anon_rmap_tree_pre_update_vma(). * * After the update, the vma will be reinserted using * anon_rmap_tree_post_update_vma(). * * The entire update must be protected by exclusive mmap_lock and by - * the root anon_vma's mutex. + * the anon rmap root lock. */ static void anon_rmap_tree_pre_update_vma(struct vm_area_struct *vma) @@ -479,7 +479,7 @@ static bool can_vma_merge_left(struct vma_merge_struct *vmg) * account the end position of the proposed range. * * In addition, if we can merge with the left VMA, ensure that left and right - * anon_vma's are also compatible. + * anon rmaps are also compatible. */ static bool can_vma_merge_right(struct vma_merge_struct *vmg, bool can_merge_left) @@ -495,7 +495,7 @@ static bool can_vma_merge_right(struct vma_merge_struct *vmg, /* * If we can merge with prev (left) and next (right), indicating that - * each VMA's anon_vma is compatible with the proposed anon_vma, this + * each VMA's anon rmap is compatible with the proposed anon rmap, this * does not mean prev and next are compatible with EACH OTHER. * * We therefore check this in addition to mergeability to either side. @@ -645,8 +645,8 @@ int split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, } /* - * dup_anon_vma() - Helper function to duplicate anon_vma on VMA merge in the - * instance that the destination VMA has no anon_vma but the source does. + * dup_anon_vma() - Helper function to duplicate the anon rmap on VMA merge in + * the instance that the destination VMA has no anon rmap but the source does. * * @dst: The destination VMA * @src: The source VMA @@ -659,17 +659,17 @@ static int dup_anon_vma(struct vm_area_struct *dst, { /* * There are three cases to consider for correctly propagating - * anon_vma's on merge. + * anon rmaps on merge. * - * The first is trivial - neither VMA has anon_vma, we need not do + * The first is trivial - neither VMA has an anon rmap, we need not do * anything. * - * The second where both have anon_vma is also a no-op, as they must + * The second where both have an anon rmap is also a no-op, as they must * then be the same, so there is simply nothing to copy. * - * Here we cover the third - if the destination VMA has no anon_vma, + * Here we cover the third - if the destination VMA has no anon rmap, * that is it is unfaulted, we need to ensure that the newly merged - * range is referenced by the anon_vma's of the source. + * range is referenced by the anon rmap of the source. */ if (vma_has_anon_rmap(src) && !vma_has_anon_rmap(dst)) { int ret; @@ -1017,9 +1017,9 @@ static __must_check struct vm_area_struct *vma_merge_existing_range( vmg->anon_pgoff = vma_start_anon_pgoff(prev); /* - * We already ensured anon_vma compatibility above, so now it's - * simply a case of, if prev has no anon_vma object, which of - * next or middle contains the anon_vma we must duplicate. + * We already ensured anon rmap compatibility above, so now it's + * simply a case of, if prev has no anon rmap, which of next or + * middle contains the anon rmap we must duplicate. */ err = dup_anon_vma(prev, vma_has_anon_rmap(next) ? next : middle, &anon_dup); @@ -1085,7 +1085,7 @@ static __must_check struct vm_area_struct *vma_merge_existing_range( unlink_anon_vmas(anon_dup); /* - * This means we have failed to clone anon_vma's correctly, but no + * This means we have failed to clone the anon rmap correctly, but no * actual changes to VMAs have occurred, so no harm no foul - if the * user doesn't want this reported and instead just wants to give up on * the merge, allow it. @@ -1282,7 +1282,7 @@ int vma_expand(struct vma_merge_struct *vmg) /* * If we are removing the next VMA or copying from a VMA - * (e.g. mremap()'ing), we must propagate anon_vma state. + * (e.g. mremap()'ing), we must propagate anon rmap state. * * Note that, by convention, callers ignore OOM for this case, so * we don't need to account for vmg->give_up_on_mm here. @@ -2059,16 +2059,16 @@ struct vm_area_struct *copy_vma(struct vm_area_struct **vmap, /* * Rough compatibility check to quickly see if it's even worth looking - * at sharing an anon_vma. + * at sharing an anon rmap. * * They need to have the same vm_file, and the flags can only differ * in things that mprotect may change. * - * NOTE! The fact that we share an anon_vma doesn't _have_ to mean that + * NOTE! The fact that we share an anon rmap doesn't _have_ to mean that * we can merge the two vma's. For example, we refuse to merge a vma if * there is a vm_ops->close() function, because that indicates that the * driver is doing some kind of reference counting. But that doesn't - * really matter for the anon_vma sharing case. + * really matter for the anon rmap sharing case. */ static int anon_vma_compatible(struct vm_area_struct *a, struct vm_area_struct *b) { @@ -2101,13 +2101,13 @@ static int anon_vma_compatible(struct vm_area_struct *a, struct vm_area_struct * } /* - * Do some basic sanity checking to see if we can re-use the anon_vma + * Do some basic sanity checking to see if we can re-use the anon rmap * from 'old'. The 'a'/'b' vma's are in VM order - one of them will be * the same as 'old', the other will be the new one that is trying - * to share the anon_vma. + * to share the anon rmap. * * NOTE! This runs with mmap_lock held for reading, so it is possible that - * the anon_vma of 'old' is concurrently in the process of being set up + * the anon rmap of 'old' is concurrently in the process of being set up * by another page fault trying to merge _that_. But that's ok: if it * is being set up, that automatically means that it will be a singleton * acceptable for merging, so we can do all of this optimistically. But @@ -2121,8 +2121,8 @@ static int anon_vma_compatible(struct vm_area_struct *a, struct vm_area_struct * * accessing an uninitialised anon_vma's fields may result in a UAF. * * IOW: that the "list_is_singular()" test on the anon_vma_chain only - * matters for the 'stable anon_vma' case (ie the thing we want to avoid - * is to return an anon_vma that is "complex" due to having gone through + * matters for the 'stable anon rmap' case (ie the thing we want to avoid + * is to return an anon rmap that is "complex" due to having gone through * a fork). * * We also make sure that the two vma's are compatible (adjacent, @@ -2145,10 +2145,10 @@ static struct anon_vma *reusable_anon_vma(struct vm_area_struct *old, /* * find_mergeable_anon_vma is used by anon_vma_prepare, to check - * neighbouring vmas for a suitable anon_vma, before it goes off - * to allocate a new anon_vma. It checks because a repetitive + * neighbouring vmas for a suitable anon rmap, before it goes off + * to allocate a new anon rmap. It checks because a repetitive * sequence of mprotects and faults may otherwise lead to distinct - * anon_vmas being allocated, preventing vma merge in subsequent + * anon rmaps being allocated, preventing vma merge in subsequent * mprotect. */ struct anon_vma *find_mergeable_anon_vma(struct vm_area_struct *vma) @@ -2174,13 +2174,13 @@ struct anon_vma *find_mergeable_anon_vma(struct vm_area_struct *vma) /* * We might reach here with anon_vma == NULL if we can't find - * any reusable anon_vma. + * any reusable anon rmap. * There's no absolute need to look only at touching neighbours: - * we could search further afield for "compatible" anon_vmas. + * we could search further afield for "compatible" anon rmaps. * But it would probably just be a waste of time searching, - * or lead to too many vmas hanging off the same anon_vma. + * or lead to too many vmas hanging off the same anon rmap. * We're trying to allow mprotect remerging later on, - * not trying to minimize memory used for anon_vmas. + * not trying to minimize memory used for anon rmaps. */ return anon_vma; } @@ -2276,7 +2276,7 @@ static void vm_lock_anon_vma(struct mm_struct *mm, struct anon_vma *anon_vma) /* * We can safely modify head.next after taking the * anon_vma->root->rwsem. If some other vma in this mm shares - * the same anon_vma we won't take it again. + * the same anon rmap we won't take it again. * * No need of atomic instructions here, head.next * can't change from under us thanks to the @@ -2318,14 +2318,14 @@ static void vm_lock_mapping(struct mm_struct *mm, struct address_space *mapping) * mmap_lock in write mode is required in order to block all operations * that could modify pagetables and free pages without need of * altering the vma layout. It's also needed in write mode to avoid new - * anon_vmas to be associated with existing vmas. + * anon rmaps being associated with existing vmas. * * A single task can't take more than one mm_take_all_locks() in a row * or it would deadlock. * * The LSB in anon_vma->rb_root.rb_node and the AS_MM_ALL_LOCKS bitflag in * mapping->flags avoid to take the same lock twice, if more than one - * vma in this mm is backed by the same anon_vma or address_space. + * vma in this mm is backed by the same anon rmap or address_space. * * We take locks in following order, accordingly to comment at beginning * of mm/rmap.c: @@ -3419,7 +3419,7 @@ int expand_upwards(struct vm_area_struct *vma, unsigned long address) if (next && vma_is_accessible(next)) { if (!vma_test(next, VMA_GROWSUP_BIT)) return -ENOMEM; - /* Check that both stack segments have the same anon_vma? */ + /* Check that both stack segments have the same anon rmap? */ } if (next) @@ -3429,7 +3429,7 @@ int expand_upwards(struct vm_area_struct *vma, unsigned long address) if (vma_iter_prealloc(&vmi, vma)) return -ENOMEM; - /* We must make sure the anon_vma is allocated. */ + /* We must make sure the anon rmap is allocated. */ if (unlikely(anon_vma_prepare(vma))) { vma_iter_free(&vmi); return -ENOMEM; @@ -3492,7 +3492,7 @@ int expand_downwards(struct vm_area_struct *vma, unsigned long address) /* Enforce stack_guard_gap */ prev = vma_prev(&vmi); - /* Check that both stack segments have the same anon_vma? */ + /* Check that both stack segments have the same anon rmap? */ if (prev) { if (!vma_test(prev, VMA_GROWSDOWN_BIT) && vma_is_accessible(prev) && @@ -3507,7 +3507,7 @@ int expand_downwards(struct vm_area_struct *vma, unsigned long address) if (vma_iter_prealloc(&vmi, vma)) return -ENOMEM; - /* We must make sure the anon_vma is allocated. */ + /* We must make sure the anon rmap is allocated. */ if (unlikely(anon_vma_prepare(vma))) { vma_iter_free(&vmi); return -ENOMEM; @@ -3587,7 +3587,7 @@ int insert_vm_struct(struct mm_struct *mm, struct vm_area_struct *vma) /* * The vm_pgoff of a purely anonymous vma should be irrelevant - * until its first write fault, when page's anon_vma and index + * until its first write fault, when page's anon rmap and index * are set. But now set the vm_pgoff it will almost certainly * end up with (unless mremap moves it elsewhere before that * first wfault), so /proc/pid/maps tells a consistent story. From e1397da8d49d6e50e1aa60b2d9a2f1478f6b929a Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Thu, 17 Sep 2026 16:39:24 -0700 Subject: [PATCH 0745/1012] mm-update-comments-to-refer-to-anon-rmap-rather-than-anon_vma-fix fix comment, per Zi Yan. Cc: "Lorenzo Stoakes (ARM)" Cc: Zi Yan Signed-off-by: Andrew Morton --- mm/ksm.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/ksm.c b/mm/ksm.c index fbeae63ced2043..f33523a84a1a13 100644 --- a/mm/ksm.c +++ b/mm/ksm.c @@ -1617,7 +1617,7 @@ static int try_to_merge_with_ksm_page(struct ksm_rmap_item *rmap_item, /* * We can consider the VMA only while still holding the mmap lock, - * so lock, so reference the anon rmap and calculate the linear + * so reference the anon rmap and calculate the linear * page index early, before stable_tree_append(). If anything goes * wrong that prevents the rmap_item from being added to the * stable_tree, break_cow() will clean it up. From 220fcdd280a7bbdc605db2570b023d3389356f16 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:21:59 -0700 Subject: [PATCH 0746/1012] mm/damon/api: remove NR_DAMOS_FILTER_TYPES Patch series "mm/damon: improve readability, clarity and test coverage". Yet another batch of miscellaneous DAMON minor improvements. Mostly focused on readability and clarity of code and document, and unit/self test coverage. No user-visible behavioral change is intended. This patch (of 10): Nobody uses NR_DAMOS_FILTER_TYPES. Remove it. Link: https://lore.kernel.org/20260917142210.90829-1-sj@kernel.org Link: https://lore.kernel.org/20260917142210.90829-2-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- include/linux/damon.h | 2 -- 1 file changed, 2 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index bbb190b4740150..836353c4ab9aab 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -400,7 +400,6 @@ struct damos_stat { * @DAMOS_FILTER_TYPE_ADDR: Address range. * @DAMOS_FILTER_TYPE_TARGET: Data Access Monitoring target. * @DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: probe_hits weighted sum range. - * @NR_DAMOS_FILTER_TYPES: Number of filter types. * * All types except &DAMOS_FILTER_TYPE_ADDR, &DAMOS_FILTER_TYPE_TARGET and * &DAMOS_FILTER_TYPE_PROBE_HITS_WSUM are handled by the underlying &struct @@ -422,7 +421,6 @@ enum damos_filter_type { DAMOS_FILTER_TYPE_ADDR, DAMOS_FILTER_TYPE_TARGET, DAMOS_FILTER_TYPE_PROBE_HITS_WSUM, - NR_DAMOS_FILTER_TYPES, }; /** From 8f93e0972a81e58be55d2769506cd5e3129da1de Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:00 -0700 Subject: [PATCH 0747/1012] mm/damon/core: use abs_diff() in damon_feed_loop_next_input() damon_feed_loop_next_input() is open-coding absolute diff calculation instead of the dedicated helper, abs_diff(), for no good reason. Use the dedicated helper. Link: https://lore.kernel.org/20260917142210.90829-3-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index e1b49c3d72b865..e49bbf7c07e85e 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2926,10 +2926,7 @@ static unsigned long damon_feed_loop_next_input(unsigned long last_input, if (score >= goal * 2) return min_input; - if (over_achieving) - score_goal_diff = score - goal; - else - score_goal_diff = goal - score; + score_goal_diff = abs_diff(score, goal); if (last_input < ULONG_MAX / score_goal_diff) compensation = last_input * score_goal_diff / goal; From d658baaa28a080531fbdf9ee653d301bb309753a Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:01 -0700 Subject: [PATCH 0748/1012] mm/damon/core: use mult_frac() in damon_feed_loop_next_input() damon_feed_loop_next_input() does its best effort overflow protection. score_goal_diff is always smaller than goal (10,000). Hence the calculation can be replaced to use mult_frac() without concerning the overflow. Use mult_frac(). Link: https://lore.kernel.org/20260917142210.90829-4-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index e49bbf7c07e85e..f3ccfc9ad0f086 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2927,11 +2927,7 @@ static unsigned long damon_feed_loop_next_input(unsigned long last_input, return min_input; score_goal_diff = abs_diff(score, goal); - - if (last_input < ULONG_MAX / score_goal_diff) - compensation = last_input * score_goal_diff / goal; - else - compensation = last_input / goal * score_goal_diff; + compensation = mult_frac(last_input, score_goal_diff, goal); if (over_achieving) return max(last_input - compensation, min_input); From 69745f9823a64f12cd1663632635c4270b390706 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:02 -0700 Subject: [PATCH 0749/1012] mm/damon/core: set damon_ctx->walk_control_obsolete in damon_new_ctx() damos_walk() should be called for a damon_ctx context that has successfully started at least once. That's because damon_ctx->walk_control_obsolete is initialized when kdamond starts. If the rule is violated, an indefinite wait can happen. There is no existing violation of the rule. damon_call() had a similar rule, and it turned out keeping the rule is not easy for damon_call()'s case. Hence, commit 8023b5f47e09 ("mm/damon/core: set ctx->call_controls_obsolete in damon_new_ctx()") added the initialization in damon_new_ctx() and removed the rule. Keeping the rule for damos_walk() is relatively easier. But having slightly different rules for similar functions could be confusing. Sashiko, for example, repeatedly asked questions about this. Do the initialization of walk_control_obsolete in damon_new_ctx() for consistency. Link: https://lore.kernel.org/20260917142210.90829-5-sj@kernel.org Link: https://lore.kernel.org/20260915011614.102342-1-sj@kernel.org [1] Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index f3ccfc9ad0f086..327277ba365818 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -939,6 +939,7 @@ struct damon_ctx *damon_new_ctx(void) INIT_LIST_HEAD(&ctx->schemes); ctx->call_controls_obsolete = true; + ctx->walk_control_obsolete = true; prandom_seed_state(&ctx->rnd_state, get_random_u64()); return ctx; @@ -2308,10 +2309,6 @@ int damon_call(struct damon_ctx *ctx, struct damon_call_control *control) * passed at least one &damos->apply_interval_us, kdamond marks the request as * completed so that damos_walk() can wakeup and return. * - * Note that this function should be called only after damon_start() with the - * @ctx has succeeded. Otherwise, this function could fall into an indefinite - * wait. - * * Return: 0 on success, negative error code otherwise. */ int damos_walk(struct damon_ctx *ctx, struct damos_walk_control *control) From 3251a35cabc1c5b70b521ebc7cf6b7a142e5ad5c Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:03 -0700 Subject: [PATCH 0750/1012] mm/damon/core: document damon_call()/damon_start() race hang issue Let's suppose damon_start() and damon_call() are executed in parallel for the same DAMON context. Then, damon_call() could show ctx->damon_calls_obsolete set while ctx->kdamond is unset. If damon_start() sets ctx->kdamond before damon_call() starts the cancelling, damon_call() can indefinitely hang. No DAMON API caller does such parallel execution of damon_start() and damon_call(), so the issue doesn't exist. But who knows what will happen in future. Add a clarification comment for caution. Link: https://lore.kernel.org/20260917142210.90829-6-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/core.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 327277ba365818..add1b7afb957ac 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2258,6 +2258,9 @@ int damon_kdamond_pid(struct damon_ctx *ctx) * * When this function is failed, the @ctx is guaranteed to be stopped. * + * This function should not be called in parallel to damon_start() for the + * @ctx. In the case, this function could indefinitely hang. + * * Return: 0 on success, negative error code otherwise. */ int damon_call(struct damon_ctx *ctx, struct damon_call_control *control) From 344f6c2120a42e3cbd433fede1ac536e63f95a3e Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:04 -0700 Subject: [PATCH 0751/1012] mm/damon/paddr: remove pa parameter from damon_pa_filter_pass() damon_pa_filter_pass() receives the 'pa' parameter, but doesn't use it. Remove the parameter from the function signature. Link: https://lore.kernel.org/20260917142210.90829-7-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/paddr.c | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/mm/damon/paddr.c b/mm/damon/paddr.c index 5abfabaa339e0e..2cfdc356b41572 100644 --- a/mm/damon/paddr.c +++ b/mm/damon/paddr.c @@ -163,8 +163,7 @@ static bool damon_pa_filter_match(struct damon_filter *filter, return matched == filter->matching; } -static bool damon_pa_filter_pass(phys_addr_t pa, struct folio *folio, - struct damon_probe *p) +static bool damon_pa_filter_pass(struct folio *folio, struct damon_probe *p) { struct damon_filter *f; bool pass = true; @@ -200,7 +199,7 @@ static unsigned int damon_pa_apply_probes(struct damon_ctx *ctx, ctx->addr_unit); folio = damon_get_folio(PHYS_PFN(pa)); damon_for_each_probe(p, ctx) { - if (damon_pa_filter_pass(pa, folio, p)) + if (damon_pa_filter_pass(folio, p)) r->probe_hits[i]++; i++; } From 77125f03414cc216cd2c748d18c6183e452abe95 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:05 -0700 Subject: [PATCH 0752/1012] mm/damon/tests/core-kunit: test eligible_mem_bp commitment There was a DAMOS quota goal commit bug [1] that doesn't update the nid field for DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP metric goal. Add a kunit test case for confirming nid commitment. Link: https://lore.kernel.org/20260917142210.90829-8-sj@kernel.org Link: https://lore.kkernel.org/20260827045035.94611-1-sj@kernel.org [1] Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/tests/core-kunit.h | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 5da84caf4124d5..1f19fefdd98c39 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -835,6 +835,9 @@ static void damos_test_commit_quota_goal_for(struct kunit *test, KUNIT_EXPECT_EQ(test, dst->nid, src->nid); KUNIT_EXPECT_EQ(test, dst->memcg_id, src->memcg_id); break; + case DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP: + KUNIT_EXPECT_EQ(test, dst->nid, src->nid); + break; default: break; } @@ -898,6 +901,13 @@ static void damos_test_commit_quota_goal(struct kunit *test) .current_value = 345, .last_psi_total = 567, }); + damos_test_commit_quota_goal_for(test, &dst, + &(struct damos_quota_goal){ + .metric = DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP, + .target_value = 12, + .current_value = 345, + .nid = 6, + }); } static void damos_test_commit_quota_goals_for(struct kunit *test, From fa969228c909056d22e16bd246d78612276d9995 Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:06 -0700 Subject: [PATCH 0753/1012] mm/damon/tests/core-kunit: add probe_hits_wsum damos filter commit test DAMOS filter commit kunit test lacks test cases for probe_hits_wsum filter type. Add test cases for probe_hits_wsum type DAMOS filter commit. Link: https://lore.kernel.org/20260917142210.90829-9-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- mm/damon/tests/core-kunit.h | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 1f19fefdd98c39..5ff0436c58441f 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1166,6 +1166,10 @@ static void damos_test_commit_filter_for(struct kunit *test, KUNIT_EXPECT_EQ(test, dst->sz_range.min, src->sz_range.min); KUNIT_EXPECT_EQ(test, dst->sz_range.max, src->sz_range.max); break; + case DAMOS_FILTER_TYPE_PROBE_HITS_WSUM: + KUNIT_EXPECT_EQ(test, dst->range_min, src->range_min); + KUNIT_EXPECT_EQ(test, dst->range_max, src->range_max); + break; default: break; } @@ -1239,6 +1243,22 @@ static void damos_test_commit_filter(struct kunit *test) .allow = true, .target_idx = 6, }, false); + damos_test_commit_filter_for(test, &dst, + &(struct damos_filter){ + .type = DAMOS_FILTER_TYPE_PROBE_HITS_WSUM, + .matching = false, + .allow = true, + .range_min = 12, + .range_max = 34, + }, false); + damos_test_commit_filter_for(test, &dst, + &(struct damos_filter){ + .type = DAMOS_FILTER_TYPE_PROBE_HITS_WSUM, + .matching = false, + .allow = true, + .range_min = 34, + .range_max = 12, + }, true); } static void damos_test_help_initailize_scheme(struct damos *scheme) From 8048981e71ea2d3829529e8b49598a8f39dd6c1b Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:07 -0700 Subject: [PATCH 0754/1012] selftests/damon/sysfs_memcg_path_leak: fail only for real DAMON leak The selftest can fail for any leak if it happens while the test is running. Remove the false positive test failures by further checking if the expected leaking function is called out on the report. Link: https://lore.kernel.org/20260917142210.90829-10-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- tools/testing/selftests/damon/sysfs_memcg_path_leak.sh | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/tools/testing/selftests/damon/sysfs_memcg_path_leak.sh b/tools/testing/selftests/damon/sysfs_memcg_path_leak.sh index 33a7ff43ed6cc7..34c37129c49fe1 100755 --- a/tools/testing/selftests/damon/sysfs_memcg_path_leak.sh +++ b/tools/testing/selftests/damon/sysfs_memcg_path_leak.sh @@ -41,5 +41,12 @@ if [ "$kmemleak_report" = "" ] then exit 0 fi +if ! echo "$kmemleak_report" | grep "memcg_path_store" --quiet +then + echo "[WARN] memleak found; apparently not from DAMON, though" + echo "$kmemleak_report" + exit 0 +fi + echo "$kmemleak_report" exit 1 From f1febec5791e7fd050eec4ab3cea92cc0ae4c0fe Mon Sep 17 00:00:00 2001 From: SJ Park Date: Thu, 17 Sep 2026 07:22:08 -0700 Subject: [PATCH 0755/1012] Docs/mm/damon/design: clarify bp is basis point DAMON design document uses "bp" for "basis point" in multiple places. Because it is not clearly mentioned, it is difficult to understand what "bp" stands for. Add the clarification. Link: https://lore.kernel.org/20260917142210.90829-11-sj@kernel.org Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reported-by: Randy Dunlap Closes: https://lore.kernel.org/107dc6ba-697e-4b25-ba3e-8ce2499cac9a@infradead.org Acked-by: Randy Dunlap Acked-by: Zenghui Yu (Huawei) Cc: Asier Gutierrez Cc: Brendan Higgins Cc: David Gow Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Mike Rapoport Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/damon/design.rst | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/Documentation/mm/damon/design.rst b/Documentation/mm/damon/design.rst index 0a86792f90a184..e82390e77a70ae 100644 --- a/Documentation/mm/damon/design.rst +++ b/Documentation/mm/damon/design.rst @@ -445,7 +445,7 @@ users to set the aimed amount of access events to observe via DAMON within given time interval. The target can be specified by the user as a ratio of DAMON-observed access events to the theoretical maximum amount of the events (``access_bp``) that measured within a given number of aggregations -(``aggrs``). +(``aggrs``). The ratio is in basis point (bp or 1/10,000). The DAMON-observed access events are calculated in byte granularity based on DAMON :ref:`region assumption `. For @@ -717,7 +717,8 @@ mechanism tries to make ``current_value`` of ``target_metric`` be same to in microseconds that measured from last quota reset to next quota reset. DAMOS does the measurement on its own, so only ``target_value`` need to be set by users at the initial time. In other words, DAMOS does self-feedback. -- ``node_mem_used_bp``: Specific NUMA node's used memory ratio in bp (1/10,000). +- ``node_mem_used_bp``: Specific NUMA node's used memory ratio in basis point + (bp or 1/10,000). - ``node_mem_free_bp``: Specific NUMA node's free memory ratio in bp (1/10,000). - ``node_memcg_used_bp``: Specific cgroup's node used memory ratio for a specific NUMA node, in bp (1/10,000). From 58eb409ec711cc7e22a14601ffe3f531f918b3f7 Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Thu, 17 Sep 2026 06:47:35 -0700 Subject: [PATCH 0756/1012] Documentation: kmemleak: describe the metadata pool, not the early log Patch series "kmemleak: fix stale documentation and raise the verbose default". The first two patches fix statements in Documentation/dev-tools/kmemleak.rst that do not match mm/kmemleak.c. The third requires one more consecutive unreferenced scan before a CONFIG_DEBUG_KMEMLEAK_VERBOSE kernel reports a leak on the console, to avoid the last false positives I am seeing when running CONFIG_DEBUG_KMEMLEAK_VERBOSE on a daily basis. This patch (of 3): commit c5665868183f ("mm: kmemleak: use the memory pool for early allocations") removed the early log buffer in favour of a static pool of kmemleak_object structures, but the documentation still describes the old mechanism. Fix the documentation by describing what the pool actually is, matching the Kconfig help text. Link: https://lore.kernel.org/20260917142210.90829-1-sj@kernel.org Link: https://lore.kernel.org/20260917-b4-kmemleak-doc-v1-1-84fde6d1f749@debian.org Fixes: c5665868183f ("mm: kmemleak: use the memory pool for early allocations") Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: Catalin Marinas Cc: Jonathan Corbet Cc: Randy Dunlap --- Documentation/dev-tools/kmemleak.rst | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/Documentation/dev-tools/kmemleak.rst b/Documentation/dev-tools/kmemleak.rst index d1b690b1716967..8dad7647742d31 100644 --- a/Documentation/dev-tools/kmemleak.rst +++ b/Documentation/dev-tools/kmemleak.rst @@ -66,9 +66,14 @@ Memory scanning parameters can be modified at run-time by writing to the Kmemleak can also be disabled at boot-time by passing ``kmemleak=off`` on the kernel command line. -Memory may be allocated or freed before kmemleak is initialised and -these actions are stored in an early log buffer. The size of this buffer -is configured via the CONFIG_DEBUG_KMEMLEAK_MEM_POOL_SIZE option. +Memory may be allocated or freed before kmemleak is initialised, so a +static pool of metadata objects is used to track those allocations. Once +kmemleak is fully initialised the pool becomes an emergency reserve, used +whenever a metadata object cannot be allocated from the slab. The number +of objects in the pool is configured via the +CONFIG_DEBUG_KMEMLEAK_MEM_POOL_SIZE option. Exhausting it at run time +prints "Cannot allocate a kmemleak_object structure" and disables +kmemleak. If CONFIG_DEBUG_KMEMLEAK_DEFAULT_OFF are enabled, the kmemleak is disabled by default. Passing ``kmemleak=on`` on the kernel command From ef31840f63539486828ef1afd3f39b072ebeac6e Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Thu, 17 Sep 2026 06:47:36 -0700 Subject: [PATCH 0757/1012] Documentation: kmemleak: fix stale statements about scanning Three statements in the "false positives/negatives" and "Limitations" sections have never matched the code: - task stack scanning is on by default (kmemleak_stack_scan = 1), as the parameter list earlier in the same document already states; - MSECS_MIN_AGE has been 5000, not 1000, since the initial commit; - scanning is done by a periodic kthread. Reading the debugfs file only lists what the last scan found; kmemleak_open() calls seq_open() and never scans. Link: https://lore.kernel.org/20260917-b4-kmemleak-doc-v1-2-84fde6d1f749@debian.org Fixes: 04f70336c80c ("kmemleak: Add documentation on the memory leak detector") Fixes: e0a2a1601bec ("kmemleak: Enable task stacks scanning by default") Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: Catalin Marinas Cc: Jonathan Corbet Cc: Randy Dunlap --- Documentation/dev-tools/kmemleak.rst | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/Documentation/dev-tools/kmemleak.rst b/Documentation/dev-tools/kmemleak.rst index 8dad7647742d31..b5fe7e671d0f82 100644 --- a/Documentation/dev-tools/kmemleak.rst +++ b/Documentation/dev-tools/kmemleak.rst @@ -190,7 +190,8 @@ reported by kmemleak because values found during the memory scanning point to such objects. To reduce the number of false negatives, kmemleak provides the kmemleak_ignore, kmemleak_scan_area, kmemleak_no_scan and kmemleak_erase functions (see above). The task stacks also increase the -amount of false negatives and their scanning is not enabled by default. +amount of false negatives and their scanning is enabled by default; it +can be turned off with ``stack=off``. The false positives are objects wrongly reported as being memory leaks (orphan). For objects known not to be leaks, kmemleak provides the @@ -200,7 +201,7 @@ longer be scanned. Some of the reported leaks are only transient, especially on SMP systems, because of pointers temporarily stored in CPU registers or -stacks. Kmemleak defines MSECS_MIN_AGE (defaulting to 1000) representing +stacks. Kmemleak defines MSECS_MIN_AGE (defaulting to 5000) representing the minimum age of an object to be reported as a memory leak. The ``min_unref_scans`` module parameter requires an object to be seen @@ -217,8 +218,10 @@ Limitations and Drawbacks ------------------------- The main drawback is the reduced performance of memory allocation and -freeing. To avoid other penalties, the memory scanning is only performed -when the /sys/kernel/debug/kmemleak file is read. Anyway, this tool is +freeing. To avoid other penalties, the memory scanning is performed by a +periodic thread rather than on every allocation. Reading the +/sys/kernel/debug/kmemleak file only lists the objects found by the last +scan; writing ``scan`` to it triggers a new one. Anyway, this tool is intended for debugging purposes where the performance might not be the most important requirement. From 0e03f2be4d4af715bd4540af10e92c3e005ed450 Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Thu, 17 Sep 2026 06:47:37 -0700 Subject: [PATCH 0758/1012] mm: kmemleak: raise min_unref_scans to 3 for verbose auto-scan CONFIG_DEBUG_KMEMLEAK_VERBOSE sends every report to the console, so a transient false positive there is broadcast to whatever collects the kernel log rather than sitting in the debugfs file until someone looks. That asymmetry justifies being more conservative than the general case. Require one more consecutive unreferenced scan before reporting. The only cost is that a genuine leak is reported one scan interval later (600s by default); the value stays writable at run time through the module parameter. Kernels without CONFIG_DEBUG_KMEMLEAK_VERBOSE keep reporting on the first unreferenced scan. I've been running constant upstream kernel with CONFIG_DEBUG_KMEMLEAK_VERBOSE set, and I am still seeing some rare false positive, that goes away with min_unref_scans=3, so, making it the default based on my heuristic. Link: https://lore.kernel.org/20260917-b4-kmemleak-doc-v1-3-84fde6d1f749@debian.org Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: Catalin Marinas Cc: Jonathan Corbet Cc: Randy Dunlap --- Documentation/dev-tools/kmemleak.rst | 2 +- mm/kmemleak.c | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/Documentation/dev-tools/kmemleak.rst b/Documentation/dev-tools/kmemleak.rst index b5fe7e671d0f82..c0d32937234251 100644 --- a/Documentation/dev-tools/kmemleak.rst +++ b/Documentation/dev-tools/kmemleak.rst @@ -206,7 +206,7 @@ the minimum age of an object to be reported as a memory leak. The ``min_unref_scans`` module parameter requires an object to be seen unreferenced in that many consecutive scans before it is reported. It -defaults to 2 when CONFIG_DEBUG_KMEMLEAK_VERBOSE is enabled, where the +defaults to 3 when CONFIG_DEBUG_KMEMLEAK_VERBOSE is enabled, where the periodic scan thread confirms a leak on its own, and to 1 otherwise. A value of 1 preserves the historical behaviour; higher values filter the transient false positives described above, at the cost of delaying genuine diff --git a/mm/kmemleak.c b/mm/kmemleak.c index 8fa409a4f9fb26..5d0daea93c471f 100644 --- a/mm/kmemleak.c +++ b/mm/kmemleak.c @@ -238,7 +238,7 @@ static struct task_struct *scan_thread; static unsigned long jiffies_min_age; /* consecutive scans an object must stay unreferenced before reporting */ static unsigned int min_unref_scans = - IS_ENABLED(CONFIG_DEBUG_KMEMLEAK_VERBOSE) ? 2 : 1; + IS_ENABLED(CONFIG_DEBUG_KMEMLEAK_VERBOSE) ? 3 : 1; module_param(min_unref_scans, uint, 0644); static unsigned long jiffies_last_scan; /* delay between automatic memory scannings */ From 46ee9529f8767c670a7dd6ea1261375dfb71b838 Mon Sep 17 00:00:00 2001 From: Lisa Wang Date: Thu, 17 Sep 2026 20:43:47 +0000 Subject: [PATCH 0759/1012] mm: memory_failure: clarify the MF_DELAYED definition Patch series "mm: Fix MF_DELAYED handling on memory failure", v6. This series addresses an issue in the memory failure handling path where MF_DELAYED is incorrectly treated as an error. This issue was discovered while testing memory failure handling for guest_memfd. The proposed solution involves - 1. Clarifying the definition of MF_DELAYED to mean that memory failure handling is only partially completed, and that the metadata for the memory that failed (as in struct page/folio) is still referenced. 2. Updating shmems handling to align with the clarified definition. 3. Updating how the result of .error_remove_folio() is interpreted. This patch (of 5): This patch clarifies the definition of MF_DELAYED to represent cases where a folio's removal is initiated but not immediately completed (e.g., due to remaining metadata references). Link: https://lore.kernel.org/20260917-memory-failure-mf-delayed-fix-v6-0-4b00856b5364@google.com Link: https://lore.kernel.org/20260917-memory-failure-mf-delayed-fix-v6-1-4b00856b5364@google.com Signed-off-by: Lisa Wang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Miaohe Lin Reviewed-by: Ackerley Tng Cc: Andi Kleen Cc: Baolin Wang Cc: Dave Hansen Cc: David Rientjes Cc: Fuad Tabba Cc: Hidehiro Kawai Cc: Hugh Dickins Cc: Isaku Yamahata Cc: Jiaqi Yan Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michael Roth Cc: Michal Hocko Cc: Mike Rapoport Cc: Naoya Horiguchi Cc: Paolo Bonzini Cc: Rik van Riel Cc: Sean Christopherson Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Xiaoyao Li Cc: Yu Zhang --- mm/memory-failure.c | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/mm/memory-failure.c b/mm/memory-failure.c index a8b03e2920ba8a..057cb9537db292 100644 --- a/mm/memory-failure.c +++ b/mm/memory-failure.c @@ -847,24 +847,25 @@ static int kill_accessing_process(struct task_struct *p, unsigned long pfn, } /* - * MF_IGNORED - The m-f() handler marks the page as PG_hwpoisoned'ed. + * MF_IGNORED - The m-f() handler marks the page as PG_hwpoison'ed. * But it could not do more to isolate the page from being accessed again, * nor does it kill the process. This is extremely rare and one of the * potential causes is that the page state has been changed due to * underlying race condition. This is the most severe outcomes. * - * MF_FAILED - The m-f() handler marks the page as PG_hwpoisoned'ed. + * MF_FAILED - The m-f() handler marks the page as PG_hwpoison'ed. * It should have killed the process, but it can't isolate the page, * due to conditions such as extra pin, unmap failure, etc. Accessing * the page again may trigger another MCE and the process will be killed * by the m-f() handler immediately. * - * MF_DELAYED - The m-f() handler marks the page as PG_hwpoisoned'ed. - * The page is unmapped, and is removed from the LRU or file mapping. - * An attempt to access the page again will trigger page fault and the - * PF handler will kill the process. + * MF_DELAYED - The m-f() handler marks the page as PG_hwpoison'ed. + * It means the page was unmapped and partially isolated (e.g. removed from + * file mapping or the LRU) but full cleanup is deferred (e.g. the metadata + * for the memory, as in struct page/folio, is still referenced). Any + * further access to the page will result in the process being killed. * - * MF_RECOVERED - The m-f() handler marks the page as PG_hwpoisoned'ed. + * MF_RECOVERED - The m-f() handler marks the page as PG_hwpoison'ed. * The page has been completely isolated, that is, unmapped, taken out of * the buddy system, or hole-punched out of the file mapping. */ From 47b246f152b10a84bc3f71c3d507fa7dc196c897 Mon Sep 17 00:00:00 2001 From: Lisa Wang Date: Thu, 17 Sep 2026 20:43:48 +0000 Subject: [PATCH 0760/1012] mm: memory_failure: Allow truncate_error_folio to return MF_DELAYED The .error_remove_folio a_ops is used by different filesystems to handle folio truncation upon discovery of a memory failure in the memory associated with the given folio. Currently, MF_DELAYED is treated as an error, causing "Failed to punch page" to be written to the console. MF_DELAYED is then relayed to the caller of truncate_error_folio() as MF_FAILED. This further causes memory_failure() to return -EBUSY, which then always causes a SIGBUS. This is also implies that regardless of whether the thread's memory corruption kill policy is PR_MCE_KILL_EARLY or PR_MCE_KILL_LATE, a memory failure with MF_DELAYED will always cause a SIGBUS. Update truncate_error_folio() to return MF_DELAYED to the caller if the .error_remove_folio() callback reports MF_DELAYED. Link: https://lore.kernel.org/20260917-memory-failure-mf-delayed-fix-v6-2-4b00856b5364@google.com Fixes: 6a46079cf57a ("HWPOISON: The high level memory error handler in the VM v7") Fixes: a7800aa80ea4 ("KVM: Add KVM_CREATE_GUEST_MEMFD ioctl() for guest-specific backing memory") Signed-off-by: Lisa Wang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Miaohe Lin Reviewed-by: Ackerley Tng Cc: Andi Kleen Cc: Baolin Wang Cc: Dave Hansen Cc: David Rientjes Cc: Fuad Tabba Cc: Hidehiro Kawai Cc: Hugh Dickins Cc: Isaku Yamahata Cc: Jiaqi Yan Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michael Roth Cc: Michal Hocko Cc: Mike Rapoport Cc: Naoya Horiguchi Cc: Paolo Bonzini Cc: Rik van Riel Cc: Sean Christopherson Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Xiaoyao Li Cc: Yu Zhang --- mm/memory-failure.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/mm/memory-failure.c b/mm/memory-failure.c index 057cb9537db292..83ab0fce22a61a 100644 --- a/mm/memory-failure.c +++ b/mm/memory-failure.c @@ -939,10 +939,12 @@ static int truncate_error_folio(struct folio *folio, unsigned long pfn, if (mapping->a_ops->error_remove_folio) { int err = mapping->a_ops->error_remove_folio(mapping, folio); - if (err != 0) + if (err == MF_DELAYED) + ret = err; + else if (err != 0) pr_info("%#lx: Failed to punch page: %d\n", pfn, err); else if (!filemap_release_folio(folio, GFP_NOIO)) - pr_info("%#lx: failed to release buffers\n", pfn); + pr_info("%#lx: Failed to release buffers\n", pfn); else ret = MF_RECOVERED; } else { From 2c0c86010a833bbbb241521eb36aea28235c54cb Mon Sep 17 00:00:00 2001 From: Lisa Wang Date: Thu, 17 Sep 2026 20:43:49 +0000 Subject: [PATCH 0761/1012] mm: shmem: Update shmem handler to the MF_DELAYED definition To align with the definition of MF_DELAYED, update shmem_error_remove_folio() to return MF_DELAYED. shmem handles memory failures but defers the actual file truncation. The function's return value should therefore be MF_DELAYED to accurately reflect the state. Currently, this logical error does not cause a bug, because: - For shmem folios, folio->private is not set. - As a result, filemap_release_folio() is a no-op and returns true. - This, in turn, causes truncate_error_folio() to incorrectly return MF_RECOVERED. - The caller then treats MF_RECOVERED as a success condition, masking the issue. The previous patch relays MF_DELAYED to the caller of truncate_error_folio() before any logging, so returning MF_DELAYED from shmem_error_remove_folio() will retain the original behavior of not adding any logs. The return value of truncate_error_folio() is consumed in action_result(), which treats MF_DELAYED the same way as MF_RECOVERED, hence action_result() also returns the same thing after this change. Link: https://lore.kernel.org/20260917-memory-failure-mf-delayed-fix-v6-3-4b00856b5364@google.com Signed-off-by: Lisa Wang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Miaohe Lin Reviewed-by: Ackerley Tng Cc: Andi Kleen Cc: Baolin Wang Cc: Dave Hansen Cc: David Rientjes Cc: Fuad Tabba Cc: Hidehiro Kawai Cc: Hugh Dickins Cc: Isaku Yamahata Cc: Jiaqi Yan Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michael Roth Cc: Michal Hocko Cc: Mike Rapoport Cc: Naoya Horiguchi Cc: Paolo Bonzini Cc: Rik van Riel Cc: Sean Christopherson Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Xiaoyao Li Cc: Yu Zhang --- mm/shmem.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/shmem.c b/mm/shmem.c index 05bc7c3aa52548..f33dbf5af2cb79 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -5363,7 +5363,7 @@ static void __init shmem_destroy_inodecache(void) static int shmem_error_remove_folio(struct address_space *mapping, struct folio *folio) { - return 0; + return MF_DELAYED; } static const struct address_space_operations shmem_aops = { From d02e2e785d7eb9754331ba86cff996c12200b3cc Mon Sep 17 00:00:00 2001 From: Lisa Wang Date: Thu, 17 Sep 2026 20:43:50 +0000 Subject: [PATCH 0762/1012] mm: memory_failure: Generalize extra_pins handling to all MF_DELAYED cases Generalize extra_pins handling to all MF_DELAYED cases not only shmem_mapping. If MF_DELAYED is returned, the filemap continues to hold refcounts on the folio. Hence, take that into account when checking for extra refcounts. As clarified in an earlier patch, a return value of MF_DELAYED implies that the page still has elevated refcounts. Hence, set extra_pins to true if the return value is MF_DELAYED. This is aligned with the implementation in me_swapcache_dirty(), where, if a folio is still in the swap cache, ret is set to MF_DELAYED and extra_pins is set to true. Link: https://lore.kernel.org/20260917-memory-failure-mf-delayed-fix-v6-4-4b00856b5364@google.com Signed-off-by: Lisa Wang Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Acked-by: Miaohe Lin Reviewed-by: Ackerley Tng Cc: Andi Kleen Cc: Baolin Wang Cc: Dave Hansen Cc: David Rientjes Cc: Fuad Tabba Cc: Hidehiro Kawai Cc: Hugh Dickins Cc: Isaku Yamahata Cc: Jiaqi Yan Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michael Roth Cc: Michal Hocko Cc: Mike Rapoport Cc: Naoya Horiguchi Cc: Paolo Bonzini Cc: Rik van Riel Cc: Sean Christopherson Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Xiaoyao Li Cc: Yu Zhang --- mm/memory-failure.c | 8 ++------ 1 file changed, 2 insertions(+), 6 deletions(-) diff --git a/mm/memory-failure.c b/mm/memory-failure.c index 83ab0fce22a61a..d237f556b3a09e 100644 --- a/mm/memory-failure.c +++ b/mm/memory-failure.c @@ -1039,18 +1039,14 @@ static int me_pagecache_clean(struct page_state *ps, struct page *p) goto out; } - /* - * The shmem page is kept in page cache instead of truncating - * so is expected to have an extra refcount after error-handling. - */ - extra_pins = shmem_mapping(mapping); - /* * Truncation is a bit tricky. Enable it per file system for now. * * Open: to take i_rwsem or not for this? Right now we don't. */ ret = truncate_error_folio(folio, page_to_pfn(p), mapping); + + extra_pins = ret == MF_DELAYED; if (has_extra_refcount(ps, p, extra_pins)) ret = MF_FAILED; From e5471ae45ffcb8134e15fb7cd989e9c14ac2196a Mon Sep 17 00:00:00 2001 From: Lisa Wang Date: Thu, 17 Sep 2026 20:43:51 +0000 Subject: [PATCH 0763/1012] mm: selftests: Add shmem into memory failure test Add a shmem memory failure selftest to test the shmem memory failure is correct after modifying shmem return value. Specifically, test the expected behavior under various scenarios combining page dirtiness (dirty vs clean) and failure types (hard vs soft): + Dirty + Hard: Trigger a SIGBUS on injection, and trigger another SIGBUS when reading the page again. + Dirty + Soft: No SIGBUS is triggered, and the original value can be read successfully. + Clean + Hard: No SIGBUS is triggered on injection, but trigger a SIGBUS when trying to read the page again. + Clean + Soft: No SIGBUS is triggered, and the page can be read successfully. Link: https://lore.kernel.org/20260917-memory-failure-mf-delayed-fix-v6-5-4b00856b5364@google.com Signed-off-by: Lisa Wang Signed-off-by: Andrew Morton Acked-by: Miaohe Lin Cc: Ackerley Tng Cc: Andi Kleen Cc: Baolin Wang Cc: Dave Hansen Cc: David Hildenbrand (Arm) Cc: David Rientjes Cc: Fuad Tabba Cc: Hidehiro Kawai Cc: Hugh Dickins Cc: Isaku Yamahata Cc: Jiaqi Yan Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michael Roth Cc: Michal Hocko Cc: Mike Rapoport Cc: Naoya Horiguchi Cc: Paolo Bonzini Cc: Rik van Riel Cc: Sean Christopherson Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vishal Annapurve Cc: Vlastimil Babka Cc: Xiaoyao Li Cc: Yu Zhang --- tools/testing/selftests/mm/memory-failure.c | 114 +++++++++++++++++++- 1 file changed, 111 insertions(+), 3 deletions(-) diff --git a/tools/testing/selftests/mm/memory-failure.c b/tools/testing/selftests/mm/memory-failure.c index f3cb578b160962..1f22869e94670a 100644 --- a/tools/testing/selftests/mm/memory-failure.c +++ b/tools/testing/selftests/mm/memory-failure.c @@ -29,9 +29,14 @@ enum result_type { MADV_HARD_ANON, MADV_HARD_CLEAN_PAGECACHE, MADV_HARD_DIRTY_PAGECACHE, + MADV_HARD_CLEAN_SHMEM, + MADV_HARD_DIRTY_SHMEM, MADV_SOFT_ANON, MADV_SOFT_CLEAN_PAGECACHE, MADV_SOFT_DIRTY_PAGECACHE, + MADV_SOFT_CLEAN_SHMEM, + MADV_SOFT_DIRTY_SHMEM, + READ_ERROR, }; static jmp_buf signal_jmp_buf; @@ -157,17 +162,22 @@ static void check(struct __test_metadata *_metadata, FIXTURE_DATA(memory_failure case MADV_HARD_CLEAN_PAGECACHE: case MADV_SOFT_CLEAN_PAGECACHE: case MADV_SOFT_DIRTY_PAGECACHE: - /* It is not expected to receive a SIGBUS signal. */ - ASSERT_EQ(setjmp, 0); - + case MADV_SOFT_DIRTY_SHMEM: /* The page content should remain unchanged. */ ASSERT_TRUE(check_memory(vaddr, self->page_size)); + /* FALLTHORUGH */ + case MADV_HARD_CLEAN_SHMEM: + case MADV_SOFT_CLEAN_SHMEM: + /* It is not expected to receive a SIGBUS signal. */ + ASSERT_EQ(setjmp, 0); /* The backing pfn of addr should have changed. */ ASSERT_NE(pagemap_get_pfn(self->pagemap_fd, vaddr), self->pfn); break; case MADV_HARD_ANON: case MADV_HARD_DIRTY_PAGECACHE: + case MADV_HARD_DIRTY_SHMEM: + case READ_ERROR: /* The SIGBUS signal should have been received. */ ASSERT_EQ(setjmp, 1); @@ -263,6 +273,20 @@ static int prepare_file(const char *fname, unsigned long size) return fd; } +static int prepare_shmem(const char *fname, unsigned long size) +{ + int fd; + + fd = memfd_create(fname, 0); + if (fd < 0) + return -1; + if (ftruncate(fd, size) < 0) { + close(fd); + return -1; + } + return fd; +} + /* Borrowed from mm/gup_longterm.c. */ static int get_fs_type(int fd) { @@ -365,4 +389,88 @@ TEST_F(memory_failure, dirty_pagecache) ASSERT_EQ(close(fd), 0); } +TEST_F(memory_failure, dirty_shmem) +{ + int fd; + char *addr; + int ret; + + fd = prepare_shmem("shmem-file", self->page_size); + if (fd < 0) + SKIP(return, "failed to open test shmem-file.\n"); + + addr = mmap(0, self->page_size, PROT_READ | PROT_WRITE, + MAP_SHARED, fd, 0); + if (addr == MAP_FAILED) { + close(fd); + SKIP(return, "mmap failed, not enough memory.\n"); + } + memset(addr, 0xce, self->page_size); + + prepare(_metadata, self, addr); + + ret = sigsetjmp(signal_jmp_buf, 1); + if (!ret && !self->injection_attempted) { + self->injection_attempted = true; + ASSERT_EQ(variant->inject(self, addr), 0); + } + + if (variant->type == MADV_HARD) { + check(_metadata, self, addr, MADV_HARD_DIRTY_SHMEM, ret); + ret = sigsetjmp(signal_jmp_buf, 1); + if (ret == 0) + FORCE_READ(*addr); + check(_metadata, self, addr, READ_ERROR, ret); + } else { + check(_metadata, self, addr, MADV_SOFT_DIRTY_SHMEM, ret); + } + + ASSERT_EQ(munmap(addr, self->page_size), 0); + + ASSERT_EQ(close(fd), 0); +} + +TEST_F(memory_failure, clean_shmem) +{ + int fd; + char *addr; + int ret; + + fd = prepare_shmem("shmem-file", self->page_size); + if (fd < 0) + SKIP(return, "failed to open test shmem-file.\n"); + + addr = mmap(0, self->page_size, PROT_READ | PROT_WRITE, + MAP_SHARED, fd, 0); + if (addr == MAP_FAILED) { + close(fd); + SKIP(return, "mmap failed, not enough memory.\n"); + } + FORCE_READ(*addr); + + prepare(_metadata, self, addr); + + ret = sigsetjmp(signal_jmp_buf, 1); + if (!ret && !self->injection_attempted) { + self->injection_attempted = true; + ASSERT_EQ(variant->inject(self, addr), 0); + } + + if (variant->type == MADV_HARD) { + check(_metadata, self, addr, MADV_HARD_CLEAN_SHMEM, ret); + ret = sigsetjmp(signal_jmp_buf, 1); + if (ret == 0) + FORCE_READ(*addr); + check(_metadata, self, addr, READ_ERROR, ret); + } else { + /* Test the address accessability without check_memory(). */ + FORCE_READ(*addr); + check(_metadata, self, addr, MADV_SOFT_CLEAN_SHMEM, ret); + } + + ASSERT_EQ(munmap(addr, self->page_size), 0); + + ASSERT_EQ(close(fd), 0); +} + TEST_HARNESS_MAIN From 57515c8fb52f97e2b80f264f90c19e763190fd47 Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Thu, 17 Sep 2026 17:06:00 -0700 Subject: [PATCH 0764/1012] mm-selftests-add-shmem-into-memory-failure-test-fix fix commant typo, per Lisa Cc: Lisa Wang Cc: Ackerley Tng Signed-off-by: Andrew Morton --- tools/testing/selftests/mm/memory-failure.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/memory-failure.c b/tools/testing/selftests/mm/memory-failure.c index 1f22869e94670a..135a2e8069ba5c 100644 --- a/tools/testing/selftests/mm/memory-failure.c +++ b/tools/testing/selftests/mm/memory-failure.c @@ -165,7 +165,7 @@ static void check(struct __test_metadata *_metadata, FIXTURE_DATA(memory_failure case MADV_SOFT_DIRTY_SHMEM: /* The page content should remain unchanged. */ ASSERT_TRUE(check_memory(vaddr, self->page_size)); - /* FALLTHORUGH */ + /* FALLTHROUGH */ case MADV_HARD_CLEAN_SHMEM: case MADV_SOFT_CLEAN_SHMEM: /* It is not expected to receive a SIGBUS signal. */ From 0db2bf8d26e67190300c0835dfe48ba22fc9bf25 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Fri, 18 Sep 2026 10:56:10 +0800 Subject: [PATCH 0765/1012] mm/hugetlb_cgroup: move per-node usage on cross node migration Patch series "mm/hugetlb_cgroup: move the per-node usage along with the folio", v2. The per-node usage reported by hugetlb..numa_stat is accounted against folio_nid() in __hugetlb_cgroup_commit_charge() and __hugetlb_cgroup_uncharge_folio(), so it is only correct while a folio stays charged on the same node and in the same hugetlb_cgroup. Two paths move a folio which stays charged, and neither moves the usage with it. hugetlb_cgroup_migrate() only moves the hugetlb_cgroup pointers of a folio migrated to another node, and hugetlb_cgroup_move_parent() only moves the page_counter charges and the hugetlb_cgroup pointer of the folios of a dying cgroup. In both cases the node (or cgroup) which was charged keeps a usage which never goes away, while the node (or cgroup) which ends up uncharging the folio underflows as soon as the folio is freed. Patch 1/2 moves the usage along with the folio on cross node migration, patch 2/2 does the same for the folios a dying cgroup reparents. This patch (of 2): hugetlb..numa_stat uses folio_nid() to account usage in __hugetlb_cgroup_commit_charge() and __hugetlb_cgroup_uncharge_folio(). hugetlb_cgroup_migrate() only moves hugetlb_cgroup pointers, leaving per-node usage behind on the source node during cross-node migration. When the migrated folio gets uncharged, we subtract usage from the destination node counter. This creates stale usage on the source node and unsigned long counter underflow on the destination node. The hugetlb..numa_stat interface exposes these incorrect per-node usage values to userspace. Add a hugetlb_cgroup_move_usage() helper which moves the usage from the old node to the new node, and call it from hugetlb_cgroup_migrate(). Link: https://lore.kernel.org/20260918-for-hugetlb-charge-v2-0-2b6d8c2bdc36@kylinos.cn Link: https://lore.kernel.org/20260918-for-hugetlb-charge-v2-1-2b6d8c2bdc36@kylinos.cn Fixes: f47761999052 ("hugetlb: add hugetlb.*.numa_stat file") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Colin Ian King Cc: David Hildenbrand Cc: Kees Cook Cc: Mina Almasry Cc: Oscar Salvador Cc: Shakeel Butt Cc: --- mm/hugetlb_cgroup.c | 31 +++++++++++++++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/mm/hugetlb_cgroup.c b/mm/hugetlb_cgroup.c index ecb6e0b7819a0d..7cf7c18119b432 100644 --- a/mm/hugetlb_cgroup.c +++ b/mm/hugetlb_cgroup.c @@ -179,6 +179,34 @@ static void hugetlb_cgroup_css_free(struct cgroup_subsys_state *css) hugetlb_cgroup_free(hugetlb_cgroup_from_css(css)); } +static void hugetlb_cgroup_move_usage(struct hugetlb_cgroup *from, + struct hugetlb_cgroup *to, + struct folio *from_folio, + struct folio *to_folio) +{ + int idx = hstate_index(folio_hstate(from_folio)); + unsigned long nr_pages = folio_nr_pages(from_folio); + int from_nid = folio_nid(from_folio); + int to_nid = folio_nid(to_folio); + unsigned long usage; + + lockdep_assert_held(&hugetlb_lock); + + if (!from || !to) + return; + + if (from == to && from_nid == to_nid) + return; + + usage = from->nodeinfo[from_nid]->usage[idx]; + if (WARN_ON_ONCE(usage < nr_pages)) + return; + WRITE_ONCE(from->nodeinfo[from_nid]->usage[idx], usage - nr_pages); + + usage = to->nodeinfo[to_nid]->usage[idx]; + WRITE_ONCE(to->nodeinfo[to_nid]->usage[idx], usage + nr_pages); +} + /* * Should be called with hugetlb_lock held. * Since we are holding hugetlb_lock, pages cannot get moved from @@ -906,6 +934,9 @@ void hugetlb_cgroup_migrate(struct folio *old_folio, struct folio *new_folio) /* move the h_cg details to new cgroup */ set_hugetlb_cgroup(new_folio, h_cg); set_hugetlb_cgroup_rsvd(new_folio, h_cg_rsvd); + + hugetlb_cgroup_move_usage(h_cg, h_cg, old_folio, new_folio); + list_move(&new_folio->lru, &h->hugepage_activelist); spin_unlock_irq(&hugetlb_lock); } From 686d2928ed9d0d0647504454abece122b72b7f57 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Fri, 18 Sep 2026 10:56:11 +0800 Subject: [PATCH 0766/1012] mm/hugetlb_cgroup: move per-node usage on cgroup reparenting hugetlb_cgroup_css_offline() hands the folios of a dying cgroup over to its parent with hugetlb_cgroup_move_parent(), which moves the page_counter charges and the hugetlb_cgroup pointer of the folio but not its per-node usage. The parent's hugetlb..numa_stat is short by that usage while they are charged, and underflows once they are freed, exposing incorrect per-node usage values to userspace. Move the per-node usage to the parent as well. The folios keep their node here, so only the cgroup which holds the usage changes. Link: https://lore.kernel.org/20260918-for-hugetlb-charge-v2-2-2b6d8c2bdc36@kylinos.cn Fixes: f47761999052 ("hugetlb: add hugetlb.*.numa_stat file") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Colin Ian King Cc: David Hildenbrand Cc: Kees Cook Cc: Mina Almasry Cc: Oscar Salvador Cc: Shakeel Butt Cc: --- mm/hugetlb_cgroup.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/hugetlb_cgroup.c b/mm/hugetlb_cgroup.c index 7cf7c18119b432..3fb41311e4c7dc 100644 --- a/mm/hugetlb_cgroup.c +++ b/mm/hugetlb_cgroup.c @@ -241,6 +241,8 @@ static void hugetlb_cgroup_move_parent(int idx, struct hugetlb_cgroup *h_cg, /* Take the pages off the local counter */ page_counter_cancel(counter, nr_pages); + hugetlb_cgroup_move_usage(h_cg, parent, folio, folio); + set_hugetlb_cgroup(folio, parent); out: return; From 0887f49a153788d9ee8d2d80dfba2b23a26cffd0 Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:12 -0700 Subject: [PATCH 0767/1012] proc/task_mmu: remove unnecessary helpers Patch series "read proc/pid/smaps_rollup under per-vma lock", v5. proc/pid/smaps_rollup can be read using the combination of RCU and VMA read locks, similar to proc/pid/{maps|smaps|numa_maps}. RCU is required to safely traverse the VMA tree and VMA lock stabilizes the VMA being processed and the pagetable walk. Note that we have to keep the logic to drop mmap_lock on contention because even when using per-VMA locks we might have to fall back to holding the mmap_lock. The first 5 patches are cleanups making later change simpler. The main change is in patch 6. Patch 7 extends existing proc-maps-race tearing test to verify smaps_rollup content. This patch (of 7): When per-vma locks were behind a config option, a number of helper functions were needed to simplify the locking code. Now that these locks are universally available, we can do a little cleanup. Remove lock_vma_range(), unlock_vma_range(), query_vma_setup(), query_vma_teardown() helpers. No functional change intended. Link: https://lore.kernel.org/20260918153318.758387-1-surenb@google.com Link: https://lore.kernel.org/20260918153318.758387-2-surenb@google.com Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Liam R. Howlett (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Usama Arif Acked-by: David Hildenbrand (Arm) Cc: Jann Horn Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Vlastimil Babka --- fs/proc/task_mmu.c | 67 ++++++++++++---------------------------------- 1 file changed, 17 insertions(+), 50 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 052e8dc796bcf8..191a054d7e4093 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -160,25 +160,6 @@ static void unlock_ctx_vma(struct proc_maps_locking_ctx *lock_ctx) } } -static inline bool lock_vma_range(struct seq_file *m, - struct proc_maps_locking_ctx *lock_ctx) -{ - rcu_read_lock(); - reset_lock_ctx(lock_ctx); - - return true; -} - -static inline void unlock_vma_range(struct proc_maps_locking_ctx *lock_ctx) -{ - if (lock_ctx->mmap_locked) { - unlock_ctx_mm(lock_ctx); - } else { - unlock_ctx_vma(lock_ctx); - rcu_read_unlock(); - } -} - static struct vm_area_struct *get_next_vma(struct proc_maps_private *priv, loff_t last_pos) { @@ -286,13 +267,8 @@ static void *m_start(struct seq_file *m, loff_t *ppos) return NULL; } - if (!lock_vma_range(m, lock_ctx)) { - mmput(mm); - put_task_struct(priv->task); - priv->task = NULL; - return ERR_PTR(-EINTR); - } - + rcu_read_lock(); + reset_lock_ctx(lock_ctx); /* * Reset current position if last_addr was set before * and it's not a sentinel. @@ -325,7 +301,12 @@ static void m_stop(struct seq_file *m, void *v) return; release_task_mempolicy(priv); - unlock_vma_range(&priv->lock_ctx); + if (priv->lock_ctx.mmap_locked) { + unlock_ctx_mm(&priv->lock_ctx); + } else { + unlock_ctx_vma(&priv->lock_ctx); + rcu_read_unlock(); + } mmput(mm); put_task_struct(priv->task); priv->task = NULL; @@ -518,21 +499,6 @@ static int pid_maps_open(struct inode *inode, struct file *file) PROCMAP_QUERY_VMA_FLAGS \ ) -static int query_vma_setup(struct proc_maps_locking_ctx *lock_ctx) -{ - reset_lock_ctx(lock_ctx); - - return 0; -} - -static void query_vma_teardown(struct proc_maps_locking_ctx *lock_ctx) -{ - if (lock_ctx->mmap_locked) - unlock_ctx_mm(lock_ctx); - else - unlock_ctx_vma(lock_ctx); -} - static struct vm_area_struct *query_vma_find_by_addr(struct proc_maps_locking_ctx *lock_ctx, unsigned long addr) { @@ -653,12 +619,7 @@ static int do_procmap_query(struct mm_struct *mm, void __user *uarg) if (!mm || !mmget_not_zero(mm)) return -ESRCH; - err = query_vma_setup(&lock_ctx); - if (err) { - mmput(mm); - return err; - } - + reset_lock_ctx(&lock_ctx); vma = query_matching_vma(&lock_ctx, karg.query_addr, karg.query_flags); if (IS_ERR(vma)) { err = PTR_ERR(vma); @@ -732,7 +693,10 @@ static int do_procmap_query(struct mm_struct *mm, void __user *uarg) vm_file = get_file(vma->vm_file); /* unlock vma or mmap_lock, and put mm_struct before copying data to user */ - query_vma_teardown(&lock_ctx); + if (lock_ctx.mmap_locked) + unlock_ctx_mm(&lock_ctx); + else + unlock_ctx_vma(&lock_ctx); mmput(mm); if (karg.build_id_size) { @@ -773,7 +737,10 @@ static int do_procmap_query(struct mm_struct *mm, void __user *uarg) return 0; out: - query_vma_teardown(&lock_ctx); + if (lock_ctx.mmap_locked) + unlock_ctx_mm(&lock_ctx); + else + unlock_ctx_vma(&lock_ctx); mmput(mm); out_file: if (vm_file) From 6ca88ac9f6e12be0ad24ece8fc2107fa97f422e0 Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:13 -0700 Subject: [PATCH 0768/1012] proc/task_mmu: remove unnecessary inlines in function definitions It was pointed out in the previous reviews of this code that many functions are specified as inline, which is unnecessary as the compiler can make that decision by itself. Cleanup these definitions. No change in the resulting binary file size with gcc v15.2.0. No functional change intended. Link: https://lore.kernel.org/20260918153318.758387-3-surenb@google.com Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Liam R. Howlett (Oracle) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: Usama Arif Acked-by: David Hildenbrand (Arm) Cc: Jann Horn Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Vlastimil Babka --- fs/proc/task_mmu.c | 34 +++++++++++++++++++--------------- 1 file changed, 19 insertions(+), 15 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 191a054d7e4093..8a72dc0dc9452b 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -130,7 +130,8 @@ static void release_task_mempolicy(struct proc_maps_private *priv) } #endif -static inline int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) +#ifdef CONFIG_PROC_PAGE_MONITOR +static int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) { int ret = mmap_read_lock_killable(lock_ctx->mm); @@ -139,8 +140,9 @@ static inline int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) return ret; } +#endif -static inline void unlock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) +static void unlock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) { mmap_read_unlock(lock_ctx->mm); lock_ctx->mmap_locked = false; @@ -177,8 +179,8 @@ static struct vm_area_struct *get_next_vma(struct proc_maps_private *priv, return vma; } -static inline bool fallback_to_mmap_lock(struct proc_maps_private *priv, - loff_t pos) +static bool fallback_to_mmap_lock(struct proc_maps_private *priv, + loff_t pos) { struct proc_maps_locking_ctx *lock_ctx = &priv->lock_ctx; @@ -194,7 +196,8 @@ static inline bool fallback_to_mmap_lock(struct proc_maps_private *priv, return true; } -static inline void drop_rcu(struct proc_maps_private *priv) +#ifdef CONFIG_PROC_PAGE_MONITOR +static void drop_rcu(struct proc_maps_private *priv) { if (priv->lock_ctx.mmap_locked) return; @@ -202,7 +205,7 @@ static inline void drop_rcu(struct proc_maps_private *priv) rcu_read_unlock(); } -static inline void reacquire_rcu(struct proc_maps_private *priv) +static void reacquire_rcu(struct proc_maps_private *priv) { if (priv->lock_ctx.mmap_locked) return; @@ -211,6 +214,7 @@ static inline void reacquire_rcu(struct proc_maps_private *priv) /* Reinitialize the iterator. */ vma_iter_set(&priv->iter, priv->lock_ctx.locked_vma->vm_end); } +#endif static struct vm_area_struct *proc_get_vma(struct seq_file *m, loff_t *ppos) { @@ -1230,7 +1234,7 @@ static const struct mm_walk_ops smaps_shmem_walk_vma_lock_ops = { .walk_lock = PGWALK_VMA_RDLOCK_VERIFY, }; -static inline const struct mm_walk_ops * +static const struct mm_walk_ops * get_smaps_walk_ops(struct proc_maps_private *priv) { if (priv->lock_ctx.mmap_locked) @@ -1238,7 +1242,7 @@ get_smaps_walk_ops(struct proc_maps_private *priv) return &smaps_walk_vma_lock_ops; } -static inline const struct mm_walk_ops * +static const struct mm_walk_ops * get_smaps_shmem_walk_ops(struct proc_maps_private *priv) { if (priv->lock_ctx.mmap_locked) @@ -1572,7 +1576,7 @@ struct clear_refs_private { enum clear_refs_types type; }; -static inline bool pte_is_pinned(struct vm_area_struct *vma, unsigned long addr, pte_t pte) +static bool pte_is_pinned(struct vm_area_struct *vma, unsigned long addr, pte_t pte) { struct folio *folio; @@ -1588,8 +1592,8 @@ static inline bool pte_is_pinned(struct vm_area_struct *vma, unsigned long addr, return folio_maybe_dma_pinned(folio); } -static inline void clear_soft_dirty(struct vm_area_struct *vma, - unsigned long addr, pte_t *pte) +static void clear_soft_dirty(struct vm_area_struct *vma, unsigned long addr, + pte_t *pte) { if (!pgtable_supports_soft_dirty()) return; @@ -1620,7 +1624,7 @@ static inline void clear_soft_dirty(struct vm_area_struct *vma, } #if defined(CONFIG_TRANSPARENT_HUGEPAGE) -static inline void clear_soft_dirty_pmd(struct vm_area_struct *vma, +static void clear_soft_dirty_pmd(struct vm_area_struct *vma, unsigned long addr, pmd_t *pmdp) { pmd_t old, pmd = *pmdp; @@ -1646,7 +1650,7 @@ static inline void clear_soft_dirty_pmd(struct vm_area_struct *vma, } } #else -static inline void clear_soft_dirty_pmd(struct vm_area_struct *vma, +static void clear_soft_dirty_pmd(struct vm_area_struct *vma, unsigned long addr, pmd_t *pmdp) { } @@ -1848,7 +1852,7 @@ struct pagemapread { #define PM_END_OF_BUFFER 1 -static inline pagemap_entry_t make_pme(u64 frame, u64 flags) +static pagemap_entry_t make_pme(u64 frame, u64 flags) { return (pagemap_entry_t) { .pme = (frame & PM_PFRAME_MASK) | flags }; } @@ -3390,7 +3394,7 @@ static const struct mm_walk_ops show_numa_vma_lock_ops = { .walk_lock = PGWALK_VMA_RDLOCK_VERIFY, }; -static inline const struct mm_walk_ops * +static const struct mm_walk_ops * get_show_numa_ops(struct proc_maps_private *priv) { if (priv->lock_ctx.mmap_locked) From be417124eb814bc0b0ce3b988f2261d2620cc74d Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Sun, 27 Sep 2026 10:59:53 -0700 Subject: [PATCH 0769/1012] proc-task_mmu-remove-unnecessary-inlines-in-function-definitions-fix fix CONFIG_NUMA=y && CONFIG_PROC_PAGE_MONITOR=n build show_numa_map() uses drop_rcu() and reacquire_rcu(), but those helpers are currently built only when CONFIG_PROC_PAGE_MONITOR is enabled. A kernel with CONFIG_NUMA=y and CONFIG_PROC_PAGE_MONITOR=n therefore fails to build with implicit declarations for both helpers. Cc: Suren Baghdasaryan Signed-off-by: Andrew Morton --- fs/proc/task_mmu.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 8a72dc0dc9452b..4ed3b76d00c871 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -196,7 +196,7 @@ static bool fallback_to_mmap_lock(struct proc_maps_private *priv, return true; } -#ifdef CONFIG_PROC_PAGE_MONITOR +#if defined(CONFIG_PROC_PAGE_MONITOR) || defined(CONFIG_NUMA) static void drop_rcu(struct proc_maps_private *priv) { if (priv->lock_ctx.mmap_locked) From cd01c5ddf0c2eddab432d640303918e45411ba18 Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:14 -0700 Subject: [PATCH 0770/1012] proc/task_mmu: clarify shmem mapping walk conditions in smap_gather_stats() smap_gather_stats() optimizes stats gathering by skipping the walk for shmem mappings in certain conditions. Update the comment to clarify these conditions and use vma_is_cow_mapping() for CoW identification instead of open-coding it. No functional change intended. Link: https://lore.kernel.org/20260918153318.758387-4-surenb@google.com Suggested-by: David Hildenbrand (Arm) Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Cc: Jann Horn Cc: Liam R. Howlett (Oracle) Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Usama Arif Cc: Vlastimil Babka --- fs/proc/task_mmu.c | 23 +++++++++-------------- 1 file changed, 9 insertions(+), 14 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 4ed3b76d00c871..dd9cccf8a4892d 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -1274,23 +1274,18 @@ static void smap_gather_stats(struct proc_maps_private *priv, if (vma->vm_file && shmem_mapping(vma->vm_file->f_mapping)) { /* - * For shared or readonly shmem mappings we know that all - * swapped out pages belong to the shmem object, and we can - * obtain the swap value much more efficiently. For private - * writable mappings, we might have COW pages that are - * not affected by the parent swapped out pages of the shmem - * object, so we have to distinguish them during the page walk. - * Unless we know that the shmem object (or the part mapped by - * our VMA) has no swapped out pages at all. + * CoW mappings might map anon folios that do not belong to + * shmem. Perform a less efficient page table walk in this + * situation, unless we know that the shmem object (or the + * part mapped by our VMA) has no swapped out pages at all. */ - unsigned long shmem_swapped = shmem_swap_usage(vma); + const unsigned long shmem_swapped = shmem_swap_usage(vma); + const bool is_cow = vma_is_cow_mapping(vma); - if (!start && (!shmem_swapped || (vma->vm_flags & VM_SHARED) || - !(vma->vm_flags & VM_WRITE))) { - mss->swap += shmem_swapped; - } else { + if (start || (shmem_swapped && is_cow)) ops = get_smaps_shmem_walk_ops(priv); - } + else + mss->swap += shmem_swapped; } if (!start) From 9ce74bcebdec72b4613407ba714d051ce508dfd5 Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:15 -0700 Subject: [PATCH 0771/1012] proc/task_mmu: remove special-casing of smap_gather_stats() start parameter smap_gather_stats() interprets its start parameter to mean vma->vm_start when it's set to 0. Eliminate this special interpretation and provide two separate functions for a partial and complete VMA walk. Since smap_gather_stats() operates within a single VMA, we can replace walk_page_vma()/walk_page_range() calls with walk_page_range_vma() which is simpler and also can be called while holding per-VMA lock. No functional change intended. Link: https://lore.kernel.org/20260918153318.758387-5-surenb@google.com Suggested-by: Lorenzo Stoakes Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Cc: Jann Horn Cc: Liam R. Howlett (Oracle) Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Usama Arif Cc: Vlastimil Babka --- fs/proc/task_mmu.c | 52 +++++++++++++++++++++++++++++++--------------- 1 file changed, 35 insertions(+), 17 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index dd9cccf8a4892d..f62359ffc3f74b 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -1250,20 +1250,26 @@ get_smaps_shmem_walk_ops(struct proc_maps_private *priv) return &smaps_shmem_walk_vma_lock_ops; } -/* - * Gather mem stats from @vma with the indicated beginning - * address @start, and keep them in @mss. +/** + * smap_gather_stats_range() - Gather mem stats from a portion of the @vma. + * @priv: proc maps private state. + * @vma: The VMA to gather stats for. + * @mss: The accumulated stats. + * @start: The address from which to start. * - * Use vm_start of @vma as the beginning address if @start is 0. + * This gathers stats for the portion of the VMA starting at the @start + * address. */ -static void smap_gather_stats(struct proc_maps_private *priv, - struct vm_area_struct *vma, - struct mem_size_stats *mss, unsigned long start) +static void smap_gather_stats_range(struct proc_maps_private *priv, + struct vm_area_struct *vma, + struct mem_size_stats *mss, + unsigned long start) { const struct mm_walk_ops *ops = get_smaps_walk_ops(priv); + const bool is_partial = start > vma->vm_start; /* Invalid start */ - if (start >= vma->vm_end) + if (start < vma->vm_start || start >= vma->vm_end) return; if (vma == get_gate_vma(priv->lock_ctx.mm)) @@ -1282,20 +1288,31 @@ static void smap_gather_stats(struct proc_maps_private *priv, const unsigned long shmem_swapped = shmem_swap_usage(vma); const bool is_cow = vma_is_cow_mapping(vma); - if (start || (shmem_swapped && is_cow)) + if (is_partial || (shmem_swapped && is_cow)) ops = get_smaps_shmem_walk_ops(priv); else mss->swap += shmem_swapped; } - if (!start) - walk_page_vma(vma, ops, mss); - else - walk_page_range(vma->vm_mm, start, vma->vm_end, ops, mss); + walk_page_range_vma(vma, start, vma->vm_end, ops, mss); reacquire_rcu(priv); } +/** + * smap_gather_stats() - Gather mem stats from the entire @vma. + * @priv: proc maps private state. + * @vma: The VMA to gather stats for. + * @mss: The accumulated stats. + * + * This gathers stats for the whole of the VMA. + */ +static void smap_gather_stats(struct proc_maps_private *priv, + struct vm_area_struct *vma, struct mem_size_stats *mss) +{ + smap_gather_stats_range(priv, vma, mss, vma->vm_start); +} + #define SEQ_PUT_DEC(str, val) \ seq_put_decimal_ull_width(m, str, (val) >> 10, 8) @@ -1346,7 +1363,7 @@ static int show_smap(struct seq_file *m, void *v) struct vm_area_struct *vma = v; struct mem_size_stats mss = {}; - smap_gather_stats(priv, vma, &mss, 0); + smap_gather_stats(priv, vma, &mss); show_map_vma(m, vma); @@ -1399,7 +1416,7 @@ static int show_smaps_rollup(struct seq_file *m, void *v) vma_start = vma->vm_start; do { - smap_gather_stats(priv, vma, &mss, 0); + smap_gather_stats(priv, vma, &mss); last_vma_end = vma->vm_end; /* @@ -1458,14 +1475,15 @@ static int show_smaps_rollup(struct seq_file *m, void *v) /* Case 1 and 2 above */ if (vma->vm_start >= last_vma_end) { - smap_gather_stats(priv, vma, &mss, 0); + smap_gather_stats(priv, vma, &mss); last_vma_end = vma->vm_end; continue; } /* Case 4 above */ if (vma->vm_end > last_vma_end) { - smap_gather_stats(priv, vma, &mss, last_vma_end); + smap_gather_stats_range(priv, vma, &mss, + last_vma_end); last_vma_end = vma->vm_end; } } From 03ff35e5f74146fb3d7eba23647a6cdfce230014 Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:16 -0700 Subject: [PATCH 0772/1012] proc/task_mmu: change proc_get_vma() to stop returning gate VMA at the end proc_get_vma() returning gate VMA at the end is desirable for the its current m_start/m_next callers, as they need to report a gate VMA at the end of the address space. This behavior is very specific to these callers and makes proc_get_vma() hard to use for other purposes. Move this usage-specific behavior into the callers themselves so that proc_get_vma() returns either a valid VMA, an error or a NULL when no more VMAs are available. This makes it more generic, simpler and usable in the later patches. Link: https://lore.kernel.org/20260918153318.758387-6-surenb@google.com Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Cc: Jann Horn Cc: Liam R. Howlett (Oracle) Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Usama Arif Cc: Vlastimil Babka --- fs/proc/task_mmu.c | 28 +++++++++++++++++++++++----- 1 file changed, 23 insertions(+), 5 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index f62359ffc3f74b..2a29bfb41ab52b 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -240,9 +240,6 @@ static struct vm_area_struct *proc_get_vma(struct seq_file *m, loff_t *ppos) * found the extended vma with the same vm_start. */ *ppos = vma->vm_end; - } else { - *ppos = SENTINEL_VMA_GATE; - vma = get_gate_vma(priv->lock_ctx.mm); } return vma; @@ -252,6 +249,7 @@ static void *m_start(struct seq_file *m, loff_t *ppos) { struct proc_maps_private *priv = m->private; struct proc_maps_locking_ctx *lock_ctx; + struct vm_area_struct *vma; loff_t last_addr = *ppos; struct mm_struct *mm; @@ -281,19 +279,39 @@ static void *m_start(struct seq_file *m, loff_t *ppos) *ppos = last_addr = priv->last_pos; vma_iter_init(&priv->iter, mm, (unsigned long)last_addr); hold_task_mempolicy(priv); + /* + * If seq_file had to flush its collected data right after m_next() set + * position to SENTINEL_VMA_GATE, m_start() will get that sentinel and + * should return gate_vma without calling proc_get_vma(). + */ if (last_addr == SENTINEL_VMA_GATE) return get_gate_vma(mm); - return proc_get_vma(m, ppos); + vma = proc_get_vma(m, ppos); + if (vma) + return vma; + + /* Return gate VMA at the end */ + *ppos = SENTINEL_VMA_GATE; + return get_gate_vma(mm); } static void *m_next(struct seq_file *m, void *v, loff_t *ppos) { + struct proc_maps_private *priv = m->private; + struct vm_area_struct *vma; + if (*ppos == SENTINEL_VMA_GATE) { *ppos = SENTINEL_VMA_END; return NULL; } - return proc_get_vma(m, ppos); + vma = proc_get_vma(m, ppos); + if (vma) + return vma; + + /* Return gate VMA at the end */ + *ppos = SENTINEL_VMA_GATE; + return get_gate_vma(priv->lock_ctx.mm); } static void m_stop(struct seq_file *m, void *v) From 3732c0d2de0fd901bb565b87d474de4a0fd4b255 Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:17 -0700 Subject: [PATCH 0773/1012] proc/task_mmu: read proc/pid/smaps_rollup under per-vma lock proc/pid/smaps_rollup can be read using the combination of RCU and VMA read locks, similar to proc/pid/{maps|smaps|numa_maps}. RCU is required to safely traverse the VMA tree and VMA lock stabilizes the VMA being processed and the pagetable walk. Note that we have to keep the logic to drop mmap_lock on contention because even when using per-VMA locks we might have to fall back to holding the mmap_lock. Running Paul's contention benchmark [1] shows considerable improvement both in median and in the worst case latencies: Execution command: run-proc-vs-map.sh --nsamples 20 --rawdata -- \ --busyduration 2 --procfile smaps_rollup Baseline: Median Minimum Maximum 0.174 0.161 2.553 0.174 0.164 2.663 0.174 0.165 2.664 0.174 0.166 2.679 0.174 0.167 2.691 0.174 0.168 2.704 0.174 0.169 2.729 0.174 0.172 2.741 0.174 0.174 2.745 0.174 0.174 2.755 0.174 0.175 2.790 0.174 0.177 2.809 0.174 0.179 3.096 0.174 0.183 3.144 0.174 0.184 3.158 0.174 0.185 3.175 0.174 0.185 4.568 0.174 0.198 4.821 0.174 0.214 5.143 0.174 0.251 5.220 Patched: Median Minimum Maximum 0.007 0.007 1.952 0.007 0.007 1.955 0.007 0.007 1.955 0.007 0.007 1.955 0.007 0.007 1.957 0.007 0.007 1.969 0.007 0.007 2.065 0.007 0.007 2.075 0.007 0.007 2.146 0.007 0.007 2.195 0.007 0.007 2.223 0.007 0.007 2.259 0.007 0.007 2.488 0.007 0.007 2.562 0.007 0.007 2.599 0.007 0.007 2.697 0.007 0.007 3.030 0.007 0.007 3.075 0.007 0.007 3.145 0.007 0.007 3.225 Remove now unused lock_ctx_mm() and move unlock_ctx_vma() next to unlock_ctx_mm() as they are logically related. Remove a long comment about 4 cases that we handle when dropping the mmap lock in the middle of VMA walk due to contention. The first 3 cases explained there are handled naturally and only case 4 needs to be handled in a special way, which is done in smap_gather_stats() by gathering stats from the portion of the VMA that has not yet been processed. For posterity, moving this comment here: After dropping the lock, there are four cases to consider. See the following example for explanation. +------+------+-----------+ | VMA1 | VMA2 | VMA3 | +------+------+-----------+ | | | | 4k 8k 16k 400k Suppose we drop the lock after reading VMA2 due to contention, then we get: last_vma_end = 16k 1) VMA2 is freed, but VMA3 exists: vma_next(vmi) will return VMA3. In this case, just continue from VMA3. 2) VMA2 still exists: vma_next(vmi) will return VMA3. In this case, just continue from VMA3. 3) No more VMAs can be found: vma_next(vmi) will return NULL. No more things to do, just break. 4) (last_vma_end - 1) is the middle of a vma (VMA'): vma_next(vmi) will return VMA' whose range contains last_vma_end. Iterate VMA' from last_vma_end. Link: https://lore.kernel.org/20260918153318.758387-7-surenb@google.com Link: https://github.com/paulmckrcu/proc-mmap_sem-test [1] Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Cc: David Hildenbrand (Arm) Cc: Jann Horn Cc: Liam R. Howlett (Oracle) Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Usama Arif Cc: Vlastimil Babka --- fs/proc/task_mmu.c | 159 ++++++++++++++++++--------------------------- 1 file changed, 63 insertions(+), 96 deletions(-) diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c index 2a29bfb41ab52b..44147ba0b899da 100644 --- a/fs/proc/task_mmu.c +++ b/fs/proc/task_mmu.c @@ -130,30 +130,12 @@ static void release_task_mempolicy(struct proc_maps_private *priv) } #endif -#ifdef CONFIG_PROC_PAGE_MONITOR -static int lock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) -{ - int ret = mmap_read_lock_killable(lock_ctx->mm); - - if (!ret) - lock_ctx->mmap_locked = true; - - return ret; -} -#endif - static void unlock_ctx_mm(struct proc_maps_locking_ctx *lock_ctx) { mmap_read_unlock(lock_ctx->mm); lock_ctx->mmap_locked = false; } -static void reset_lock_ctx(struct proc_maps_locking_ctx *lock_ctx) -{ - lock_ctx->locked_vma = NULL; - lock_ctx->mmap_locked = false; -} - static void unlock_ctx_vma(struct proc_maps_locking_ctx *lock_ctx) { if (lock_ctx->locked_vma) { @@ -162,6 +144,12 @@ static void unlock_ctx_vma(struct proc_maps_locking_ctx *lock_ctx) } } +static void reset_lock_ctx(struct proc_maps_locking_ctx *lock_ctx) +{ + lock_ctx->locked_vma = NULL; + lock_ctx->mmap_locked = false; +} + static struct vm_area_struct *get_next_vma(struct proc_maps_private *priv, loff_t last_pos) { @@ -1406,12 +1394,14 @@ static int show_smap(struct seq_file *m, void *v) static int show_smaps_rollup(struct seq_file *m, void *v) { struct proc_maps_private *priv = m->private; + struct proc_maps_locking_ctx *lock_ctx = &priv->lock_ctx; + struct mm_struct *mm = lock_ctx->mm; struct mem_size_stats mss = {}; - struct mm_struct *mm = priv->lock_ctx.mm; + unsigned long last_vma_end = 0; + unsigned long vma_start = 0; struct vm_area_struct *vma; - unsigned long vma_start = 0, last_vma_end = 0; + loff_t pos = 0; int ret = 0; - VMA_ITERATOR(vmi, mm, 0); priv->task = get_proc_task(priv->inode); if (!priv->task) @@ -1422,90 +1412,63 @@ static int show_smaps_rollup(struct seq_file *m, void *v) goto out_put_task; } - ret = lock_ctx_mm(&priv->lock_ctx); - if (ret) - goto out_put_mm; - hold_task_mempolicy(priv); - vma = vma_next(&vmi); + rcu_read_lock(); + reset_lock_ctx(lock_ctx); + vma_iter_init(&priv->iter, mm, 0); + vma = proc_get_vma(m, &pos); if (unlikely(!vma)) goto empty_set; - vma_start = vma->vm_start; - do { - smap_gather_stats(priv, vma, &mss); + if (!IS_ERR(vma)) + vma_start = vma->vm_start; + + while (vma) { + if (IS_ERR(vma)) { + ret = PTR_ERR(vma); + goto out_unlock; + } + + if (vma->vm_start < last_vma_end) { + /* + * After retaking the lock, already reported VMA grew + * or got merged with the next one and we found it + * again. Gather stats for the remaining portion by + * starting at last_vma_end. + */ + smap_gather_stats_range(priv, vma, &mss, last_vma_end); + } else { + /* Found next unreported VMA, start from its beginning */ + smap_gather_stats(priv, vma, &mss); + } last_vma_end = vma->vm_end; /* - * Release mmap_lock temporarily if someone wants to - * access it for write request. + * If the VMA lock is not taken, we hold the often contended + * mmap lock. This can happen if we had to fall back to the + * mmap lock. + * + * To relieve pressure, check if it is indeed contended, then + * temporarily release it. */ - if (mmap_lock_is_contended(mm)) { - vma_iter_invalidate(&vmi); - unlock_ctx_mm(&priv->lock_ctx); - ret = lock_ctx_mm(&priv->lock_ctx); - if (ret) { - release_task_mempolicy(priv); - goto out_put_mm; - } - + if (lock_ctx->mmap_locked && + mmap_lock_is_contended(lock_ctx->mm)) { + unlock_ctx_mm(lock_ctx); /* - * After dropping the lock, there are four cases to - * consider. See the following example for explanation. - * - * +------+------+-----------+ - * | VMA1 | VMA2 | VMA3 | - * +------+------+-----------+ - * | | | | - * 4k 8k 16k 400k - * - * Suppose we drop the lock after reading VMA2 due to - * contention, then we get: - * - * last_vma_end = 16k - * - * 1) VMA2 is freed, but VMA3 exists: - * - * vma_next(vmi) will return VMA3. - * In this case, just continue from VMA3. - * - * 2) VMA2 still exists: - * - * vma_next(vmi) will return VMA3. - * In this case, just continue from VMA3. - * - * 3) No more VMAs can be found: - * - * vma_next(vmi) will return NULL. - * No more things to do, just break. - * - * 4) (last_vma_end - 1) is the middle of a vma (VMA'): - * - * vma_next(vmi) will return VMA' whose range - * contains last_vma_end. - * Iterate VMA' from last_vma_end. + * Even though we previously fell back to mmap lock, + * we try taking VMA lock for the next VMA, since it + * might not be under modification. In the worst case + * we will fall back to mmap lock again. */ - vma = vma_next(&vmi); - /* Case 3 above */ - if (!vma) - break; - - /* Case 1 and 2 above */ - if (vma->vm_start >= last_vma_end) { - smap_gather_stats(priv, vma, &mss); - last_vma_end = vma->vm_end; - continue; - } - - /* Case 4 above */ - if (vma->vm_end > last_vma_end) { - smap_gather_stats_range(priv, vma, &mss, - last_vma_end); - last_vma_end = vma->vm_end; - } + rcu_read_lock(); + reset_lock_ctx(lock_ctx); + /* Resume from the last position. */ + pos = last_vma_end; + vma_iter_init(&priv->iter, mm, pos); } - } for_each_vma(vmi, vma); + vma = proc_get_vma(m, &pos); + } empty_set: show_vma_header_prefix(m, vma_start, last_vma_end, 0, 0, 0, 0); @@ -1514,10 +1477,14 @@ static int show_smaps_rollup(struct seq_file *m, void *v) __show_smap(m, &mss, true); +out_unlock: + if (lock_ctx->mmap_locked) { + unlock_ctx_mm(lock_ctx); + } else { + unlock_ctx_vma(lock_ctx); + rcu_read_unlock(); + } release_task_mempolicy(priv); - unlock_ctx_mm(&priv->lock_ctx); - -out_put_mm: mmput(mm); out_put_task: put_task_struct(priv->task); From b0d39635fda5d0566cb2b053b9691d08cc4c4339 Mon Sep 17 00:00:00 2001 From: Suren Baghdasaryan Date: Fri, 18 Sep 2026 08:33:18 -0700 Subject: [PATCH 0774/1012] selftests/proc: add /proc/pid/smaps_rollup tearing tests During tearing tests, smaps_rollup Pss* metrics should stay constant. Extend /proc/pid/smaps tearing tests to also check for smaps_rollup consistency. Link: https://lore.kernel.org/20260918153318.758387-8-surenb@google.com Signed-off-by: Suren Baghdasaryan Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes (ARM) Cc: David Hildenbrand (Arm) Cc: Jann Horn Cc: Liam R. Howlett (Oracle) Cc: Matthew Wilcox (Oracle) Cc: "Paul E . McKenney" Cc: Pedro Falcato Cc: Usama Arif Cc: Vlastimil Babka --- tools/testing/selftests/proc/proc-maps-race.c | 186 +++++++++++++++++- 1 file changed, 181 insertions(+), 5 deletions(-) diff --git a/tools/testing/selftests/proc/proc-maps-race.c b/tools/testing/selftests/proc/proc-maps-race.c index 415eccb7046848..bf4c5073f6fc86 100644 --- a/tools/testing/selftests/proc/proc-maps-race.c +++ b/tools/testing/selftests/proc/proc-maps-race.c @@ -80,6 +80,61 @@ enum maps_file { struct vma_modifier_info; +enum smaps_rollup_stat { + Rss, + Pss, + Pss_Dirty, + Pss_Anon, + Pss_File, + Pss_Shmem, + Shared_Clean, + Shared_Dirty, + Private_Clean, + Private_Dirty, + Referenced, + Anonymous, + KSM, + LazyFree, + AnonHugePages, + ShmemPmdMapped, + FilePmdMapped, + Shared_Hugetlb, + Private_Hugetlb, + Swap, + SwapPss, + Locked, + RollupFieldCount +}; + +static const char *smaps_rollup_stat_names[RollupFieldCount] = { + "Rss", + "Pss", + "Pss_Dirty", + "Pss_Anon", + "Pss_File", + "Pss_Shmem", + "Shared_Clean", + "Shared_Dirty", + "Private_Clean", + "Private_Dirty", + "Referenced", + "Anonymous", + "KSM", + "LazyFree", + "AnonHugePages", + "ShmemPmdMapped", + "FilePmdMapped", + "Shared_Hugetlb", + "Private_Hugetlb", + "Swap", + "SwapPss", + "Locked", +}; + +struct smaps_rollup_stats { + unsigned long values[RollupFieldCount]; +}; + FIXTURE(proc_maps_race) { struct vma_modifier_info *mod_info; @@ -91,6 +146,7 @@ FIXTURE(proc_maps_race) enum maps_file maps_file; int shared_mem_size; int skip_pages; + int rollup_fd; int page_size; int vma_count; bool verbose; @@ -132,12 +188,12 @@ struct vma_modifier_info { void *child_mapped_addr[]; }; -static bool read_page(FIXTURE_DATA(proc_maps_race) *self, +static bool read_page(FIXTURE_DATA(proc_maps_race) *self, int fd, struct page_content *page) { ssize_t bytes_read; - bytes_read = read(self->maps_fd, page->data, self->page_size); + bytes_read = read(fd, page->data, self->page_size); if (bytes_read <= 0) return false; @@ -175,7 +231,7 @@ static int locate_containing_page(FIXTURE_DATA(proc_maps_race) *self, char *curr_pos; char *end_pos; - if (!read_page(self, &self->page1)) + if (!read_page(self, self->maps_fd, &self->page1)) return -1; curr_pos = self->page1.data; @@ -205,10 +261,11 @@ static bool read_two_pages(FIXTURE_DATA(proc_maps_race) *self) return false; for (int i = 0; i < self->skip_pages; i++) - if (!read_page(self, &self->page1)) + if (!read_page(self, self->maps_fd, &self->page1)) return false; - return read_page(self, &self->page1) && read_page(self, &self->page2); + return read_page(self, self->maps_fd, &self->page1) && + read_page(self, self->maps_fd, &self->page2); } static void copy_line(const char *line_start, const char *line_end, @@ -317,6 +374,61 @@ static bool read_boundary_lines(FIXTURE_DATA(proc_maps_race) *self, &first_line->end_addr) == 2; } +static bool parse_smaps_rollup(FIXTURE_DATA(proc_maps_race) *self, + struct smaps_rollup_stats *stats) +{ + unsigned int dev_maj, dev_min, inode; + unsigned long start, end, offs; + unsigned long value; + char name[32], perm[5]; + char *curr_pos; + char *end_pos; + char *line_end; + + if (lseek(self->rollup_fd, 0, SEEK_SET) < 0) + return false; + + if (!read_page(self, self->rollup_fd, &self->page1)) + return false; + + curr_pos = self->page1.data; + end_pos = self->page1.data + self->page1.size; + + line_end = strchr(curr_pos, '\n'); + if (!line_end) + return false; + + if (sscanf(curr_pos, "%lx-%lx %4s %lx %u:%u %u %31s", + &start, &end, perm, &offs, &dev_maj, &dev_min, &inode, name) != 8) + return false; + + if (strcmp(name, "[rollup]")) + return false; + + for (int stat = 0; stat < ARRAY_SIZE(smaps_rollup_stat_names); stat++) { + int len; + + curr_pos = line_end + 1; + if (curr_pos >= end_pos) + return false; + + line_end = strchr(curr_pos, '\n'); + if (!line_end) + return false; + + if (sscanf(curr_pos, "%31s %lu kB", name, &value) != 2) + return false; + + len = strlen(name); + if (name[len - 1] != ':' || strncmp(name, smaps_rollup_stat_names[stat], len - 1)) + return false; + + stats->values[stat] = value; + } + + return true; +} + /* Thread synchronization routines */ static void wait_for_state(struct vma_modifier_info *mod_info, enum test_state state) { @@ -397,6 +509,40 @@ static bool print_boundaries_on(bool condition, const char *title, return condition; } +static void print_smaps_rollup_stats(const char *title, FIXTURE_DATA(proc_maps_race) *self, + struct smaps_rollup_stats *stats) +{ + printf("%s", title); + for (int stat = 0; stat < ARRAY_SIZE(smaps_rollup_stat_names); stat++) + printf("%64s %lu kB\n", smaps_rollup_stat_names[stat], stats->values[stat]); +} + +static bool cmp_smaps_rollup_stat(struct smaps_rollup_stats *s1, + struct smaps_rollup_stats *s2, enum smaps_rollup_stat stat) +{ + return s1->values[stat] == s2->values[stat]; +} + +static bool compare_smaps_rollup(FIXTURE_DATA(proc_maps_race) *self, + struct smaps_rollup_stats *expected, + struct smaps_rollup_stats *actual) +{ + /* + * Clean/dirty metrics might change but Pss-related ones + * should stay constant. + */ + if (cmp_smaps_rollup_stat(expected, actual, Pss) && + cmp_smaps_rollup_stat(expected, actual, Pss_Anon) && + cmp_smaps_rollup_stat(expected, actual, Pss_File) && + cmp_smaps_rollup_stat(expected, actual, Pss_Shmem)) + return true; + + print_smaps_rollup_stats("Expected stats:", self, expected); + print_smaps_rollup_stats("Actual stats:", self, actual); + + return false; +} + static void report_test_start(const char *name, bool verbose) { if (verbose) @@ -572,6 +718,7 @@ FIXTURE_SETUP(proc_maps_race) unsigned long first_map_addr; unsigned long last_map_addr; unsigned long duration_sec; + char rollup_fname[32]; char fname[32]; self->page_size = (unsigned long)sysconf(_SC_PAGESIZE); @@ -649,6 +796,9 @@ FIXTURE_SETUP(proc_maps_race) break; case SMAPS: sprintf(fname, "/proc/%d/smaps", self->pid); + sprintf(rollup_fname, "/proc/%d/smaps_rollup", self->pid); + self->rollup_fd = open(rollup_fname, O_RDONLY); + ASSERT_NE(self->rollup_fd, -1); break; default: ksft_exit_fail(); @@ -711,6 +861,8 @@ FIXTURE_TEARDOWN(proc_maps_race) for (int i = 0; i < self->vma_count; i++) munmap(self->mod_info->child_mapped_addr[i], self->page_size); close(self->maps_fd); + if (self->maps_file == SMAPS) + close(self->rollup_fd); waitpid(self->pid, &status, 0); munmap(self->mod_info, self->shared_mem_size); } @@ -723,6 +875,7 @@ TEST_F(proc_maps_race, test_maps_tearing_from_split) struct line_content split_first_line; struct line_content restored_last_line; struct line_content restored_first_line; + struct smaps_rollup_stats orig_stats; wait_for_state(mod_info, SETUP_READY); @@ -736,6 +889,8 @@ TEST_F(proc_maps_race, test_maps_tearing_from_split) report_test_start("Tearing from split", self->verbose); ASSERT_TRUE(capture_mod_pattern(self, &split_last_line, &split_first_line, &restored_last_line, &restored_first_line)); + if (self->maps_file == SMAPS) + ASSERT_TRUE(parse_smaps_rollup(self, &orig_stats)); /* Now start concurrent modifications for self->duration_sec */ signal_state(mod_info, TEST_READY); @@ -799,6 +954,11 @@ TEST_F(proc_maps_race, test_maps_tearing_from_split) vma_end == self->last_line.end_addr) || (vma_start == split_first_line.start_addr && vma_end == split_first_line.end_addr)); + } else { + struct smaps_rollup_stats stats; + + ASSERT_TRUE(parse_smaps_rollup(self, &stats)); + ASSERT_TRUE(compare_smaps_rollup(self, &orig_stats, &stats)); } clock_gettime(CLOCK_MONOTONIC_COARSE, &end_ts); end_test_iteration(&end_ts, self->verbose); @@ -817,6 +977,7 @@ TEST_F(proc_maps_race, test_maps_tearing_from_resize) struct line_content shrunk_first_line; struct line_content restored_last_line; struct line_content restored_first_line; + struct smaps_rollup_stats orig_stats; wait_for_state(mod_info, SETUP_READY); @@ -830,6 +991,8 @@ TEST_F(proc_maps_race, test_maps_tearing_from_resize) report_test_start("Tearing from resize", self->verbose); ASSERT_TRUE(capture_mod_pattern(self, &shrunk_last_line, &shrunk_first_line, &restored_last_line, &restored_first_line)); + if (self->maps_file == SMAPS) + ASSERT_TRUE(parse_smaps_rollup(self, &orig_stats)); /* Now start concurrent modifications for self->duration_sec */ signal_state(mod_info, TEST_READY); @@ -880,6 +1043,11 @@ TEST_F(proc_maps_race, test_maps_tearing_from_resize) ASSERT_TRUE(vma_start == self->last_line.start_addr && (vma_end - vma_start == self->page_size * 3 || vma_end - vma_start == self->page_size)); + } else { + struct smaps_rollup_stats stats; + + ASSERT_TRUE(parse_smaps_rollup(self, &stats)); + ASSERT_TRUE(compare_smaps_rollup(self, &orig_stats, &stats)); } clock_gettime(CLOCK_MONOTONIC_COARSE, &end_ts); end_test_iteration(&end_ts, self->verbose); @@ -898,6 +1066,7 @@ TEST_F(proc_maps_race, test_maps_tearing_from_remap) struct line_content remapped_first_line; struct line_content restored_last_line; struct line_content restored_first_line; + struct smaps_rollup_stats orig_stats; wait_for_state(mod_info, SETUP_READY); @@ -911,6 +1080,8 @@ TEST_F(proc_maps_race, test_maps_tearing_from_remap) report_test_start("Tearing from remap", self->verbose); ASSERT_TRUE(capture_mod_pattern(self, &remapped_last_line, &remapped_first_line, &restored_last_line, &restored_first_line)); + if (self->maps_file == SMAPS) + ASSERT_TRUE(parse_smaps_rollup(self, &orig_stats)); /* Now start concurrent modifications for self->duration_sec */ signal_state(mod_info, TEST_READY); @@ -963,6 +1134,11 @@ TEST_F(proc_maps_race, test_maps_tearing_from_remap) vma_end - vma_start == self->page_size * 3) || (vma_start == self->last_line.start_addr + self->page_size && vma_end - vma_start == self->page_size)); + } else { + struct smaps_rollup_stats stats; + + ASSERT_TRUE(parse_smaps_rollup(self, &stats)); + ASSERT_TRUE(compare_smaps_rollup(self, &orig_stats, &stats)); } clock_gettime(CLOCK_MONOTONIC_COARSE, &end_ts); end_test_iteration(&end_ts, self->verbose); From e39bb1077ade3375e0300c76891e73ec7bb6fea6 Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Sat, 19 Sep 2026 01:56:40 +0900 Subject: [PATCH 0775/1012] mm/swapops: remove unused is_hwpoison_entry() Since commit 93976a20345b ("mm: eliminate further swapops predicates"), is_hwpoison_entry() has no callers. So remove it. No functional change. Link: https://lore.kernel.org/20260918165642.1014988-1-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Reviewed-by: SJ Park Reviewed-by: Zenghui Yu (Huawei) Reviewed-by: Barry Song Acked-by: Chris Li Cc: Baoquan He Cc: Kairui Song Cc: Kemeng Shi Cc: Nhat Pham Cc: Youngjun Park --- include/linux/swapops.h | 9 --------- 1 file changed, 9 deletions(-) diff --git a/include/linux/swapops.h b/include/linux/swapops.h index e7d0d529f3e0f9..603f9e3f909cd1 100644 --- a/include/linux/swapops.h +++ b/include/linux/swapops.h @@ -261,11 +261,6 @@ static inline swp_entry_t make_hwpoison_entry(struct page *page) return swp_entry(SWP_HWPOISON, page_to_pfn(page)); } -static inline int is_hwpoison_entry(swp_entry_t entry) -{ - return swp_type(entry) == SWP_HWPOISON; -} - #else static inline swp_entry_t make_hwpoison_entry(struct page *page) @@ -273,10 +268,6 @@ static inline swp_entry_t make_hwpoison_entry(struct page *page) return swp_entry(0, 0); } -static inline int is_hwpoison_entry(swp_entry_t swp) -{ - return 0; -} #endif typedef unsigned long pte_marker; From 0142aae3f8d468ac234fed7dee6f36aabb9dfc7a Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 16:52:29 +0530 Subject: [PATCH 0776/1012] selftests/mm: make file helpers return errors Patch series "selftests/mm: separate GUP microbenchmarking from functional testing", v11. gup_test.c currently serves two separate purposes: benchmarking (GUP_FAST_BENCHMARK, PIN_FAST_BENCHMARK and PIN_LONGTERM_BENCHMARK) and functional testing (GUP_BASIC_TEST, PIN_BASIC_TEST and DUMP_USER_PAGES_TEST). Keeping both in one program makes the functional tests harder to run and report individually, while run_vmtests.sh has to invoke the program repeatedly with different options. Separate these roles into tools/mm/gup_bench for benchmarking and tools/testing/selftests/mm/gup for functional testing. Move the shared file and hugepage helpers to tools/lib/mm/ so both programs can use them without duplicating the implementation. Patch 1 makes read_file(), write_file(), read_num(), write_num() and write_num_ignore_einval() return errors to their callers instead of exiting. It also makes read_num() reject negative and malformed values and updates the existing callers to handle failures. Patch 2 moves these file helpers from vm_util.c to tools/lib/mm/. It keeps them available to the mm selftests through vm_util.h and adjusts the selftests build accordingly. Patch 3 moves hugepage_settings.[ch] from selftests/mm to tools/lib/mm/. It also removes its kselftest dependency while preserving TAP-compatible diagnostics for selftest users. Patch 4 moves the existing gup_test implementation from selftests/mm to tools/mm as gup_bench. This keeps the code movement separate from the subsequent changes and makes it easier to review. Patch 5 removes the functional test modes and kselftest dependency from gup_bench. When run without arguments, it performs one GUP_FAST benchmark using the existing defaults instead of running the whole matrix. Other benchmark configurations can be selected through command-line options. Patch 6 adds a new harness-based GUP selftest. It covers THP, non-THP and HugeTLB mappings across private/shared and read/write variants. For each variant, it tests get_user_pages(), get_user_pages_fast(), pin_user_pages(), pin_user_pages_fast() and long-term pinning modes using four batch sizes. The HugeTLB variants share a one-time setup of two hugeTLB pages. This patch (of 6): Change read_file(), write_file(), read_num(), write_num() and write_num_ignore_einval() in vm_util.c to report failures to callers instead of exiting from the helper. Make read_file() return a negative errno on failure and 0 on success, so callers can distinguish a successful read from an I/O error. Also make read_num() reject negative and malformed values. Keep write_num_ignore_einval() silent for -EINVAL while returning other errors to its caller. Update callers to print diagnostics and fail wherever required. Modify a comment which implies write_num() uses ksft_exit_fail_msg(). Also add a helper print_file_access_error() in hugepage_settings.c to print TAP-compatible errors without a kselftest dependency. This prepares the helpers to be moved to tools/lib/mm without a kselftest dependency. Link: https://lore.kernel.org/20260918112234.195857-1-sarthak.sharma@arm.com Link: https://lore.kernel.org/20260918112234.195857-2-sarthak.sharma@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Muhammad Usama Anjum Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- .../testing/selftests/mm/hugepage_settings.c | 98 +++++++++++--- .../selftests/mm/hugetlb-soft-offline.c | 16 ++- tools/testing/selftests/mm/khugepaged.c | 14 +- .../selftests/mm/split_huge_page_test.c | 5 +- tools/testing/selftests/mm/vm_util.c | 122 ++++++++++++------ tools/testing/selftests/mm/vm_util.h | 8 +- 6 files changed, 192 insertions(+), 71 deletions(-) diff --git a/tools/testing/selftests/mm/hugepage_settings.c b/tools/testing/selftests/mm/hugepage_settings.c index 584054736ce99f..9a63420d0744fb 100644 --- a/tools/testing/selftests/mm/hugepage_settings.c +++ b/tools/testing/selftests/mm/hugepage_settings.c @@ -8,6 +8,7 @@ #include #include #include +#include #include "vm_util.h" #include "hugepage_settings.h" @@ -48,6 +49,11 @@ static const char * const shmem_enabled_strings[] = { NULL }; +static void print_file_access_error(const char *path, int ret) +{ + printf("# %s: %s (%d)\n", path, strerror(-ret), -ret); +} + int thp_read_string(const char *name, const char * const strings[]) { char path[PATH_MAX]; @@ -61,8 +67,9 @@ int thp_read_string(const char *name, const char * const strings[]) exit(EXIT_FAILURE); } - if (!read_file(path, buf, sizeof(buf))) { - perror(path); + ret = read_file(path, buf, sizeof(buf)); + if (ret) { + print_file_access_error(path, ret); exit(EXIT_FAILURE); } @@ -103,12 +110,17 @@ void thp_write_string(const char *name, const char *val) printf("%s: Pathname is too long\n", __func__); exit(EXIT_FAILURE); } - write_file(path, val, strlen(val) + 1); + ret = write_file(path, val, strlen(val) + 1); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } } unsigned long thp_read_num(const char *name) { char path[PATH_MAX]; + unsigned long num; int ret; ret = snprintf(path, PATH_MAX, THP_SYSFS "%s", name); @@ -116,7 +128,13 @@ unsigned long thp_read_num(const char *name) printf("%s: Pathname is too long\n", __func__); exit(EXIT_FAILURE); } - return read_num(path); + ret = read_num(path, &num); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } + + return num; } void thp_write_num(const char *name, unsigned long num) @@ -129,7 +147,11 @@ void thp_write_num(const char *name, unsigned long num) printf("%s: Pathname is too long\n", __func__); exit(EXIT_FAILURE); } - write_num(path, num); + ret = write_num(path, num); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } } void thp_read_settings(struct thp_settings *settings) @@ -157,8 +179,15 @@ void thp_read_settings(struct thp_settings *settings) .max_ptes_shared = thp_read_num("khugepaged/max_ptes_shared"), .pages_to_scan = thp_read_num("khugepaged/pages_to_scan"), }; - if (dev_queue_read_ahead_path[0]) - settings->read_ahead_kb = read_num(dev_queue_read_ahead_path); + if (dev_queue_read_ahead_path[0]) { + int ret = read_num(dev_queue_read_ahead_path, + &settings->read_ahead_kb); + + if (ret) { + print_file_access_error(dev_queue_read_ahead_path, ret); + exit(EXIT_FAILURE); + } + } for (i = 0; i < NR_ORDERS; i++) { if (!((1 << i) & orders)) { @@ -208,8 +237,15 @@ void thp_write_settings(struct thp_settings *settings) thp_write_num("khugepaged/max_ptes_shared", khugepaged->max_ptes_shared); thp_write_num("khugepaged/pages_to_scan", khugepaged->pages_to_scan); - if (dev_queue_read_ahead_path[0]) - write_num(dev_queue_read_ahead_path, settings->read_ahead_kb); + if (dev_queue_read_ahead_path[0]) { + int ret = write_num(dev_queue_read_ahead_path, + settings->read_ahead_kb); + + if (ret) { + print_file_access_error(dev_queue_read_ahead_path, ret); + exit(EXIT_FAILURE); + } + } for (i = 0; i < NR_ORDERS; i++) { if (!((1 << i) & orders)) @@ -307,8 +343,15 @@ static unsigned long __thp_supported_orders(bool is_shmem) } ret = read_file(path, buf, sizeof(buf)); - if (ret) - orders |= 1UL << i; + if (ret) { + if (ret != -ENOENT) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } + continue; + } + + orders |= 1UL << i; } return orders; @@ -382,8 +425,7 @@ int detect_hugetlb_page_sizes(unsigned long sizes[], int max) if (sscanf(entry->d_name, "hugepages-%zukB", &kb) != 1) continue; sizes[count++] = kb * 1024; - ksft_print_msg("[INFO] detected hugetlb page size: %zu KiB\n", - kb); + printf("# [INFO] detected hugetlb page size: %zu KiB\n", kb); } closedir(dir); return count; @@ -425,28 +467,49 @@ static void hugetlb_sysfs_path(char *buf, size_t buflen, unsigned long hugetlb_nr_pages(unsigned long size) { char path[PATH_MAX]; + unsigned long nr; + int ret; hugetlb_sysfs_path(path, sizeof(path), size, "nr_hugepages"); - return read_num(path); + ret = read_num(path, &nr); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } + + return nr; } void hugetlb_set_nr_pages(unsigned long size, unsigned long nr) { char path[PATH_MAX]; + int ret; hugetlb_sysfs_path(path, sizeof(path), size, "nr_hugepages"); - write_num_ignore_einval(path, nr); + ret = write_num_ignore_einval(path, nr); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } } unsigned long hugetlb_free_pages(unsigned long size) { char path[PATH_MAX]; + unsigned long nr; + int ret; hugetlb_sysfs_path(path, sizeof(path), size, "free_hugepages"); - return read_num(path); + ret = read_num(path, &nr); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } + + return nr; } unsigned long hugetlb_nr_resv_pages(unsigned long size) @@ -511,7 +574,8 @@ unsigned long hugetlb_setup(unsigned long nr, unsigned long sizes[], return 0; if (nr_enabled > max) { - ksft_print_msg("detected %d huge page sizes, will only test %d\n", nr_enabled, max); + printf("# detected %d huge page sizes, will only test %d\n", + nr_enabled, max); nr_enabled = max; } diff --git a/tools/testing/selftests/mm/hugetlb-soft-offline.c b/tools/testing/selftests/mm/hugetlb-soft-offline.c index 4af9d3db7b5b6f..ffc85b958c6920 100644 --- a/tools/testing/selftests/mm/hugetlb-soft-offline.c +++ b/tools/testing/selftests/mm/hugetlb-soft-offline.c @@ -85,8 +85,7 @@ static unsigned long orig_enable_soft_offline = -1UL; /* * Runs from an atexit handler, so it must not call anything that - * exits on failure: write_num() would re-enter exit() through - * ksft_exit_fail_msg(). + * exits on failure. */ static void restore_enable_soft_offline(void) { @@ -152,7 +151,10 @@ static void test_soft_offline_common(int enable_soft_offline) hugepagesize_kb = file_stat.f_bsize / 1024; ksft_print_msg("Hugepagesize is %ldkB\n", hugepagesize_kb); - write_num(ENABLE_SOFT_OFFLINE_PATH, enable_soft_offline); + ret = write_num(ENABLE_SOFT_OFFLINE_PATH, enable_soft_offline); + if (ret) + ksft_exit_fail_msg("Failed to write to %s: %s\n", + ENABLE_SOFT_OFFLINE_PATH, strerror(-ret)); nr_hugepages_before = hugetlb_nr_default_pages(); @@ -189,6 +191,8 @@ static void test_soft_offline_common(int enable_soft_offline) int main(int argc, char **argv) { + int ret; + ksft_print_header(); if (!hugetlb_setup_default(8)) @@ -196,7 +200,11 @@ int main(int argc, char **argv) ksft_set_plan(2); - orig_enable_soft_offline = read_num(ENABLE_SOFT_OFFLINE_PATH); + ret = read_num(ENABLE_SOFT_OFFLINE_PATH, &orig_enable_soft_offline); + if (ret) + ksft_exit_fail_msg("Failed to read %s: %s\n", + ENABLE_SOFT_OFFLINE_PATH, strerror(-ret)); + atexit(restore_enable_soft_offline); test_soft_offline_common(1); diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index f82673f5f6b47e..6daa22f6da2f3f 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -122,6 +122,7 @@ static void get_finfo(const char *dir) char buf[1 << 10]; char path[PATH_MAX]; char *str, *end; + int ret; finfo.dir = dir; if (stat(finfo.dir, &path_stat)) @@ -142,8 +143,9 @@ static void get_finfo(const char *dir) major(path_stat.st_dev), minor(path_stat.st_dev)) >= sizeof(path)) ksft_exit_fail_msg("%s: Pathname is too long\n", __func__); - if (!read_file(path, buf, sizeof(buf))) - ksft_exit_fail_perror("read_file(uevent)"); + ret = read_file(path, buf, sizeof(buf)); + if (ret) + ksft_exit_fail_msg("read_file(%s): %s\n", path, strerror(-ret)); if (strstr(buf, "DEVTYPE=disk")) { /* Found it */ if (snprintf(finfo.dev_queue_read_ahead_path, @@ -324,7 +326,7 @@ static void *file_setup_area_common(int nr_hpages, enum file_setup_ops setup) { const int open_opt = setup == FILE_SETUP_READ_ONLY_FS ? O_RDONLY : O_RDWR; const int mmap_prot = setup == FILE_SETUP_READ_ONLY_FS ? PROT_READ : (PROT_READ | PROT_WRITE); - int fd; + int fd, ret; void *p; unsigned long size; @@ -362,7 +364,11 @@ static void *file_setup_area_common(int nr_hpages, enum file_setup_ops setup) ksft_exit_fail_perror("mmap()"); /* Drop page cache */ - write_file("/proc/sys/vm/drop_caches", "3", 2); + ret = write_file("/proc/sys/vm/drop_caches", "3", 2); + if (ret) + ksft_exit_fail_msg("write_file(drop_caches): %s\n", + strerror(-ret)); + success("OK"); return p; } diff --git a/tools/testing/selftests/mm/split_huge_page_test.c b/tools/testing/selftests/mm/split_huge_page_test.c index c01d227d7fd6dd..a30927514b4f6c 100644 --- a/tools/testing/selftests/mm/split_huge_page_test.c +++ b/tools/testing/selftests/mm/split_huge_page_test.c @@ -145,7 +145,10 @@ static void write_debugfs(const char *fmt, ...) if (ret >= INPUT_MAX) ksft_exit_fail_msg("%s: Debugfs input is too long\n", __func__); - write_file(SPLIT_DEBUGFS, input, ret + 1); + ret = write_file(SPLIT_DEBUGFS, input, ret + 1); + if (ret) + ksft_exit_fail_msg("write_file(%s): %s\n", SPLIT_DEBUGFS, + strerror(-ret)); } static char *allocate_zero_filled_hugepage(size_t len) diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index 4821a356303631..d04c904e76c35e 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -887,109 +887,149 @@ int unpoison_memory(unsigned long pfn) int read_file(const char *path, char *buf, size_t buflen) { - int fd; + int fd, err; ssize_t numread; fd = open(path, O_RDONLY); if (fd == -1) - return 0; + return -errno; numread = read(fd, buf, buflen - 1); if (numread < 1) { + err = numread ? errno : ENODATA; close(fd); - return 0; + return -err; } buf[numread] = '\0'; close(fd); - return (unsigned int) numread; + return 0; } -static void __write_file(const char *path, const char *buf, size_t buflen, bool ignore_einval) +int write_file(const char *path, const char *buf, size_t buflen) { int fd, saved_errno; ssize_t numwritten; if (buflen < 2) - ksft_exit_fail_msg("Incorrect buffer len: %zu\n", buflen); + return -EINVAL; fd = open(path, O_WRONLY); if (fd == -1) - ksft_exit_fail_msg("%s open failed: %s\n", path, strerror(errno)); + return -errno; numwritten = write(fd, buf, buflen - 1); saved_errno = errno; close(fd); - errno = saved_errno; - if (numwritten < 0) { - if (ignore_einval && errno == EINVAL) - return; - ksft_exit_fail_msg("%s write(%.*s) failed: %s\n", path, (int)(buflen - 1), - buf, strerror(errno)); - } - if (numwritten != buflen - 1) - ksft_exit_fail_msg("%s write(%.*s) is truncated, expected %zu bytes, got %zd bytes\n", - path, (int)(buflen - 1), buf, buflen - 1, numwritten); -} -void write_file(const char *path, const char *buf, size_t buflen) -{ - __write_file(path, buf, buflen, /* ignore_einval = */ false); + if (numwritten < 0) + return -saved_errno; + + if (numwritten != (ssize_t)(buflen - 1)) + return -EIO; + + return 0; } -unsigned long read_num(const char *path) +int read_num(const char *path, unsigned long *num) { + unsigned long val; + int ret; char buf[21]; + char *end; - if (!read_file(path, buf, sizeof(buf))) - ksft_exit_fail_perror("read_file()"); + if (!num) + return -EINVAL; - return strtoul(buf, NULL, 10); + ret = read_file(path, buf, sizeof(buf)); + if (ret) + return ret; + + /* Reject signs and leading whitespace that are accepted by strtoul() */ + if (buf[0] < '0' || buf[0] > '9') + return -EINVAL; + + errno = 0; + val = strtoul(buf, &end, 10); + if (errno) + return -errno; + + /* Only allow a newline after the number */ + if (*end == '\n') + end++; + + if (*end != '\0') + return -EINVAL; + + *num = val; + return 0; } -static void __write_num(const char *path, unsigned long num, bool ignore_einval) +int write_num(const char *path, unsigned long num) { char buf[21]; sprintf(buf, "%lu", num); - __write_file(path, buf, strlen(buf) + 1, ignore_einval); + return write_file(path, buf, strlen(buf) + 1); } -void write_num(const char *path, unsigned long num) +int write_num_ignore_einval(const char *path, unsigned long num) { - return __write_num(path, num, /* ignore_einval = */ false); -} + int ret; -void write_num_ignore_einval(const char *path, unsigned long num) -{ - return __write_num(path, num, /* ignore_einval = */ true); + ret = write_num(path, num); + return ret == -EINVAL ? 0 : ret; } static unsigned long shmall, shmmax; void __shm_limits_restore(void) { - if (shmmax) - write_num("/proc/sys/kernel/shmmax", shmmax); - if (shmall) - write_num("/proc/sys/kernel/shmall", shmall); + int ret; + + if (shmmax) { + ret = write_num("/proc/sys/kernel/shmmax", shmmax); + if (ret < 0) + ksft_exit_fail_msg("Failed to restore shmmax: %s\n", + strerror(-ret)); + } + if (shmall) { + ret = write_num("/proc/sys/kernel/shmall", shmall); + if (ret < 0) + ksft_exit_fail_msg("Failed to restore shmall: %s\n", + strerror(-ret)); + } } void shm_limits_prepare(unsigned long length) { unsigned long nr = length / psize(); unsigned long val; + int ret; + + ret = read_num("/proc/sys/kernel/shmmax", &val); + if (ret < 0) + ksft_exit_fail_msg("Failed to read /proc/sys/kernel/shmmax: %s\n", + strerror(-ret)); - val = read_num("/proc/sys/kernel/shmmax"); if (val < length) { - write_num("/proc/sys/kernel/shmmax", length); + ret = write_num("/proc/sys/kernel/shmmax", length); + if (ret < 0) + ksft_exit_fail_msg("Failed to write %lu to /proc/sys/kernel/shmmax: %s\n", + length, strerror(-ret)); shmmax = val; } - val = read_num("/proc/sys/kernel/shmall"); + ret = read_num("/proc/sys/kernel/shmall", &val); + if (ret < 0) + ksft_exit_fail_msg("Failed to read /proc/sys/kernel/shmall: %s\n", + strerror(-ret)); if (val < nr) { - write_num("/proc/sys/kernel/shmall", nr); + ret = write_num("/proc/sys/kernel/shmall", nr); + if (ret < 0) + ksft_exit_fail_msg("Failed to write %lu to /proc/sys/kernel/shmall: %s\n", + nr, strerror(-ret)); shmall = val; } } diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index 9a49af88702e4c..62f6f5b4264924 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -166,11 +166,11 @@ int unpoison_memory(unsigned long pfn); #define PAGEMAP_PRESENT(ent) (((ent) & (1ull << 63)) != 0) #define PAGEMAP_PFN(ent) ((ent) & ((1ull << 55) - 1)) -void write_file(const char *path, const char *buf, size_t buflen); +int write_file(const char *path, const char *buf, size_t buflen); int read_file(const char *path, char *buf, size_t buflen); -unsigned long read_num(const char *path); -void write_num(const char *path, unsigned long num); -void write_num_ignore_einval(const char *path, unsigned long num); +int read_num(const char *path, unsigned long *num); +int write_num(const char *path, unsigned long num); +int write_num_ignore_einval(const char *path, unsigned long num); void shm_limits_prepare(unsigned long length); void __shm_limits_restore(void); From 91aeb5710f6aee2f31ba35697fd36b5ec51f201e Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 17:40:50 +0530 Subject: [PATCH 0777/1012] selftests-mm-make-file-helpers-return-errors-fix Convert hugetlb_nr_resv_pages(), which was missed when read_num() changed to return an error and store the parsed value through an output pointer. Link: https://lore.kernel.org/937939c3-ae9a-4148-a601-0f8876216423@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: David Hildenbrand (Arm) Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Muhammad Usama Anjum Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- tools/testing/selftests/mm/hugepage_settings.c | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/hugepage_settings.c b/tools/testing/selftests/mm/hugepage_settings.c index 9a63420d0744fb..6f3abd35738592 100644 --- a/tools/testing/selftests/mm/hugepage_settings.c +++ b/tools/testing/selftests/mm/hugepage_settings.c @@ -515,10 +515,18 @@ unsigned long hugetlb_free_pages(unsigned long size) unsigned long hugetlb_nr_resv_pages(unsigned long size) { char path[PATH_MAX]; + unsigned long nr; + int ret; hugetlb_sysfs_path(path, sizeof(path), size, "resv_hugepages"); - return read_num(path); + ret = read_num(path, &nr); + if (ret) { + print_file_access_error(path, ret); + exit(EXIT_FAILURE); + } + + return nr; } static bool __hugetlb_setup(unsigned long size, unsigned long nr) From ed78fa3de2415072452b834d598eace70f7348ef Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 16:52:30 +0530 Subject: [PATCH 0778/1012] tools/lib/mm: add shared file helpers Move read_file(), write_file(), read_num(), write_num() and write_num_ignore_einval() out of tools/testing/selftests/mm/vm_util.c into a new shared helper under tools/lib/mm/. These helpers are used by mm selftests today and will also be needed by shared hugepage helpers in subsequent patches. Move them to a generic location so they can be reused outside selftests as well. Keep the helpers exposed to mm selftests through vm_util.h by including the new shared header there, and link the new helper into the selftests/mm build. Update the explicit x86 protection_keys 32-bit and 64-bit build rules to preserve prerequisite paths, now that file_utils.c is built from tools/lib/mm. Add tools/lib/mm/ to the MEMORY MANAGEMENT - MISC entry in MAINTAINERS. Link: https://lore.kernel.org/20260918112234.195857-3-sarthak.sharma@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Muhammad Usama Anjum Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- MAINTAINERS | 1 + tools/lib/mm/file_utils.c | 106 +++++++++++++++++++++++++++ tools/lib/mm/file_utils.h | 13 ++++ tools/testing/selftests/mm/Makefile | 11 +-- tools/testing/selftests/mm/vm_util.c | 97 ------------------------ tools/testing/selftests/mm/vm_util.h | 7 +- 6 files changed, 127 insertions(+), 108 deletions(-) create mode 100644 tools/lib/mm/file_utils.c create mode 100644 tools/lib/mm/file_utils.h diff --git a/MAINTAINERS b/MAINTAINERS index 4689a021006002..dfd9f948390708 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17289,6 +17289,7 @@ F: mm/mapping_dirty_helpers.c F: mm/page_idle.c F: mm/pgalloc-track.h F: mm/process_vm_access.c +F: tools/lib/mm/ F: tools/testing/selftests/mm/ MEMORY MANAGEMENT - NUMA MEMBLOCKS AND NUMA EMULATION diff --git a/tools/lib/mm/file_utils.c b/tools/lib/mm/file_utils.c new file mode 100644 index 00000000000000..9b2237e9823e98 --- /dev/null +++ b/tools/lib/mm/file_utils.c @@ -0,0 +1,106 @@ +// SPDX-License-Identifier: GPL-2.0 +#include +#include +#include +#include +#include +#include + +#include "file_utils.h" + +int read_file(const char *path, char *buf, size_t buflen) +{ + int fd, err; + ssize_t numread; + + fd = open(path, O_RDONLY); + if (fd == -1) + return -errno; + + numread = read(fd, buf, buflen - 1); + if (numread < 1) { + err = numread ? errno : ENODATA; + close(fd); + return -err; + } + + buf[numread] = '\0'; + close(fd); + + return 0; +} + +int write_file(const char *path, const char *buf, size_t buflen) +{ + int fd, saved_errno; + ssize_t numwritten; + + if (buflen < 2) + return -EINVAL; + + fd = open(path, O_WRONLY); + if (fd == -1) + return -errno; + + numwritten = write(fd, buf, buflen - 1); + saved_errno = errno; + close(fd); + + if (numwritten < 0) + return -saved_errno; + + if (numwritten != (ssize_t)(buflen - 1)) + return -EIO; + + return 0; +} + +int read_num(const char *path, unsigned long *num) +{ + unsigned long val; + int ret; + char buf[21]; + char *end; + + if (!num) + return -EINVAL; + + ret = read_file(path, buf, sizeof(buf)); + if (ret) + return ret; + + /* Reject signs and leading whitespace that are accepted by strtoul() */ + if (buf[0] < '0' || buf[0] > '9') + return -EINVAL; + + errno = 0; + val = strtoul(buf, &end, 10); + if (errno) + return -errno; + + /* Only allow a newline after the number */ + if (*end == '\n') + end++; + + if (*end != '\0') + return -EINVAL; + + *num = val; + return 0; +} + +int write_num(const char *path, unsigned long num) +{ + char buf[21]; + + sprintf(buf, "%lu", num); + return write_file(path, buf, strlen(buf) + 1); +} + +int write_num_ignore_einval(const char *path, unsigned long num) +{ + int ret; + + ret = write_num(path, num); + return ret == -EINVAL ? 0 : ret; +} diff --git a/tools/lib/mm/file_utils.h b/tools/lib/mm/file_utils.h new file mode 100644 index 00000000000000..50daa82c2b2b48 --- /dev/null +++ b/tools/lib/mm/file_utils.h @@ -0,0 +1,13 @@ +/* SPDX-License-Identifier: GPL-2.0 */ +#ifndef __MM_FILE_UTILS_H__ +#define __MM_FILE_UTILS_H__ + +#include + +int read_file(const char *path, char *buf, size_t buflen); +int write_file(const char *path, const char *buf, size_t buflen); +int read_num(const char *path, unsigned long *num); +int write_num(const char *path, unsigned long num); +int write_num_ignore_einval(const char *path, unsigned long num); + +#endif diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index d3e9bd67904aa0..c36a9a7089bb26 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -37,7 +37,8 @@ endif # LDLIBS. MAKEFLAGS += --no-builtin-rules -CFLAGS = -Wall -O2 -I $(top_srcdir) $(EXTRA_CFLAGS) $(KHDR_INCLUDES) $(TOOLS_INCLUDES) +CFLAGS = -Wall -O2 -I $(top_srcdir) -I $(top_srcdir)/tools/lib +CFLAGS += $(EXTRA_CFLAGS) $(KHDR_INCLUDES) $(TOOLS_INCLUDES) CFLAGS += -Wunreachable-code LDLIBS = -lrt -lpthread -lm @@ -184,8 +185,8 @@ TEST_FILES += write_hugetlb_memory.sh include ../lib.mk -$(TEST_GEN_PROGS): vm_util.c hugepage_settings.c -$(TEST_GEN_FILES): vm_util.c hugepage_settings.c +$(TEST_GEN_PROGS): vm_util.c hugepage_settings.c $(top_srcdir)/tools/lib/mm/file_utils.c +$(TEST_GEN_FILES): vm_util.c hugepage_settings.c $(top_srcdir)/tools/lib/mm/file_utils.c $(OUTPUT)/uffd-stress: uffd-common.c $(OUTPUT)/uffd-unit-tests: uffd-common.c @@ -214,7 +215,7 @@ $(BINARIES_32): CFLAGS += -m32 -mxsave $(BINARIES_32): LDLIBS += -lrt -ldl -lm $(BINARIES_32): $(OUTPUT)/%_32: %.c $(call msg,CC,,$@) - $(Q)$(CC) $(CFLAGS) $(EXTRA_CFLAGS) $(notdir $^) $(LDLIBS) -o $@ + $(Q)$(CC) $(CFLAGS) $(EXTRA_CFLAGS) $^ $(LDLIBS) -o $@ $(foreach t,$(VMTARGETS),$(eval $(call gen-target-rule-32,$(t)))) endif @@ -223,7 +224,7 @@ $(BINARIES_64): CFLAGS += -m64 -mxsave $(BINARIES_64): LDLIBS += -lrt -ldl $(BINARIES_64): $(OUTPUT)/%_64: %.c $(call msg,CC,,$@) - $(Q)$(CC) $(CFLAGS) $(EXTRA_CFLAGS) $(notdir $^) $(LDLIBS) -o $@ + $(Q)$(CC) $(CFLAGS) $(EXTRA_CFLAGS) $^ $(LDLIBS) -o $@ $(foreach t,$(VMTARGETS),$(eval $(call gen-target-rule-64,$(t)))) endif diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index d04c904e76c35e..f8916b2abc2efb 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -885,103 +885,6 @@ int unpoison_memory(unsigned long pfn) return ret > 0 ? 0 : -errno; } -int read_file(const char *path, char *buf, size_t buflen) -{ - int fd, err; - ssize_t numread; - - fd = open(path, O_RDONLY); - if (fd == -1) - return -errno; - - numread = read(fd, buf, buflen - 1); - if (numread < 1) { - err = numread ? errno : ENODATA; - close(fd); - return -err; - } - - buf[numread] = '\0'; - close(fd); - - return 0; -} - -int write_file(const char *path, const char *buf, size_t buflen) -{ - int fd, saved_errno; - ssize_t numwritten; - - if (buflen < 2) - return -EINVAL; - - fd = open(path, O_WRONLY); - if (fd == -1) - return -errno; - - numwritten = write(fd, buf, buflen - 1); - saved_errno = errno; - close(fd); - - if (numwritten < 0) - return -saved_errno; - - if (numwritten != (ssize_t)(buflen - 1)) - return -EIO; - - return 0; -} - -int read_num(const char *path, unsigned long *num) -{ - unsigned long val; - int ret; - char buf[21]; - char *end; - - if (!num) - return -EINVAL; - - ret = read_file(path, buf, sizeof(buf)); - if (ret) - return ret; - - /* Reject signs and leading whitespace that are accepted by strtoul() */ - if (buf[0] < '0' || buf[0] > '9') - return -EINVAL; - - errno = 0; - val = strtoul(buf, &end, 10); - if (errno) - return -errno; - - /* Only allow a newline after the number */ - if (*end == '\n') - end++; - - if (*end != '\0') - return -EINVAL; - - *num = val; - return 0; -} - -int write_num(const char *path, unsigned long num) -{ - char buf[21]; - - sprintf(buf, "%lu", num); - return write_file(path, buf, strlen(buf) + 1); -} - -int write_num_ignore_einval(const char *path, unsigned long num) -{ - int ret; - - ret = write_num(path, num); - return ret == -EINVAL ? 0 : ret; -} - static unsigned long shmall, shmmax; void __shm_limits_restore(void) diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index 62f6f5b4264924..fe0475f2bdf288 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -8,6 +8,7 @@ #include /* _SC_PAGESIZE */ #include "kselftest.h" #include +#include #define BIT_ULL(nr) (1ULL << (nr)) #define PM_SOFT_DIRTY BIT_ULL(55) @@ -166,12 +167,6 @@ int unpoison_memory(unsigned long pfn); #define PAGEMAP_PRESENT(ent) (((ent) & (1ull << 63)) != 0) #define PAGEMAP_PFN(ent) ((ent) & ((1ull << 55) - 1)) -int write_file(const char *path, const char *buf, size_t buflen); -int read_file(const char *path, char *buf, size_t buflen); -int read_num(const char *path, unsigned long *num); -int write_num(const char *path, unsigned long num); -int write_num_ignore_einval(const char *path, unsigned long num); - void shm_limits_prepare(unsigned long length); void __shm_limits_restore(void); From af666ae48e61a5a6ca1375f3d74b111568cb9a9d Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 16:52:31 +0530 Subject: [PATCH 0779/1012] tools/lib/mm: move hugepage_settings out of selftests Move hugepage_settings.[ch] from tools/testing/selftests/mm/ to tools/lib/mm/ so the THP and HugeTLB helpers can be shared more easily between selftests and other tools. Keep the helpers exposed to mm selftests through vm_util.h where possible, and use direct includes for files that do not include vm_util.h. Adjust the selftests/mm build to compile the moved implementation from its new location. Remove the remaining kselftest dependency by including file_utils.h directly and using EXIT_FAILURE in the signal handler. Link: https://lore.kernel.org/20260918112234.195857-4-sarthak.sharma@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Muhammad Usama Anjum Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- .../{testing/selftests => lib}/mm/hugepage_settings.c | 11 +++++++++-- .../{testing/selftests => lib}/mm/hugepage_settings.h | 0 tools/testing/selftests/mm/Makefile | 6 ++++-- tools/testing/selftests/mm/compaction_test.c | 2 +- tools/testing/selftests/mm/cow.c | 1 - tools/testing/selftests/mm/folio_split_race_test.c | 1 - tools/testing/selftests/mm/guard-regions.c | 1 - tools/testing/selftests/mm/gup_longterm.c | 1 - tools/testing/selftests/mm/gup_test.c | 1 - tools/testing/selftests/mm/hmm-tests.c | 6 +++--- tools/testing/selftests/mm/hugetlb-madvise.c | 1 - tools/testing/selftests/mm/hugetlb-mmap.c | 1 - tools/testing/selftests/mm/hugetlb-mremap.c | 1 - tools/testing/selftests/mm/hugetlb-shm.c | 1 - tools/testing/selftests/mm/hugetlb-soft-offline.c | 2 +- tools/testing/selftests/mm/hugetlb_dio.c | 1 - tools/testing/selftests/mm/hugetlb_fault_after_madv.c | 1 - tools/testing/selftests/mm/hugetlb_madv_vs_map.c | 1 - tools/testing/selftests/mm/khugepaged.c | 1 - tools/testing/selftests/mm/ksm_tests.c | 1 - tools/testing/selftests/mm/migration.c | 2 +- tools/testing/selftests/mm/pagemap_ioctl.c | 1 - tools/testing/selftests/mm/prctl_thp_disable.c | 1 - tools/testing/selftests/mm/protection_keys.c | 2 +- tools/testing/selftests/mm/soft-dirty.c | 1 - tools/testing/selftests/mm/split_huge_page_test.c | 1 - tools/testing/selftests/mm/thuge-gen.c | 1 - tools/testing/selftests/mm/transhuge-stress.c | 1 - tools/testing/selftests/mm/uffd-common.h | 1 - tools/testing/selftests/mm/uffd-wp-mremap.c | 2 +- tools/testing/selftests/mm/va_high_addr_switch.c | 1 - tools/testing/selftests/mm/vm_util.h | 1 + 32 files changed, 22 insertions(+), 34 deletions(-) rename tools/{testing/selftests => lib}/mm/hugepage_settings.c (99%) rename tools/{testing/selftests => lib}/mm/hugepage_settings.h (100%) diff --git a/tools/testing/selftests/mm/hugepage_settings.c b/tools/lib/mm/hugepage_settings.c similarity index 99% rename from tools/testing/selftests/mm/hugepage_settings.c rename to tools/lib/mm/hugepage_settings.c index 6f3abd35738592..656442c8d3954a 100644 --- a/tools/testing/selftests/mm/hugepage_settings.c +++ b/tools/lib/mm/hugepage_settings.c @@ -10,11 +10,16 @@ #include #include -#include "vm_util.h" +#include "file_utils.h" #include "hugepage_settings.h" #define THP_SYSFS "/sys/kernel/mm/transparent_hugepage/" #define MAX_SETTINGS_DEPTH 4 + +#ifndef ARRAY_SIZE +#define ARRAY_SIZE(arr) (sizeof(arr) / sizeof((arr)[0])) +#endif + static struct thp_settings settings_stack[MAX_SETTINGS_DEPTH]; static int settings_index; static struct thp_settings saved_settings; @@ -655,8 +660,10 @@ static void hugepage_restore_settings_atexit(void) static void hugepage_restore_settings_sighandler(int sig) { + (void)sig; + /* exit() will invoke the hugepage_restore_settings_atexit handler. */ - exit(KSFT_FAIL); + exit(EXIT_FAILURE); } void hugepage_save_settings(bool thp, bool hugetlb) diff --git a/tools/testing/selftests/mm/hugepage_settings.h b/tools/lib/mm/hugepage_settings.h similarity index 100% rename from tools/testing/selftests/mm/hugepage_settings.h rename to tools/lib/mm/hugepage_settings.h diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index c36a9a7089bb26..67882e52d4ff0a 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -185,8 +185,10 @@ TEST_FILES += write_hugetlb_memory.sh include ../lib.mk -$(TEST_GEN_PROGS): vm_util.c hugepage_settings.c $(top_srcdir)/tools/lib/mm/file_utils.c -$(TEST_GEN_FILES): vm_util.c hugepage_settings.c $(top_srcdir)/tools/lib/mm/file_utils.c +$(TEST_GEN_PROGS): vm_util.c $(top_srcdir)/tools/lib/mm/hugepage_settings.c \ + $(top_srcdir)/tools/lib/mm/file_utils.c +$(TEST_GEN_FILES): vm_util.c $(top_srcdir)/tools/lib/mm/hugepage_settings.c \ + $(top_srcdir)/tools/lib/mm/file_utils.c $(OUTPUT)/uffd-stress: uffd-common.c $(OUTPUT)/uffd-unit-tests: uffd-common.c diff --git a/tools/testing/selftests/mm/compaction_test.c b/tools/testing/selftests/mm/compaction_test.c index 30d4ace7155ae9..b3f5377119cb80 100644 --- a/tools/testing/selftests/mm/compaction_test.c +++ b/tools/testing/selftests/mm/compaction_test.c @@ -15,9 +15,9 @@ #include #include #include +#include #include "kselftest.h" -#include "hugepage_settings.h" #define MAP_SIZE_MB 100 #define MAP_SIZE (MAP_SIZE_MB * 1024 * 1024) diff --git a/tools/testing/selftests/mm/cow.c b/tools/testing/selftests/mm/cow.c index 8aa5249d9bef63..3264a828575bd4 100644 --- a/tools/testing/selftests/mm/cow.c +++ b/tools/testing/selftests/mm/cow.c @@ -29,7 +29,6 @@ #include "../../../../mm/gup_test.h" #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" static size_t pagesize; static int pagemap_fd; diff --git a/tools/testing/selftests/mm/folio_split_race_test.c b/tools/testing/selftests/mm/folio_split_race_test.c index 1960635a953eb5..e4660bf89b624a 100644 --- a/tools/testing/selftests/mm/folio_split_race_test.c +++ b/tools/testing/selftests/mm/folio_split_race_test.c @@ -25,7 +25,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" uint64_t page_size; uint64_t pmd_pagesize; diff --git a/tools/testing/selftests/mm/guard-regions.c b/tools/testing/selftests/mm/guard-regions.c index b724d62d2b7555..f7d53ea3c25270 100644 --- a/tools/testing/selftests/mm/guard-regions.c +++ b/tools/testing/selftests/mm/guard-regions.c @@ -21,7 +21,6 @@ #include #include #include "vm_util.h" -#include "hugepage_settings.h" #include "../pidfd/pidfd.h" diff --git a/tools/testing/selftests/mm/gup_longterm.c b/tools/testing/selftests/mm/gup_longterm.c index 510de93be6814f..c9d8b449126388 100644 --- a/tools/testing/selftests/mm/gup_longterm.c +++ b/tools/testing/selftests/mm/gup_longterm.c @@ -29,7 +29,6 @@ #include "../../../../mm/gup_test.h" #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" static size_t pagesize; static int nr_hugetlbsizes; diff --git a/tools/testing/selftests/mm/gup_test.c b/tools/testing/selftests/mm/gup_test.c index 3f841a96f87068..5f44761dbec0be 100644 --- a/tools/testing/selftests/mm/gup_test.c +++ b/tools/testing/selftests/mm/gup_test.c @@ -14,7 +14,6 @@ #include #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" #define MB (1UL << 20) diff --git a/tools/testing/selftests/mm/hmm-tests.c b/tools/testing/selftests/mm/hmm-tests.c index e2642eca0d02b4..fa1a651963fd01 100644 --- a/tools/testing/selftests/mm/hmm-tests.c +++ b/tools/testing/selftests/mm/hmm-tests.c @@ -10,9 +10,6 @@ * bugs. */ -#include "kselftest_harness.h" -#include "hugepage_settings.h" - #include #include #include @@ -33,6 +30,9 @@ #include #include #include +#include + +#include "kselftest_harness.h" /* * This is a private UAPI to the kernel test module so it isn't exported diff --git a/tools/testing/selftests/mm/hugetlb-madvise.c b/tools/testing/selftests/mm/hugetlb-madvise.c index 555b4b3d14307e..57cf790ca478d4 100644 --- a/tools/testing/selftests/mm/hugetlb-madvise.c +++ b/tools/testing/selftests/mm/hugetlb-madvise.c @@ -14,7 +14,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" #define MIN_FREE_PAGES 20 #define NR_HUGE_PAGES 10 /* common number of pages to map/allocate */ diff --git a/tools/testing/selftests/mm/hugetlb-mmap.c b/tools/testing/selftests/mm/hugetlb-mmap.c index 2edbc992e6bcf0..8853271fc8259d 100644 --- a/tools/testing/selftests/mm/hugetlb-mmap.c +++ b/tools/testing/selftests/mm/hugetlb-mmap.c @@ -18,7 +18,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" #define LENGTH (256UL*1024*1024) #define PROTECTION (PROT_READ | PROT_WRITE) diff --git a/tools/testing/selftests/mm/hugetlb-mremap.c b/tools/testing/selftests/mm/hugetlb-mremap.c index ed3d92e862d876..9b724af66e9388 100644 --- a/tools/testing/selftests/mm/hugetlb-mremap.c +++ b/tools/testing/selftests/mm/hugetlb-mremap.c @@ -26,7 +26,6 @@ #include #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" #define DEFAULT_LENGTH_MB 10UL #define MB_TO_BYTES(x) (x * 1024 * 1024) diff --git a/tools/testing/selftests/mm/hugetlb-shm.c b/tools/testing/selftests/mm/hugetlb-shm.c index 3ff7f062b7eb45..f4514da49e1df7 100644 --- a/tools/testing/selftests/mm/hugetlb-shm.c +++ b/tools/testing/selftests/mm/hugetlb-shm.c @@ -29,7 +29,6 @@ #include #include "vm_util.h" -#include "hugepage_settings.h" #define LENGTH (256UL*1024*1024) diff --git a/tools/testing/selftests/mm/hugetlb-soft-offline.c b/tools/testing/selftests/mm/hugetlb-soft-offline.c index ffc85b958c6920..d9565219378aae 100644 --- a/tools/testing/selftests/mm/hugetlb-soft-offline.c +++ b/tools/testing/selftests/mm/hugetlb-soft-offline.c @@ -22,10 +22,10 @@ #include #include #include +#include #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" #ifndef MADV_SOFT_OFFLINE #define MADV_SOFT_OFFLINE 101 diff --git a/tools/testing/selftests/mm/hugetlb_dio.c b/tools/testing/selftests/mm/hugetlb_dio.c index fb4600570e1319..9495974eccbea5 100644 --- a/tools/testing/selftests/mm/hugetlb_dio.c +++ b/tools/testing/selftests/mm/hugetlb_dio.c @@ -20,7 +20,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" #ifndef STATX_DIOALIGN #define STATX_DIOALIGN 0x00002000U diff --git a/tools/testing/selftests/mm/hugetlb_fault_after_madv.c b/tools/testing/selftests/mm/hugetlb_fault_after_madv.c index 2dc158054f666b..56c5a8533e9d90 100644 --- a/tools/testing/selftests/mm/hugetlb_fault_after_madv.c +++ b/tools/testing/selftests/mm/hugetlb_fault_after_madv.c @@ -10,7 +10,6 @@ #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" #define INLOOP_ITER 100 diff --git a/tools/testing/selftests/mm/hugetlb_madv_vs_map.c b/tools/testing/selftests/mm/hugetlb_madv_vs_map.c index 0f15eff1da0403..1d111f42dd5963 100644 --- a/tools/testing/selftests/mm/hugetlb_madv_vs_map.c +++ b/tools/testing/selftests/mm/hugetlb_madv_vs_map.c @@ -14,7 +14,6 @@ #include #include "vm_util.h" -#include "hugepage_settings.h" #define INLOOP_ITER 100 diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 6daa22f6da2f3f..525108cace54ab 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -22,7 +22,6 @@ #include "linux/magic.h" #include "vm_util.h" -#include "hugepage_settings.h" #define BASE_ADDR ((void *)(1UL << 30)) static unsigned long hpage_pmd_size; diff --git a/tools/testing/selftests/mm/ksm_tests.c b/tools/testing/selftests/mm/ksm_tests.c index 5fd7792a0d4794..6711f4c6137110 100644 --- a/tools/testing/selftests/mm/ksm_tests.c +++ b/tools/testing/selftests/mm/ksm_tests.c @@ -15,7 +15,6 @@ #include "kselftest.h" #include #include "vm_util.h" -#include "hugepage_settings.h" #define KSM_SYSFS_PATH "/sys/kernel/mm/ksm/" #define KSM_FP(s) (KSM_SYSFS_PATH s) diff --git a/tools/testing/selftests/mm/migration.c b/tools/testing/selftests/mm/migration.c index f19d53c6957644..a35e2b57e05b2d 100644 --- a/tools/testing/selftests/mm/migration.c +++ b/tools/testing/selftests/mm/migration.c @@ -5,7 +5,6 @@ */ #include "kselftest_harness.h" -#include "hugepage_settings.h" #include #include @@ -16,6 +15,7 @@ #include #include #include + #include "vm_util.h" #define TWOMEG (2<<20) diff --git a/tools/testing/selftests/mm/pagemap_ioctl.c b/tools/testing/selftests/mm/pagemap_ioctl.c index d9a4fb782ecfe7..03898b4f6cdab4 100644 --- a/tools/testing/selftests/mm/pagemap_ioctl.c +++ b/tools/testing/selftests/mm/pagemap_ioctl.c @@ -24,7 +24,6 @@ #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" #define PAGEMAP_BITS_ALL (PAGE_IS_WPALLOWED | PAGE_IS_WRITTEN | \ PAGE_IS_FILE | PAGE_IS_PRESENT | \ diff --git a/tools/testing/selftests/mm/prctl_thp_disable.c b/tools/testing/selftests/mm/prctl_thp_disable.c index 82c6e96ea6eb37..f9ec1408a6e307 100644 --- a/tools/testing/selftests/mm/prctl_thp_disable.c +++ b/tools/testing/selftests/mm/prctl_thp_disable.c @@ -14,7 +14,6 @@ #include #include "kselftest_harness.h" -#include "hugepage_settings.h" #include "vm_util.h" #ifndef PR_THP_DISABLE_EXCEPT_ADVISED diff --git a/tools/testing/selftests/mm/protection_keys.c b/tools/testing/selftests/mm/protection_keys.c index ae6e1530b35484..b7882ab97683f2 100644 --- a/tools/testing/selftests/mm/protection_keys.c +++ b/tools/testing/selftests/mm/protection_keys.c @@ -45,8 +45,8 @@ #include #include #include +#include -#include "hugepage_settings.h" #include "pkey-helpers.h" u64 shadow_pkey_reg; diff --git a/tools/testing/selftests/mm/soft-dirty.c b/tools/testing/selftests/mm/soft-dirty.c index 5f278913c4d754..7f649b67335539 100644 --- a/tools/testing/selftests/mm/soft-dirty.c +++ b/tools/testing/selftests/mm/soft-dirty.c @@ -9,7 +9,6 @@ #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" #define PAGEMAP_FILE_PATH "/proc/self/pagemap" #define TEST_ITERATIONS 10000 diff --git a/tools/testing/selftests/mm/split_huge_page_test.c b/tools/testing/selftests/mm/split_huge_page_test.c index a30927514b4f6c..68f508c9a355fc 100644 --- a/tools/testing/selftests/mm/split_huge_page_test.c +++ b/tools/testing/selftests/mm/split_huge_page_test.c @@ -21,7 +21,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" uint64_t pagesize; unsigned int pageshift; diff --git a/tools/testing/selftests/mm/thuge-gen.c b/tools/testing/selftests/mm/thuge-gen.c index 50d0805b65db94..a04f588df780f1 100644 --- a/tools/testing/selftests/mm/thuge-gen.c +++ b/tools/testing/selftests/mm/thuge-gen.c @@ -14,7 +14,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" #if !defined(MAP_HUGETLB) #define MAP_HUGETLB 0x40000 diff --git a/tools/testing/selftests/mm/transhuge-stress.c b/tools/testing/selftests/mm/transhuge-stress.c index 8eb0c5630e7e31..96f72898ebe0a5 100644 --- a/tools/testing/selftests/mm/transhuge-stress.c +++ b/tools/testing/selftests/mm/transhuge-stress.c @@ -17,7 +17,6 @@ #include #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" int backing_fd = -1; int mmap_flags = MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE; diff --git a/tools/testing/selftests/mm/uffd-common.h b/tools/testing/selftests/mm/uffd-common.h index 92a21b97f745af..0723843a7626b1 100644 --- a/tools/testing/selftests/mm/uffd-common.h +++ b/tools/testing/selftests/mm/uffd-common.h @@ -37,7 +37,6 @@ #include "kselftest.h" #include "vm_util.h" -#include "hugepage_settings.h" #define UFFD_FLAGS (O_CLOEXEC | O_NONBLOCK | UFFD_USER_MODE_ONLY) diff --git a/tools/testing/selftests/mm/uffd-wp-mremap.c b/tools/testing/selftests/mm/uffd-wp-mremap.c index 572c2516e874d7..c48eaab8e75cf9 100644 --- a/tools/testing/selftests/mm/uffd-wp-mremap.c +++ b/tools/testing/selftests/mm/uffd-wp-mremap.c @@ -7,8 +7,8 @@ #include #include #include +#include #include "kselftest.h" -#include "hugepage_settings.h" #include "uffd-common.h" static int pagemap_fd; diff --git a/tools/testing/selftests/mm/va_high_addr_switch.c b/tools/testing/selftests/mm/va_high_addr_switch.c index e24d7ba00b4417..5a354a664d1f7d 100644 --- a/tools/testing/selftests/mm/va_high_addr_switch.c +++ b/tools/testing/selftests/mm/va_high_addr_switch.c @@ -11,7 +11,6 @@ #include "vm_util.h" #include "kselftest.h" -#include "hugepage_settings.h" /* * The hint addr value is used to allocate addresses diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index fe0475f2bdf288..64a86e8a0c41bf 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -9,6 +9,7 @@ #include "kselftest.h" #include #include +#include #define BIT_ULL(nr) (1ULL << (nr)) #define PM_SOFT_DIRTY BIT_ULL(55) From c784d6dde8a2021f35b2dc928311350cf7e45292 Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 16:52:32 +0530 Subject: [PATCH 0780/1012] tools/mm: move gup_test from selftests/mm to tools/mm Move tools/testing/selftests/mm/gup_test.c to tools/mm/gup_bench.c. This is the first step in separating its benchmarking and functional testing components. Later patches will make this a purely benchmarking tool and introduce a new functional selftest under selftests/mm. Include hugepage_settings.h directly instead of vm_util.h and use getpagesize() instead of psize(). Adjust the Makefiles in both locations and add gup_bench to tools/mm/.gitignore. Remove the gup_test invocations from run_vmtests.sh and update MAINTAINERS. Also remove the gup_test reference from Documentation/core-api/pin_user_pages.rst. The selftest added later in the series is standalone and does not need per command documentation here. Link: https://lore.kernel.org/20260918112234.195857-5-sarthak.sharma@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Muhammad Usama Anjum Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- Documentation/core-api/pin_user_pages.rst | 9 ----- MAINTAINERS | 2 +- tools/mm/.gitignore | 1 + tools/mm/Makefile | 11 ++++-- .../mm/gup_test.c => mm/gup_bench.c} | 8 ++--- tools/testing/selftests/mm/Makefile | 1 - tools/testing/selftests/mm/run_vmtests.sh | 36 ------------------- 7 files changed, 14 insertions(+), 54 deletions(-) rename tools/{testing/selftests/mm/gup_test.c => mm/gup_bench.c} (97%) diff --git a/Documentation/core-api/pin_user_pages.rst b/Documentation/core-api/pin_user_pages.rst index c16ca163b55e3c..e0acedbd1d4860 100644 --- a/Documentation/core-api/pin_user_pages.rst +++ b/Documentation/core-api/pin_user_pages.rst @@ -226,15 +226,6 @@ will be pinned longterm, and whose data will be accessed. Unit testing ============ -This file:: - - tools/testing/selftests/mm/gup_test.c - -has the following new calls to exercise the new pin*() wrapper functions: - -* PIN_FAST_BENCHMARK (./gup_test -a) -* PIN_BASIC_TEST (./gup_test -b) - You can monitor how many total dma-pinned pages have been acquired and released since the system was booted, via two new /proc/vmstat entries: :: diff --git a/MAINTAINERS b/MAINTAINERS index dfd9f948390708..3a16011c3f7f75 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17190,8 +17190,8 @@ T: git git://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm F: mm/gup.c F: mm/gup_test.c F: mm/gup_test.h +F: tools/mm/gup_bench.c F: tools/testing/selftests/mm/gup_longterm.c -F: tools/testing/selftests/mm/gup_test.c MEMORY MANAGEMENT - KSM (Kernel Samepage Merging) M: Andrew Morton diff --git a/tools/mm/.gitignore b/tools/mm/.gitignore index 1446a659e54088..154d740be02e82 100644 --- a/tools/mm/.gitignore +++ b/tools/mm/.gitignore @@ -3,3 +3,4 @@ slabinfo page-types page_owner_sort thp_swap_allocator_test +gup_bench diff --git a/tools/mm/Makefile b/tools/mm/Makefile index 858186a6eefdbd..f20a32d8cc22e2 100644 --- a/tools/mm/Makefile +++ b/tools/mm/Makefile @@ -3,13 +3,15 @@ # include ../scripts/Makefile.include -BUILD_TARGETS=page-types slabinfo page_owner_sort page_owner_filter thp_swap_allocator_test +BUILD_TARGETS=page-types slabinfo page_owner_sort page_owner_filter +BUILD_TARGETS += thp_swap_allocator_test gup_bench INSTALL_TARGETS = $(BUILD_TARGETS) thpmaps LIB_DIR = ../lib/api LIBS = $(LIB_DIR)/libapi.a +GUP_BENCH_OBJS = gup_bench.c ../lib/mm/hugepage_settings.c ../lib/mm/file_utils.c -CFLAGS += -Wall -Wextra -I../lib/ -pthread +CFLAGS += -Wall -Wextra -I../lib/ -I../.. -pthread LDFLAGS += $(LIBS) -pthread all: $(BUILD_TARGETS) @@ -22,8 +24,11 @@ $(LIBS): %: %.c $(CC) $(CFLAGS) -o $@ $< $(LDFLAGS) +gup_bench: $(GUP_BENCH_OBJS) $(LIBS) + $(CC) $(CFLAGS) -o $@ $(GUP_BENCH_OBJS) $(LDFLAGS) + clean: - $(RM) page-types slabinfo page_owner_sort page_owner_filter thp_swap_allocator_test + $(RM) page-types slabinfo page_owner_sort page_owner_filter thp_swap_allocator_test gup_bench make -C $(LIB_DIR) clean sbindir ?= /usr/sbin diff --git a/tools/testing/selftests/mm/gup_test.c b/tools/mm/gup_bench.c similarity index 97% rename from tools/testing/selftests/mm/gup_test.c rename to tools/mm/gup_bench.c index 5f44761dbec0be..da56aa5324d3fa 100644 --- a/tools/testing/selftests/mm/gup_test.c +++ b/tools/mm/gup_bench.c @@ -12,8 +12,8 @@ #include #include #include -#include "kselftest.h" -#include "vm_util.h" +#include +#include "../testing/selftests/kselftest.h" #define MB (1UL << 20) @@ -140,7 +140,7 @@ int main(int argc, char **argv) case 'n': nr_pages = atoi(optarg); if (nr_pages < 0) - nr_pages = size / psize(); + nr_pages = size / getpagesize(); break; case 't': thp = 1; @@ -254,7 +254,7 @@ int main(int argc, char **argv) madvise(p, size, MADV_NOHUGEPAGE); /* Fault them in here, from user space. */ - for (; (unsigned long)p < gup.addr + size; p += psize()) + for (; (unsigned long)p < gup.addr + size; p += getpagesize()) p[0] = 0; tid = malloc(sizeof(pthread_t) * nthreads); diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index 67882e52d4ff0a..d6337156111a4d 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -59,7 +59,6 @@ endif TEST_GEN_FILES = cow TEST_GEN_FILES += compaction_test TEST_GEN_FILES += gup_longterm -TEST_GEN_FILES += gup_test TEST_GEN_FILES += hmm-tests TEST_GEN_FILES += hugetlb-madvise TEST_GEN_FILES += hugetlb-mmap diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index a1db516557023f..4281bb6a859e03 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -148,30 +148,6 @@ test_selected() { fi } -run_gup_matrix() { - # -t: thp=on, -T: thp=off, -H: hugetlb=on - local hugetlb_mb=256 - - for huge in -t -T "-H -m $hugetlb_mb"; do - # -u: gup-fast, -U: gup-basic, -a: pin-fast, -b: pin-basic, -L: pin-longterm - for test_cmd in -u -U -a -b -L; do - # -w: write=1, -W: write=0 - for write in -w -W; do - # -S: shared - for share in -S " "; do - # -n: How many pages to fetch together? 512 is special - # because it's default thp size (or 2M on x86), 123 to - # just test partial gup when hit a huge in whatever form - for num in "-n 1" "-n 512" "-n 123" "-n -1"; do - CATEGORY="gup_test" run_test ./gup_test \ - $huge $test_cmd $write $share $num - done - done - done - done - done -} - # filter 64bit architectures ARCH64STR="arm64 mips64 parisc64 ppc64 ppc64le riscv64 s390x sparc64 x86_64" if [ -z "$ARCH" ]; then @@ -293,18 +269,6 @@ fi CATEGORY="mmap" run_test ./map_fixed_noreplace -if $RUN_ALL; then - run_gup_matrix -else - # get_user_pages_fast() benchmark - CATEGORY="gup_test" run_test ./gup_test -u -n 1 - CATEGORY="gup_test" run_test ./gup_test -u -n -1 - # pin_user_pages_fast() benchmark - CATEGORY="gup_test" run_test ./gup_test -a -n 1 - CATEGORY="gup_test" run_test ./gup_test -a -n -1 -fi -# Dump pages 0, 19, and 4096, using pin_user_pages: -CATEGORY="gup_test" run_test ./gup_test -ct -F 0x1 0 19 0x1000 CATEGORY="gup_test" run_test ./gup_longterm CATEGORY="userfaultfd" run_test ./uffd-unit-tests From 9d5bc79fe349952cec3249d254d22b2ba2541b36 Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 16:52:33 +0530 Subject: [PATCH 0781/1012] tools/mm: make gup_bench a benchmark only tool Remove the functional modes (GUP_BASIC_TEST, PIN_BASIC_TEST and DUMP_USER_PAGES_TEST) from gup_bench. Drop the kselftest dependency and use normal diagnostics and exit statuses. When no arguments are supplied, run a single GUP_FAST_BENCHMARK with the existing default values. Let users select other configurations through command-line options. Report ioctl failures and handle errors without relying on assert(). Link: https://lore.kernel.org/20260918112234.195857-6-sarthak.sharma@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Muhammad Usama Anjum Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- tools/mm/gup_bench.c | 183 ++++++++++++++++++------------------------- 1 file changed, 78 insertions(+), 105 deletions(-) diff --git a/tools/mm/gup_bench.c b/tools/mm/gup_bench.c index da56aa5324d3fa..ff6e466546813a 100644 --- a/tools/mm/gup_bench.c +++ b/tools/mm/gup_bench.c @@ -10,10 +10,10 @@ #include #include #include -#include +#include +#include #include #include -#include "../testing/selftests/kselftest.h" #define MB (1UL << 20) @@ -37,12 +37,6 @@ static char *cmd_to_str(unsigned long cmd) return "PIN_FAST_BENCHMARK"; case PIN_LONGTERM_BENCHMARK: return "PIN_LONGTERM_BENCHMARK"; - case GUP_BASIC_TEST: - return "GUP_BASIC_TEST"; - case PIN_BASIC_TEST: - return "PIN_BASIC_TEST"; - case DUMP_USER_PAGES_TEST: - return "DUMP_USER_PAGES_TEST"; } return "Unknown command"; } @@ -52,39 +46,29 @@ void *gup_thread(void *data) struct gup_test gup = *(struct gup_test *)data; int i, status; - /* Only report timing information on the *_BENCHMARK commands: */ - if ((cmd == PIN_FAST_BENCHMARK) || (cmd == GUP_FAST_BENCHMARK) || - (cmd == PIN_LONGTERM_BENCHMARK)) { - for (i = 0; i < repeats; i++) { - gup.size = size; - status = ioctl(gup_fd, cmd, &gup); - if (status) - break; + for (i = 0; i < repeats; i++) { + gup.size = size; + status = ioctl(gup_fd, cmd, &gup); + if (status) { + int err = errno; pthread_mutex_lock(&print_mutex); - ksft_print_msg("%s: Time: get:%lld put:%lld us", - cmd_to_str(cmd), gup.get_delta_usec, - gup.put_delta_usec); - if (gup.size != size) - ksft_print_msg(", truncated (size: %lld)", gup.size); - ksft_print_msg("\n"); + fprintf(stderr, "%s ioctl failed: %s\n", cmd_to_str(cmd), + strerror(err)); pthread_mutex_unlock(&print_mutex); + return data; } - } else { - gup.size = size; - status = ioctl(gup_fd, cmd, &gup); - if (status) - goto return_; pthread_mutex_lock(&print_mutex); - ksft_print_msg("%s: done\n", cmd_to_str(cmd)); + printf("%s: Time: get:%lld put:%lld us", + cmd_to_str(cmd), gup.get_delta_usec, + gup.put_delta_usec); if (gup.size != size) - ksft_print_msg("Truncated (size: %lld)\n", gup.size); + printf(", truncated (size: %lld)", gup.size); + printf("\n"); pthread_mutex_unlock(&print_mutex); } -return_: - ksft_test_result(!status, "ioctl status %d\n", status); return NULL; } @@ -92,38 +76,21 @@ int main(int argc, char **argv) { struct gup_test gup = { 0 }; int filed, i, opt, nr_pages = 1, thp = -1, write = 1, nthreads = 1, ret; - int flags = MAP_PRIVATE; + int flags = MAP_PRIVATE, started_threads = 0, exit_status = 1; char *file = "/dev/zero"; - bool hugetlb = false; + bool hugetlb = false, thread_error = false; + void *thread_result; pthread_t *tid; char *p; - while ((opt = getopt(argc, argv, "m:r:n:F:f:abcj:tTLUuwWSHpz")) != -1) { + while ((opt = getopt(argc, argv, "m:r:n:F:f:aj:tTLuwWSH")) != -1) { switch (opt) { case 'a': cmd = PIN_FAST_BENCHMARK; break; - case 'b': - cmd = PIN_BASIC_TEST; - break; case 'L': cmd = PIN_LONGTERM_BENCHMARK; break; - case 'c': - cmd = DUMP_USER_PAGES_TEST; - /* - * Dump page 0 (index 1). May be overridden later, by - * user's non-option arguments. - * - * .which_pages is zero-based, so that zero can mean "do - * nothing". - */ - gup.which_pages[0] = 1; - break; - case 'p': - /* works only with DUMP_USER_PAGES_TEST */ - gup.test_flags |= GUP_TEST_FLAG_DUMP_PAGES_USE_PIN; - break; case 'F': /* strtol, so you can pass flags in hex form */ gup.gup_flags = strtol(optarg, 0, 0); @@ -148,9 +115,6 @@ int main(int argc, char **argv) case 'T': thp = 0; break; - case 'U': - cmd = GUP_BASIC_TEST; - break; case 'u': cmd = GUP_FAST_BENCHMARK; break; @@ -172,52 +136,41 @@ int main(int argc, char **argv) hugetlb = true; break; default: - ksft_exit_fail_msg("Wrong argument\n"); + fprintf(stderr, "Wrong argument\n"); + exit(1); } } - if (optind < argc) { - int extra_arg_count = 0; - /* - * For example: - * - * ./gup_test -c 0 1 0x1001 - * - * ...to dump pages 0, 1, and 4097 - */ - - while ((optind < argc) && - (extra_arg_count < GUP_TEST_MAX_PAGES_TO_DUMP)) { - /* - * Do the 1-based indexing here, so that the user can - * use normal 0-based indexing on the command line. - */ - long page_index = strtol(argv[optind], 0, 0) + 1; - - gup.which_pages[extra_arg_count] = page_index; - extra_arg_count++; - optind++; - } + if (optind != argc) { + fprintf(stderr, "Unexpected argument '%s'\n", argv[optind]); + exit(1); } - ksft_print_header(); + if (geteuid()) { + fprintf(stderr, "Please run this test as root\n"); + exit(1); + } if (hugetlb) { unsigned long hp_size = default_huge_page_size(); - if (!hp_size) - ksft_exit_skip("HugeTLB is unavailable\n"); + if (!hp_size) { + fprintf(stderr, "Could not determine huge page size\n"); + return 1; + } size = (size + hp_size - 1) & ~(hp_size - 1); - if (!hugetlb_setup_default(size / hp_size)) - ksft_exit_skip("Not enough huge pages\n"); + if (!hugetlb_setup_default(size / hp_size)) { + fprintf(stderr, "Not enough huge pages\n"); + return 1; + } } - ksft_set_plan(nthreads); - filed = open(file, O_RDWR|O_CREAT, 0664); - if (filed < 0) - ksft_exit_fail_msg("Unable to open %s: %s\n", file, strerror(errno)); + if (filed < 0) { + fprintf(stderr, "Unable to open %s: %s\n", file, strerror(errno)); + return 1; + } gup.nr_pages_per_call = nr_pages; if (write) @@ -226,26 +179,24 @@ int main(int argc, char **argv) gup_fd = open(GUP_TEST_FILE, O_RDWR); if (gup_fd == -1) { switch (errno) { - case EACCES: - if (getuid()) - ksft_print_msg("Please run this test as root\n"); - break; case ENOENT: if (opendir("/sys/kernel/debug") == NULL) - ksft_print_msg("mount debugfs at /sys/kernel/debug\n"); - ksft_print_msg("check if CONFIG_GUP_TEST is enabled in kernel config\n"); + fprintf(stderr, "mount debugfs at /sys/kernel/debug\n"); + fprintf(stderr, "check if CONFIG_GUP_TEST is enabled in kernel config\n"); break; default: - ksft_print_msg("failed to open %s: %s\n", GUP_TEST_FILE, strerror(errno)); + fprintf(stderr, "failed to open %s: %s\n", GUP_TEST_FILE, + strerror(errno)); break; } - ksft_test_result_skip("Please run this test as root\n"); - ksft_exit_pass(); + goto err_close_filed; } p = mmap(NULL, size, PROT_READ | PROT_WRITE, flags, filed, 0); - if (p == MAP_FAILED) - ksft_exit_fail_msg("mmap: %s\n", strerror(errno)); + if (p == MAP_FAILED) { + fprintf(stderr, "mmap: %s\n", strerror(errno)); + goto err_close_gup_fd; + } gup.addr = (unsigned long)p; if (thp == 1) @@ -258,17 +209,39 @@ int main(int argc, char **argv) p[0] = 0; tid = malloc(sizeof(pthread_t) * nthreads); - assert(tid); + if (!tid) { + fprintf(stderr, "Failed to allocate %d threads: %s\n", + nthreads, strerror(errno)); + goto err_unmap; + } + for (i = 0; i < nthreads; i++) { ret = pthread_create(&tid[i], NULL, gup_thread, &gup); - assert(ret == 0); + if (ret) { + fprintf(stderr, "pthread_create failed: %s\n", strerror(ret)); + thread_error = true; + break; + } + started_threads++; } - for (i = 0; i < nthreads; i++) { - ret = pthread_join(tid[i], NULL); - assert(ret == 0); + for (i = 0; i < started_threads; i++) { + ret = pthread_join(tid[i], &thread_result); + if (ret) { + fprintf(stderr, "pthread_join failed: %s\n", strerror(ret)); + thread_error = true; + } else if (thread_result) + thread_error = true; } free(tid); - - ksft_exit_pass(); + if (!thread_error) + exit_status = 0; + +err_unmap: + munmap((void *)gup.addr, size); +err_close_gup_fd: + close(gup_fd); +err_close_filed: + close(filed); + return exit_status; } From 9c5f15d6b14a3e5c5a1d596e9e6d0c200cc61abc Mon Sep 17 00:00:00 2001 From: Sarthak Sharma Date: Fri, 18 Sep 2026 16:52:34 +0530 Subject: [PATCH 0782/1012] selftests/mm: add a GUP selftest Add a new GUP selftest which uses kselftest_harness.h. Cover 12 mapping configurations: THP enabled, THP disabled and HugeTLB, each across private/shared mappings and with/without FOLL_WRITE. Run 5 test cases for every variant: get_user_pages, get_user_pages_fast, pin_user_pages, pin_user_pages_fast and pin_user_pages_longterm. Use two default hugeTLB pages and derive the mapping size from their size. This exercises GUP both within a single HugeTLB page and across a HugeTLB boundary, without reserving an excessive number of pages. Sweep four nr_pages_per_call values for each test: 1, 512, 123 and all pages. This preserves the coverage previously provided by run_gup_matrix(): 12 mapping combinations x 5 GUP/PUP operations x 4 batch sizes. In total the selftest reports 60 TAP cases and issues 240 ioctls. Do not carry DUMP_USER_PAGES_TEST into the new selftest because its output is written to the kernel log and the selftest does not verify that output. Add the new gup binary to the selftests/mm build, run_vmtests.sh and MAINTAINERS. Update mm/Kconfig to describe the benchmark and selftest split. Link: https://lore.kernel.org/20260918112234.195857-7-sarthak.sharma@arm.com Signed-off-by: Sarthak Sharma Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: Mike Rapoport (Microsoft) Acked-by: David Hildenbrand (Arm) Tested-by: Muhammad Usama Anjum Cc: Anshuman Khandual Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Jason Gunthorpe Cc: John Hubbard Cc: Jonathan Corbet Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Mark Brown Cc: Michal Hocko Cc: Nico Pache Cc: Peter Xu Cc: Ryan Roberts Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Zi Yan --- MAINTAINERS | 1 + mm/Kconfig | 19 +- tools/testing/selftests/mm/Makefile | 1 + tools/testing/selftests/mm/gup.c | 262 ++++++++++++++++++++++ tools/testing/selftests/mm/run_vmtests.sh | 1 + 5 files changed, 272 insertions(+), 12 deletions(-) create mode 100644 tools/testing/selftests/mm/gup.c diff --git a/MAINTAINERS b/MAINTAINERS index 3a16011c3f7f75..596efa354ea235 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -17191,6 +17191,7 @@ F: mm/gup.c F: mm/gup_test.c F: mm/gup_test.h F: tools/mm/gup_bench.c +F: tools/testing/selftests/mm/gup.c F: tools/testing/selftests/mm/gup_longterm.c MEMORY MANAGEMENT - KSM (Kernel Samepage Merging) diff --git a/mm/Kconfig b/mm/Kconfig index 30170a936f1fc0..edb4a6c0a87021 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -1292,24 +1292,19 @@ config PERCPU_STATS be used to help understand percpu memory usage. config GUP_TEST - bool "Enable infrastructure for get_user_pages()-related unit tests" + bool "Enable infrastructure for get_user_pages()-related unit tests and benchmarks" depends on DEBUG_FS help Provides /sys/kernel/debug/gup_test, which in turn provides a way - to make ioctl calls that can launch kernel-based unit tests for - the get_user_pages*() and pin_user_pages*() family of API calls. + to make ioctl calls that can launch kernel-based unit tests and + benchmarks for the get_user_pages*() and pin_user_pages*() families + of API calls. - These tests include benchmark testing of the _fast variants of - get_user_pages*() and pin_user_pages*(), as well as smoke tests of + These include benchmark testing of the _fast variants of + get_user_pages*() and pin_user_pages*(), as well as tests of the non-_fast variants. - There is also a sub-test that allows running dump_page() on any - of up to eight pages (selected by command line args) within the - range of user-space addresses. These pages are either pinned via - pin_user_pages*(), or pinned via get_user_pages*(), as specified - by other command line arguments. - - See tools/testing/selftests/mm/gup_test.c + See tools/testing/selftests/mm/gup.c and tools/mm/gup_bench.c. comment "GUP_TEST needs to have DEBUG_FS enabled" depends on !GUP_TEST && !DEBUG_FS diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index d6337156111a4d..7d69baeb93f430 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -58,6 +58,7 @@ endif TEST_GEN_FILES = cow TEST_GEN_FILES += compaction_test +TEST_GEN_FILES += gup TEST_GEN_FILES += gup_longterm TEST_GEN_FILES += hmm-tests TEST_GEN_FILES += hugetlb-madvise diff --git a/tools/testing/selftests/mm/gup.c b/tools/testing/selftests/mm/gup.c new file mode 100644 index 00000000000000..a6a8ca47d12eb3 --- /dev/null +++ b/tools/testing/selftests/mm/gup.c @@ -0,0 +1,262 @@ +// SPDX-License-Identifier: GPL-2.0 +#define __SANE_USERSPACE_TYPES__ // Use ll64 +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "vm_util.h" +#include "kselftest_harness.h" + +#define MB (1UL << 20) + +/* Just the flags we need, copied from the kernel internals. */ +#define FOLL_WRITE 0x01 /* check pte is writable */ + +/* Page counts exercising single, THP-batch, partial, and full-mapping GUP. */ +static const int nr_pages_list[] = { 1, 512, 123, -1 }; + +#define GUP_TEST_FILE "/sys/kernel/debug/gup_test" +#define NR_HUGETLB_PAGES 2 + +static unsigned long hp_size; + +FIXTURE(gup_test) +{ + int gup_fd; + char *addr; + unsigned long size; +}; + +FIXTURE_VARIANT(gup_test) +{ + bool thp; + bool hugetlb; + bool write; + bool shared; +}; + +FIXTURE_VARIANT_ADD(gup_test, private_write) +{ + .thp = false, + .hugetlb = false, + .write = true, + .shared = false, +}; + +FIXTURE_VARIANT_ADD(gup_test, private_read) +{ + .thp = false, + .hugetlb = false, + .write = false, + .shared = false, +}; + +FIXTURE_VARIANT_ADD(gup_test, private_write_thp) +{ + .thp = true, + .hugetlb = false, + .write = true, + .shared = false, +}; + +FIXTURE_VARIANT_ADD(gup_test, private_read_thp) +{ + .thp = true, + .hugetlb = false, + .write = false, + .shared = false, +}; + +FIXTURE_VARIANT_ADD(gup_test, private_write_hugetlb) +{ + .thp = false, + .hugetlb = true, + .write = true, + .shared = false, +}; + +FIXTURE_VARIANT_ADD(gup_test, private_read_hugetlb) +{ + .thp = false, + .hugetlb = true, + .write = false, + .shared = false, +}; + +FIXTURE_VARIANT_ADD(gup_test, shared_write) +{ + .thp = false, + .hugetlb = false, + .write = true, + .shared = true, +}; + +FIXTURE_VARIANT_ADD(gup_test, shared_read) +{ + .thp = false, + .hugetlb = false, + .write = false, + .shared = true, +}; + +FIXTURE_VARIANT_ADD(gup_test, shared_write_thp) +{ + .thp = true, + .hugetlb = false, + .write = true, + .shared = true, +}; + +FIXTURE_VARIANT_ADD(gup_test, shared_read_thp) +{ + .thp = true, + .hugetlb = false, + .write = false, + .shared = true, +}; + +FIXTURE_VARIANT_ADD(gup_test, shared_write_hugetlb) +{ + .thp = false, + .hugetlb = true, + .write = true, + .shared = true, +}; + +FIXTURE_VARIANT_ADD(gup_test, shared_read_hugetlb) +{ + .thp = false, + .hugetlb = true, + .write = false, + .shared = true, +}; + +FIXTURE_SETUP(gup_test) +{ + int mmap_flags = MAP_PRIVATE | MAP_ANONYMOUS; + char *p; + + self->size = 128 * MB; + + if (variant->hugetlb) { + if (!hp_size) + SKIP(return, "HugeTLB not available\n"); + + if (hugetlb_free_default_pages() < NR_HUGETLB_PAGES) + SKIP(return, "Not enough huge pages\n"); + + self->size = NR_HUGETLB_PAGES * hp_size; + mmap_flags |= MAP_HUGETLB; + } + + if (variant->shared) + mmap_flags = (mmap_flags & ~MAP_PRIVATE) | MAP_SHARED; + + /* gup_fd has to be >= 0. Already checked in main() */ + self->gup_fd = open(GUP_TEST_FILE, O_RDWR); + ASSERT_GE(self->gup_fd, 0); + + self->addr = mmap(NULL, self->size, PROT_READ | PROT_WRITE, + mmap_flags, -1, 0); + + ASSERT_NE(self->addr, MAP_FAILED) { + int err = errno; + + close(self->gup_fd); + TH_LOG("mmap failed: %s", strerror(err)); + } + + if (variant->thp) + madvise(self->addr, self->size, MADV_HUGEPAGE); + else if (!variant->hugetlb) + madvise(self->addr, self->size, MADV_NOHUGEPAGE); + + for (p = self->addr; (unsigned long)p < (unsigned long)self->addr + + self->size; p += psize()) + p[0] = 0; +} + +FIXTURE_TEARDOWN(gup_test) +{ + munmap(self->addr, self->size); + close(self->gup_fd); +} + +static void run_gup_cmd(struct __test_metadata *_metadata, + FIXTURE_DATA(gup_test) *self, + const FIXTURE_VARIANT(gup_test) *variant, + unsigned long command) +{ + int i; + + for (i = 0; i < (int)ARRAY_SIZE(nr_pages_list); i++) { + struct gup_test gup = { + .addr = (unsigned long)self->addr, + .size = self->size, + .nr_pages_per_call = nr_pages_list[i] < 0 ? + self->size / psize() : nr_pages_list[i], + .gup_flags = variant->write ? FOLL_WRITE : 0, + }; + + TH_LOG("nr_pages_per_call=%u", gup.nr_pages_per_call); + ASSERT_EQ(ioctl(self->gup_fd, command, &gup), 0); + ASSERT_EQ(gup.size, self->size); + } +} + +TEST_F(gup_test, get_user_pages) +{ + run_gup_cmd(_metadata, self, variant, GUP_BASIC_TEST); +} + +TEST_F(gup_test, pin_user_pages) +{ + run_gup_cmd(_metadata, self, variant, PIN_BASIC_TEST); +} + +TEST_F(gup_test, get_user_pages_fast) +{ + run_gup_cmd(_metadata, self, variant, GUP_FAST_BENCHMARK); +} + +TEST_F(gup_test, pin_user_pages_fast) +{ + run_gup_cmd(_metadata, self, variant, PIN_FAST_BENCHMARK); +} + +TEST_F(gup_test, pin_user_pages_longterm) +{ + run_gup_cmd(_metadata, self, variant, PIN_LONGTERM_BENCHMARK); +} + +int main(int argc, char **argv) +{ + const int fd = open(GUP_TEST_FILE, O_RDWR); + + if (fd == -1) { + ksft_print_header(); + if (errno == EACCES) + ksft_exit_skip("Please run this test as root\n"); + if (errno == ENOENT) { + DIR *debugfs = opendir("/sys/kernel/debug"); + + if (!debugfs) + ksft_exit_skip("Mount debugfs at /sys/kernel/debug\n"); + closedir(debugfs); + ksft_exit_skip("Check CONFIG_GUP_TEST in kernel config\n"); + } + ksft_exit_fail_msg("Failed to open %s: %s\n", GUP_TEST_FILE, strerror(errno)); + } + close(fd); + + hp_size = default_huge_page_size(); + if (hp_size) + hugetlb_setup_default(NR_HUGETLB_PAGES); + + return test_harness_run(argc, argv); +} diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 4281bb6a859e03..0d8c93087d0529 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -269,6 +269,7 @@ fi CATEGORY="mmap" run_test ./map_fixed_noreplace +CATEGORY="gup_test" run_test ./gup CATEGORY="gup_test" run_test ./gup_longterm CATEGORY="userfaultfd" run_test ./uffd-unit-tests From 0d7cbb9c5764a7777288d11e65764adc2c35c103 Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Fri, 18 Sep 2026 04:24:37 -0700 Subject: [PATCH 0783/1012] mm/shmem: report RCU-tasks quiescent states while undoing a range This shows up in the Meta fleet on ftruncate() of large tmpfs files: INFO: rcu_tasks detected stalls on tasks: 00000000752fd185: .. nvcsw: 59455/59455 holdout: 1 idle_cpu: -1/0 task:rocksdb:bottom state:R running task __folio_split find_get_entry find_get_entries truncate_inode_partial_folio shmem_undo_range shmem_setattr notify_change do_ftruncate __x64_sys_ftruncate do_syscall_64 shmem_undo_range() walks the whole of the requested range in folio_batch sized steps, twice, and its two cond_resched() calls are the only reschedule points in that walk. cond_resched() is not an RCU-tasks quiescent state. Use cond_resched_tasks_rcu_qs() at both points so the walk reports an RCU-tasks quiescent state as it proceeds. Link: https://lore.kernel.org/20260918-shmem-tasks-rcu-v1-1-79acf91a2569@debian.org Fixes: 8315f42295d2 ("rcu: Add call_rcu_tasks()") Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Baolin Wang Cc: Hugh Dickins Cc: "Paul E . McKenney" Cc: --- mm/shmem.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/shmem.c b/mm/shmem.c index f33dbf5af2cb79..f2a36a1b537506 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -1367,7 +1367,7 @@ static void shmem_undo_range(struct inode *inode, loff_t lstart, uoff_t lend, } folio_batch_remove_exceptionals(&fbatch); folio_batch_release(&fbatch); - cond_resched(); + cond_resched_tasks_rcu_qs(); } /* @@ -1408,7 +1408,7 @@ static void shmem_undo_range(struct inode *inode, loff_t lstart, uoff_t lend, index = start; while (index < end) { - cond_resched(); + cond_resched_tasks_rcu_qs(); if (!find_get_entries(mapping, &index, end - 1, &fbatch, indices)) { From 9a036049b2121bf7203b5f2aa9523c1a83080554 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Fri, 18 Sep 2026 12:19:28 +0530 Subject: [PATCH 0784/1012] mm: constify arguments in default pxdp_get() Generic MM default pxdp_get() helpers fetch the values contained in pgtable entries via READ_ONCE() without modifying them. Just make their arguments explicitly 'const' for some additional protection. Link: https://lore.kernel.org/20260918064928.793742-1-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand Acked-by: David Hildenbrand (Arm) Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: SJ Park Cc: Pedro Falcato --- include/linux/pgtable.h | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/include/linux/pgtable.h b/include/linux/pgtable.h index e3c8ab96941c5e..780fe849ff8b7e 100644 --- a/include/linux/pgtable.h +++ b/include/linux/pgtable.h @@ -490,35 +490,35 @@ static inline int pudp_set_access_flags(struct vm_area_struct *vma, #endif #ifndef ptep_get -static inline pte_t ptep_get(pte_t *ptep) +static inline pte_t ptep_get(const pte_t *ptep) { return READ_ONCE(*ptep); } #endif #ifndef pmdp_get -static inline pmd_t pmdp_get(pmd_t *pmdp) +static inline pmd_t pmdp_get(const pmd_t *pmdp) { return READ_ONCE(*pmdp); } #endif #ifndef pudp_get -static inline pud_t pudp_get(pud_t *pudp) +static inline pud_t pudp_get(const pud_t *pudp) { return READ_ONCE(*pudp); } #endif #ifndef p4dp_get -static inline p4d_t p4dp_get(p4d_t *p4dp) +static inline p4d_t p4dp_get(const p4d_t *p4dp) { return READ_ONCE(*p4dp); } #endif #ifndef pgdp_get -static inline pgd_t pgdp_get(pgd_t *pgdp) +static inline pgd_t pgdp_get(const pgd_t *pgdp) { return READ_ONCE(*pgdp); } From 7c639e3732cc514245cbc108b682fefa30c336c2 Mon Sep 17 00:00:00 2001 From: Hao Ge Date: Wed, 16 Sep 2026 15:55:57 +0800 Subject: [PATCH 0785/1012] mm/alloc_tag: account for reserved tag ids in the kernel tag check The tag ids stored in the page flags include two reserved markers. Id 0 means the page has no tag and id 1 means the tag was cleared, so real tags start at CODETAG_ID_FIRST. The kernel-side check in alloc_tag_sec_init() compared kernel_tags.count alone against the addressable limit, so with the count at or just under the limit the last tag ids wrapped into those markers. Pages allocated through them then look the same as untagged pages on free, nothing is ever subtracted from the real tag and /proc/allocinfo shows that memory as still allocated. Add the missing CODETAG_ID_FIRST, same as tags_addressable(). Link: https://lore.kernel.org/20260916075557.121316-1-hao.ge@linux.dev Fixes: 4835f747d3ed ("alloc_tag: support for page allocation tag compression") Signed-off-by: Hao Ge Signed-off-by: Andrew Morton Acked-by: Suren Baghdasaryan Cc: --- mm/alloc_tag.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/alloc_tag.c b/mm/alloc_tag.c index b3341031047796..82e2c3448dcf2c 100644 --- a/mm/alloc_tag.c +++ b/mm/alloc_tag.c @@ -621,7 +621,7 @@ void __init alloc_tag_sec_init(void) kernel_tags.count = last_codetag - kernel_tags.first_tag; /* Check if kernel tags fit into page flags */ - if (kernel_tags.count > (1UL << NR_UNUSED_PAGEFLAG_BITS)) { + if (CODETAG_ID_FIRST + kernel_tags.count > (1UL << NR_UNUSED_PAGEFLAG_BITS)) { shutdown_mem_profiling(false); /* allocinfo file does not exist yet */ pr_err("%lu allocation tags cannot be references using %d available page flag bits. Memory allocation profiling is disabled!\n", kernel_tags.count, NR_UNUSED_PAGEFLAG_BITS); From 53655d5594a97ec66a4cac25cd26034b470bc10f Mon Sep 17 00:00:00 2001 From: David Carlier Date: Sun, 20 Sep 2026 16:50:02 +0100 Subject: [PATCH 0786/1012] mm/shmem: don't release a swapin-error marker as a swap entry A failed shmem swapin (eg, EIO) frees the swap slot and leaves a PTE_MARKER_POISONED entry in the page cache. On truncate or eviction shmem_free_swap() passes that marker to swap_put_entries_direct(), which warns because it is not a swap entry. There is nothing left to release either way. Skip the release for non-swap entries, as every other caller already does. Link: https://lore.kernel.org/20260920155002.1030454-1-devnexen@gmail.com Fixes: ac2d3268284b ("mm/swapfile.c: remove the unneeded checking") Signed-off-by: David Carlier Signed-off-by: Andrew Morton Reported-by: syzbot+23b25ba3c6bf971f9c57@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=23b25ba3c6bf971f9c57 Reviewed-by: Baolin Wang Cc: Baoquan he Cc: Barry Song Cc: Chis Li (Google) Cc: Hugh Dickens Cc: Kairui Song Cc: Kemeng Shi Cc: Nhat Pham Cc: --- mm/shmem.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/mm/shmem.c b/mm/shmem.c index f2a36a1b537506..07b2855dfb7bd9 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -1184,6 +1184,7 @@ static long shmem_free_swap(struct address_space *mapping, pgoff_t index, pgoff_t end, void *radswap) { XA_STATE(xas, &mapping->i_pages, index); + const softleaf_t swp = radix_to_swp_entry(radswap); unsigned int nr_pages = 0; pgoff_t base; void *entry; @@ -1200,8 +1201,9 @@ static long shmem_free_swap(struct address_space *mapping, } xas_unlock_irq(&xas); - if (nr_pages) - swap_put_entries_direct(radix_to_swp_entry(radswap), nr_pages); + /* A swapin-error marker holds no swap slot, so just drop it. */ + if (nr_pages && softleaf_is_swap(swp)) + swap_put_entries_direct(swp, nr_pages); return nr_pages; } From caa56f975c87788a4a4df0bf4d84de9cd7634b95 Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Sat, 19 Sep 2026 13:43:11 +0300 Subject: [PATCH 0787/1012] docs/mm: describe set_memory() and set_direct_map() APIs The set_memory() and set_direct_map() APIs change permissions of existing kernel mappings, but their semantics are only described by the code, and that code differs from architecture to architecture. Add Documentation/mm/kernel-page-tables.rst that briefly describes what the kernel page tables consist of, defines the semantics both APIs have in common, including the parts that are easy to get wrong, and lists the differences between the architecture implementations. Add kernel-doc comments for the generic set_memory() and set_direct_map() stubs and link them into Documentation/core-api/mm-api.rst. Link: https://lore.kernel.org/20260919-set-memory-docs-v2-1-a2a4b3657690@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Reviewed-by: Kevin Brodsky Assisted-by: copilot:claude-opus Cc: "David Hildenbrand (arm)" Cc: Jonathan Corbet Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Randy Dunlap Cc: Shuah khan Cc: Suren Baghdasaryan Cc: "Vlastimil Babka (SUSE)" --- Documentation/core-api/mm-api.rst | 9 + Documentation/mm/index.rst | 1 + Documentation/mm/kernel-page-tables.rst | 410 ++++++++++++++++++++++++ include/linux/set_memory.h | 110 +++++++ 4 files changed, 530 insertions(+) create mode 100644 Documentation/mm/kernel-page-tables.rst diff --git a/Documentation/core-api/mm-api.rst b/Documentation/core-api/mm-api.rst index c1d03a5a2a192e..e135b61bc87da8 100644 --- a/Documentation/core-api/mm-api.rst +++ b/Documentation/core-api/mm-api.rst @@ -52,6 +52,15 @@ Virtually Contiguous Mappings .. kernel-doc:: mm/vmalloc.c :export: +Kernel Page Table Permissions +============================= + +.. kernel-doc:: include/linux/set_memory.h + :doc: Kernel page table permissions + +.. kernel-doc:: include/linux/set_memory.h + :internal: + File Mapping and Page Cache =========================== diff --git a/Documentation/mm/index.rst b/Documentation/mm/index.rst index 13a79f5d092c0e..e9ae5cd82163af 100644 --- a/Documentation/mm/index.rst +++ b/Documentation/mm/index.rst @@ -25,6 +25,7 @@ see the :doc:`admin guide <../admin-guide/mm/index>`. physical_memory page_tables + kernel-page-tables process_addrs bootmem page_allocation diff --git a/Documentation/mm/kernel-page-tables.rst b/Documentation/mm/kernel-page-tables.rst new file mode 100644 index 00000000000000..b3148df07fc8b4 --- /dev/null +++ b/Documentation/mm/kernel-page-tables.rst @@ -0,0 +1,410 @@ +.. SPDX-License-Identifier: GPL-2.0 + +================== +Kernel Page Tables +================== + +Introduction +============ + +The kernel page tables are created early during boot and, unlike the page +tables of user processes, most of them remain static throughout the system +lifetime. + +Every architecture has a direct map (also called linear map) that maps the +physical memory at a fixed offset, so that a physical address can be +translated to a kernel virtual address with simple arithmetic. On most +architectures the direct map is a part of the kernel page tables, with a few +exceptions described in the `Direct map`_ section below. + +On 32-bit systems with high memory the direct map covers only a part of the +physical memory, see Documentation/mm/highmem.rst. + +The vmalloc area, present on every architecture with an MMU, is used for +allocations of virtually contiguous memory whose backing pages are not +necessarily physically contiguous, and for mapping of the device memory. Its +page tables are created and torn down at runtime, see +Documentation/mm/vmalloc.rst. + +Architectures that use the `SPARSEMEM_VMEMMAP` memory model reserve a range of +kernel address space for the memory map, so that `struct page` objects appear +as a virtually contiguous array indexed by the page frame number, see +Documentation/mm/memory-model.rst. + +Besides these, the kernel image may be mapped in a dedicated part of the +kernel address space rather than accessed through the direct map. In that +case its mapping is an alias of the direct map of the physical memory the +image occupies. + +The rest of the kernel address space is architecture specific. For instance, +x86 has a region for the EFI runtime services and s390 has a region for the +code that has to run in the 31-bit addressing mode. + +Direct map +========== + +On most architectures the direct map is an ordinary part of the kernel page +tables. It is created early during boot with the largest pages the hardware +and the kernel configuration allow. + +Several architectures are different. + +MIPS +---- + +MIPS does not map the physical memory with page tables at all. Instead, a +part of the kernel virtual address space is a window into the physical +address space: the hardware translates the addresses that fall into that +window by a fixed transformation of the address bits, without walking the +page tables and without using the TLB. The memory attributes, such as +cacheability and the privilege level required to access the memory, are a +property of the window rather than of an individual page. + +There are no page table entries describing the direct map, so its properties +cannot be changed for an individual page. On 32-bit systems the window covers +only 512 MiB of the physical memory, so everything above that is high memory. + +LoongArch +--------- + +Like MIPS, LoongArch maps the physical memory with a hardware window. The +window covers 256 TiB of the physical address space on 64-bit and 512 MiB on +32-bit systems, and everything above that is high memory. + +The window occupies the lower part of the kernel address space. The upper +part, which includes the vmalloc area, is mapped with kernel page tables. + +PowerPC with the hash MMU +------------------------- + +On 64-bit PowerPC systems with the hash MMU the direct map does not exist in +the Linux page tables. It is installed into the hardware hash page table early +during boot. + +Modifying such mappings requires updating the hash page table directly, and +the hash MMU code implements this only for the kernel image permissions, +`debug_pagealloc` and KFENCE. + +With the radix MMU the direct map is a part of the ordinary kernel page +tables. + +Modifying the kernel page tables +================================ + +Except for the vmalloc area, the kernel page tables are mostly static. Still, +there are cases when the permissions of existing kernel mappings have to be +updated, for instance when a module is loaded and its text becomes read-only +and executable, or when a page is temporarily removed from the direct map to +reduce its exposure. + +There are two families of functions for this, both declared in +`include/linux/set_memory.h`: + +* `set_memory_*()` change permissions of an arbitrary kernel mapping. They + take a kernel virtual address and the number of pages. + +* `set_direct_map_*()` change permissions of the direct mapping of the page + frame represented by a `struct page`. They take a `struct page` pointer and + the number of pages. + +Architectures that implement `set_memory()` select `CONFIG_ARCH_HAS_SET_MEMORY` + +Architectures that implement `set_direct_map()` select +`CONFIG_ARCH_HAS_SET_DIRECT_MAP`. + +Common semantics +---------------- + +Ranges +~~~~~~ + +The `set_memory()` functions expect a range described by a start address and a +number of pages. The address must be page aligned and the entire range must be +covered by page table entries the architecture knows how to update. + +The entries may be marked as not present, set_memory_p() and set_memory_valid() +exist exactly to bring such a mapping back. + +When a range does not qualify, an architecture will usually say so by returning +an error and sometimes by a WARN()ing as well, unless it prefers to keep it to +itself and return success, see `Architecture specific differences`_. + +The `set_direct_map()` functions expect a range described by the first +`struct page` and a number of pages, and they update the direct map starting +at that page. + +The pages that follow the first one are updated regardless of what they are, so +the caller has to make sure that the range does not extend beyond the memory it +owns. + +Some architectures cannot split a large mapping, and they reject a range that +is a part of one, see `Architecture specific differences`_. + +Calling `set_memory()` with the number of pages set to zero is a no-op that +returns success, except on arm64, where doing nothing to the wrong address is +still an error. + +Aliases +~~~~~~~ + +A physical page may be mapped several times, for instance in the direct map +and in the vmalloc area, and the permissions of these mappings may differ. + +Whether the direct map alias is updated by a `set_memory()` call, and +which permission bits make it there, is entirely up to the architecture, and +there is not much agreement between them, see +`Architecture specific differences`_. + +Relying on that is a gamble; code that needs the direct map alias to change +should say so with the `set_direct_map()` APIs. + +Failures and partial updates +~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +Both families of functions return 0 on success and a negative error code on +failure. The most common failures are `-EINVAL` for a range that cannot be +handled and `-ENOMEM` when splitting a large mapping fails to allocate a page +table. + +Some of the range checks happen upfront, so their failure leaves the page +tables unchanged. + +**The update is not atomic and there is no rollback.** + +The architecture implementations walk the range and update the page tables as +they go, and they stop at the first entry that cannot be updated. When an error +is returned, an arbitrary prefix of the range may have been updated already, +and the same is true for the direct map alias when the architecture updates it. + +For example, when a range spans two large mappings and splitting the second +one fails because there is no memory for a page table, the first one is +already split and updated by the time the error is returned. + +None of the APIs inform the caller where in the range they failed, so reverting +such a partial update is possible in principle but unreliable in practice. + +The caller may try to restore the original permissions over the entire range, +but that revert goes through the very code that has just failed, which does +not inspire much confidence. + +The callers should therefore be prepared to give up on the memory in question: +leak it or panic, but never return it to the allocator before the permissions +are restored and never assume that the requested permissions are in effect. +Neither option is appealing, but both beat handing out a page whose +permissions nobody knows. + +TLB flushing +~~~~~~~~~~~~ + +The `set_memory()` functions flush the TLB for the affected range before +they return, so that the new permissions are in effect for every CPU. + +The `set_direct_map()` functions have `_noflush` in their names because when +they were first introduced on x86, the intention was that the TLB flushing +could be optimized by letting the caller handle it. + +For example, vfree() batches the TLB flushes for the areas allocated with +`VM_FLUSH_RESET_PERMS`, folding the flush of the direct map into the flush it +has to do for the vmalloc mapping anyway. + +Some architectures flush the TLB in the `_noflush` functions anyway, so the +name is best read as a suggestion. It does not make the flush by the caller +unnecessary, it only makes it more expensive. + +A caller that changes the permissions to more restrictive ones must flush the +TLB itself. + +Context +~~~~~~~ + +Architectures use different locking mechanisms to synchronize kernel page table +updates, and both families of functions may sleep, for instance when they +allocate memory to split a large mapping. + +The caller cannot presume it is safe to call these APIs from an atomic context. + +The `set_direct_map()` functions must not be called for high memory pages, +which have no direct map alias to update. + +Unimplemented APIs silently succeed +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +When an architecture does not implement these APIs, the generic stubs in +`include/linux/set_memory.h` return 0, that is, they report success for work +they have no intention of doing. + +The same happens inside several architecture implementations. The arm64 +`set_direct_map()` functions return 0 when can_set_direct_map() is false, and +the LoongArch `set_memory()` functions return 0 for the addresses in its +windowed direct mapping, which is not backed by page tables at all. + +Returning success without actually updating the page tables is a deliberate +trade-off that keeps the callers free of `#ifdef`\ s, but they have to realize: + +**a return value of 0 does not imply that the permissions were actually +changed.** + +For best-effort hardening that is good enough. When correctness or security +depends on the permissions, the caller has to make sure the architecture +really implements what it needs. For instance, secretmem depends on +`CONFIG_ARCH_HAS_SET_DIRECT_MAP` and calls can_set_direct_map() at runtime. + +Architecture specific differences +================================= + +The APIs are implemented by seven architectures and, beyond the common +semantics described above, their behaviour differs in several respects. + +Which of the APIs are implemented: + +========= ===================== ========================= +Arch `ARCH_HAS_SET_MEMORY` `ARCH_HAS_SET_DIRECT_MAP` +========= ===================== ========================= +arm yes no +arm64 yes yes +loongarch yes yes +powerpc yes no +riscv yes (MMU only) yes (MMU only) +s390 yes yes +x86 yes yes +========= ===================== ========================= + +Only set_memory_ro(), set_memory_rw(), set_memory_x() and set_memory_nx() are +available everywhere, and even these are not universal: some architectures +restrict the ranges that can be modified, for instance arm64 rejects direct map +addresses and accepts only addresses in vmalloc space. The architectures that +may run on hardware without an execute permission bit, like x86 and s390, +silently skip the update of the executable bit. + +set_memory_rox() has a generic implementation that calls set_memory_ro() and +set_memory_x() in turn; PowerPC, s390 and x86 override it with a single-pass +version. + +Several architectures define additional APIs. Some of those may have identical +semantics but different names. For example, making a mapping present or not +present is spelled differently: set_memory_p() and set_memory_np() on x86 and +PowerPC, set_memory_valid() on arm and arm64. + +The direct map and the kernel image are normally mapped with the largest +possible pages, and changing the permissions of a single page inside such a +mapping requires splitting it, which not every architecture can do. + +arm +--- + +* Does not implement `set_direct_map()`. +* Provides set_memory_valid(). +* set_memory_ro(), set_memory_rw(), set_memory_x() and set_memory_nx() accept + only vmalloc and module addresses. +* set_memory_valid() accepts any address. +* Does not update mapping aliases. + +arm64 +----- + +* Provides set_memory_valid(). +* Provides the memory encryption helpers, which are effective only when the + kernel runs as a confidential guest. +* set_memory_ro(), set_memory_rw(), set_memory_x() and set_memory_nx() accept + only vmalloc and module addresses: + + - the range must fit in the VM area that contains its start + - the VM area must have `VM_ALLOC` set and `VM_ALLOW_HUGE_VMAP` clear + +* set_memory_valid() accepts any address. +* The encryption helpers accept only direct map addresses. +* The `set_memory()` functions propagate the read-only and the read-write + changes to the direct map alias when `rodata=full` is in effect. +* Splits leaf mappings before the update on the hardware that supports it. + The split itself may be partial when it fails midway, but the permissions are + left untouched in that case. + Without support for splitting large mappings, a range that covers a leaf + entry only partially fails with a WARN()ing and `-EINVAL`. + If a range spans one or more full leaf entries and a partial leaf entry, the + permissions of the full leaf entries are updated before the failure. +* The `set_direct_map()` functions return 0 without doing anything when the + direct map cannot be modified, see can_set_direct_map(). +* Skips the TLB flush in `set_memory()` when the update only turns an invalid + mapping into a valid one. +* Does not flush TLB in `set_direct_map()`. + +LoongArch +--------- + +* Accepts only the addresses above the hardware window and silently returns + success for the rest, see `Direct map`_. +* Does not update mapping aliases. +* Does not split anything: a leaf entry is updated as a whole, which changes + the permissions of the entire large mapping. +* Flushes the TLB in `set_direct_map()`. + +PowerPC +------- + +* Does not implement `set_direct_map()`. +* Provides set_memory_np() and set_memory_p(). +* Rejects huge vmalloc mappings. +* With the hash MMU on 64-bit systems accepts nothing but the vmalloc and the + I/O regions. +* With the radix MMU accepts direct map addresses, but still cannot split a + large mapping. +* Does not update mapping aliases. + +riscv +----- + +* Implements both APIs only when the MMU is enabled. +* Provides set_memory_rw_nx(). +* The `set_memory()` functions accept any mapped kernel address, including the + direct map, but a vmalloc range must have the `pages` array of its VM area + populated, which rules out vmap() and ioremap() mappings. +* On 64-bit systems the `set_memory()` functions update the direct map alias of + a vmalloc range, including the executable bit. +* Does not split vmalloc ranges: a leaf entry is updated as a whole, which + changes the permissions of the entire large mapping. +* Splits the direct map on 64-bit systems. +* Flushes the TLB in `set_direct_map()`. + +s390 +---- + +* Provides set_memory_4k(), set_memory_rwnx() and the + `__set_memory_*(start, end)` variants that take a range rather than a page + count. +* The `set_memory()` functions accept any mapped kernel address, including the + direct map. +* Skips the update of the executable bit when the hardware has no support for + it. +* The `set_memory()` functions propagate only the read-only and read-write + changes to the direct map alias of a `VM_ALLOC` area, and deliberately not + the executable bit. +* Splits leaf PUD and PMD entries when the range is not aligned to them or when + set_memory_4k() is requested. +* Updates the page table entries with instructions that invalidate the + corresponding TLB entries, so no separate flush is needed anywhere. + +x86 +--- + +* Provides the largest set of operations on top of the common ones: + + - the cache attribute helpers: set_memory_uc(), set_memory_wc(), + set_memory_wb() + - presence control: set_memory_np() and set_memory_p() + - set_memory_4k() + - set_memory_global() and set_memory_nonglobal() + - the array variants that operate on `struct page` arrays or arrays of + virtual addresses + - memory encryption: set_memory_encrypted() and set_memory_decrypted() + +* The `set_memory()` functions accept any mapped kernel address, including the + direct map, and silently succeed for the unmapped holes inside it. +* Does nothing in set_memory_x() and set_memory_nx() when the CPU has no + execute permission bit. +* The `set_memory()` functions apply the change to the direct map alias and, + for the kernel image, to the high kernel mapping. The NX bit is never + propagated, so that the direct map stays non-executable. +* Splits large mappings on demand and can collapse them back when the + permissions become uniform again. +* Does not flush the TLB in `set_direct_map()`, but splitting a large + mapping flushes it anyway. diff --git a/include/linux/set_memory.h b/include/linux/set_memory.h index 3fe293cfed8cc8..27c32fbabfec90 100644 --- a/include/linux/set_memory.h +++ b/include/linux/set_memory.h @@ -5,16 +5,92 @@ #ifndef _LINUX_SET_MEMORY_H_ #define _LINUX_SET_MEMORY_H_ +/** + * DOC: Kernel page table permissions + * + * The set_memory() and set_direct_map() APIs update permissions of existing + * kernel mappings. + * + * The set_memory() functions operate on a range of kernel virtual addresses, + * the set_direct_map() functions operate on the direct map. + * + * The updates are not atomic: when a call fails, an arbitrary prefix of the + * range may have been updated already and there is no automatic rollback. + * A caller must restore the required permissions before reusing or freeing + * the memory. + * + * When an architecture does not implement these APIs they succeed without + * doing anything, so a return value of 0 does not mean that the permissions + * were actually changed. + * + * Callers that depend on the permissions being applied must ensure that the + * architecture supports the required operation for the target addresses. The + * Kconfig symbols alone do not guarantee this. + * + * See Documentation/mm/kernel-page-tables.rst for the details and for the + * differences between the architecture implementations. + */ + #ifdef CONFIG_ARCH_HAS_SET_MEMORY #include #else +/** + * set_memory_ro - make a kernel mapping read-only + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int __must_check set_memory_ro(unsigned long addr, int numpages) { return 0; } + +/** + * set_memory_rw - make a kernel mapping writable + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int __must_check set_memory_rw(unsigned long addr, int numpages) { return 0; } + +/** + * set_memory_x - make a kernel mapping executable + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int __must_check set_memory_x(unsigned long addr, int numpages) { return 0; } + +/** + * set_memory_nx - make a kernel mapping non-executable + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int __must_check set_memory_nx(unsigned long addr, int numpages) { return 0; } #endif #ifndef set_memory_rox +/** + * set_memory_rox - make a kernel mapping read-only and executable + * @addr: page aligned start of the kernel virtual address range + * @numpages: number of pages in the range + * + * A failure may leave the range read-only but not executable. + * + * Flushes the TLB for the range. + * + * Return: 0 on success, negative error code on failure. + */ static inline int set_memory_rox(unsigned long addr, int numpages) { int ret = set_memory_ro(addr, numpages); @@ -25,11 +101,33 @@ static inline int set_memory_rox(unsigned long addr, int numpages) #endif #ifndef CONFIG_ARCH_HAS_SET_DIRECT_MAP +/** + * set_direct_map_invalid_noflush - remove pages from the direct map + * @page: first page to update + * @nr: number of pages to update + * + * Makes the direct mapping of @nr pages starting at @page not present. + * The caller is responsible for any required TLB flushing. + * + * Return: 0 on success, negative error code on failure. + */ static inline int set_direct_map_invalid_noflush(struct page *page, unsigned int nr) { return 0; } + +/** + * set_direct_map_default_noflush - restore the direct map of pages + * @page: first page to update + * @nr: number of pages to update + * + * Restores the default kernel permissions of the direct mapping of @nr + * pages starting at @page. + * The caller is responsible for any required TLB flushing. + * + * Return: 0 on success, negative error code on failure. + */ static inline int set_direct_map_default_noflush(struct page *page, unsigned int nr) { @@ -46,6 +144,18 @@ static inline bool kernel_page_present(struct page *page) * boot time. Let them overrive this query. */ #ifndef can_set_direct_map +/** + * can_set_direct_map - check if the direct map can be modified + * + * Available with CONFIG_ARCH_HAS_SET_DIRECT_MAP. Architectures may override + * this to report whether direct map updates are enabled at runtime. + * Even though the generic implementation returns true this does not guarantee + * that every address can be updated. + * + * See Documentation/mm/kernel-page-tables.rst for the details + * + * Return: true unless the architecture reports direct map updates disabled. + */ static inline bool can_set_direct_map(void) { return true; From 9c39e087b9511f3d93479541d9fd0471491176bc Mon Sep 17 00:00:00 2001 From: Mike Rapoport Date: Tue, 22 Sep 2026 12:42:55 +0300 Subject: [PATCH 0788/1012] docs-mm-describe-set_memory-and-set_direct_map-apis-fix fix phrasing, per Kevin Link: https://lore.kernel.org/arJNn2QD_pY6e14R@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Cc: Kevin Brodsky Cc: David Hildenbrand Cc: Jonathan Corbet Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Randy Dunlap Cc: Shuah Khan Cc: Suren Baghdasaryan Cc: Vlastimil Babka --- Documentation/mm/kernel-page-tables.rst | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/Documentation/mm/kernel-page-tables.rst b/Documentation/mm/kernel-page-tables.rst index b3148df07fc8b4..c16de34b4f79d8 100644 --- a/Documentation/mm/kernel-page-tables.rst +++ b/Documentation/mm/kernel-page-tables.rst @@ -103,9 +103,9 @@ There are two families of functions for this, both declared in * `set_memory_*()` change permissions of an arbitrary kernel mapping. They take a kernel virtual address and the number of pages. -* `set_direct_map_*()` change permissions of the direct mapping of the page - frame represented by a `struct page`. They take a `struct page` pointer and - the number of pages. +* `set_direct_map_*()` change permissions of the direct mapping for the range + of page frames starting at the page represented by a `struct page`. They take + a `struct page` pointer and the number of pages. Architectures that implement `set_memory()` select `CONFIG_ARCH_HAS_SET_MEMORY` From 287964d32a1f3be91a03bdba4f6d23e518eceddc Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:31 +0100 Subject: [PATCH 0789/1012] selftests/mm: raise the khugepaged test-case cap Patch series "selftests/mm: improve khugepaged coverage", v6. khugepaged collapses to mTHP orders since 7.2, and 7.3 added three selftest cases for it: the generic collapse cases run at one order named by -c, with the result detected by counting folios of that order. That leaves the collapse path largely untested. The suite does not run where a PMD is 512M. A folio count cannot say where a collapse landed. Fixed sleeps cannot tell "not collapsed" from "not scanned yet". And nothing exercises collapse under contention. Close those gaps in order: - Make the suite run at a 512M PMD: scale the collapse wait with the PMD size, skip what such a PMD cannot serve, make the swapout the swap cases depend on deterministic, and keep khugepaged out of the MADV_COLLAPSE cases. - Detect results per window rather than by count, with folio-order helpers in vm_util that are checked against the kernel before any collapse test trusts them. - Drive khugepaged deterministically: a completion barrier that wakes the daemon and waits for a full pass, and a check that one pass yields one attributed collapse. - Cover collapse at every supported order by default: which window collapses, occupancy at both limits, sources that are already large folios, and a fork-shared source under concurrent writes. - Race collapse against everything that can touch its sources, at both occupancy limits and over whole-table zaps, checked by content and by the kernel's own assertions. Everything passes on an unmodified kernel. This patch (of 19): TEST() ends the run with "MAX_TEST_CASES is too small" when the table fills, and the table holds 64. A full invocation already registers 63, so the next case added anywhere aborts the whole suite before a single test runs. Raise the cap to 256. The table is a static array of small structs, so the room costs nothing worth counting. Link: https://lore.kernel.org/20260919002451.496763-1-kirill@shutemov.name Link: https://lore.kernel.org/20260919002451.496763-2-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Usama Arif Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Michal Hocko Cc: Muhammad Usama Anjum Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 525108cace54ab..c44bc18f753659 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1290,7 +1290,7 @@ struct test_case { test_fn fn; }; -#define MAX_TEST_CASES 64 +#define MAX_TEST_CASES 256 static struct test_case test_cases[MAX_TEST_CASES]; static int nr_test_cases; From 30a2b818d94dd137a97f046b60b9defd8d797163 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:32 +0100 Subject: [PATCH 0790/1012] selftests/mm: skip collapse_compound_extreme() where the PMD is too large collapse_compound_extreme() builds a PTE table full of distinct PTE-mapped compound pages by cycling hpage_pmd_nr fault-time THPs through mremap. It therefore needs hpage_pmd_nr PMD-order allocations in a row. That is fine at a 2M PMD (4K base pages) or a 32M one (16K). A 512M PMD -- arm64 with 64K base pages -- makes each of those an order-13 allocation, which the allocator cannot reliably hand out even once, let alone 8192 times. The failure is not a quiet one: the case calls ksft_exit_fail_msg(), so the whole binary stops and every case after it is lost. Skip the case where the PMD is larger than 32M. The MADV_COLLAPSE cases still cover PMD-order collapse on those configurations, and 4K and 16K PMDs are unaffected. Link: https://lore.kernel.org/20260919002451.496763-3-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Michal Hocko Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index c44bc18f753659..8a6d708026b725 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -940,6 +940,16 @@ static void collapse_compound_extreme(struct collapse_context *c, struct mem_ops void *p; int i; + /* + * This needs hpage_pmd_nr PMD-order allocations in a row, which the + * allocator will not supply if the PMD is very large. + */ + if (hpage_pmd_size > (32UL << 20)) { + ksft_test_result_skip("%s: PMD too large for fault-time THP construction\n", + __func__); + return; + } + p = ops->setup_area(1); ksft_print_msg("Construct PTE page table full of different PTE-mapped compound pages\n"); for (i = 0; i < hpage_pmd_nr; i++) { From a1e46e5e4ace911903eaaa4bb99c3b4ac4a1278d Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:33 +0100 Subject: [PATCH 0791/1012] selftests/mm: scale khugepaged's collapse wait with the PMD size wait_for_scan() gives every case the same three seconds, whatever the huge page costs to build. collapse_full() asks for four of them: 8M at a 2M PMD, but 2G at a 512M PMD -- arm64 with 64K base pages. Three seconds is thin at that size, and the case has reported a failure for a collapse that was still going. The timeout is a ceiling on a poll loop, not a sleep: the loop stops as soon as ops->check_huge() sees the collapse, or as soon as full_scans has advanced by two. Raising it costs a passing case nothing. Across 80 runs of collapse_full() on arm64 with 64K pages the wait was half a second in 73 of them, with a tail to two seconds. Keep three seconds as the floor and add a second per 128M collapsed. A 2M PMD is unchanged, so x86-64 is too; a 512M PMD gets 19 seconds. On arm64 with 64K pages a passing ./khugepaged all:anon takes 49 seconds under TCG before and after this change. Link: https://lore.kernel.org/20260919002451.496763-4-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Michal Hocko Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 8a6d708026b725..189cc4fee18ffe 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -561,8 +561,11 @@ static bool wait_for_scan(const char *msg, char *p, size_t len, int nr_hpages, int collap_order, struct mem_ops *ops) { unsigned long hpage_size = page_size << collap_order; - int full_scans; - int timeout = 6; /* 3 seconds */ + unsigned long bytes = (unsigned long)nr_hpages * hpage_size; + int timeout, full_scans; + + /* Half-second ticks: three seconds floor, plus a second per 128M */ + timeout = 6 + 2 * (bytes / (128UL << 20)); /* Sanity check */ if (!ops->check_huge(p, len, 0, hpage_size)) From 2a6e2bae098e52da2a90ce31319c65cdd99c724f Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:34 +0100 Subject: [PATCH 0792/1012] selftests/mm: skip khugepaged page cache cases without a PMD folio The page cache caps folio order at MAX_PAGECACHE_ORDER, which is below the PMD order on arm64 with 64K pages, where a PMD is 512M. A PMD-sized page cache folio is impossible there, so the kernel refuses these collapses: MADV_COLLAPSE answers -EINVAL and khugepaged passes over the range. Four shmem cases ask for a PMD-sized folio anyway, fail, and the run bails out in the middle. Skip the shmem and file mem types where the cap is below the PMD order. The cap is not shmem-specific: it applies to every file folio. Add thp_file_supported_orders() to read the orders the page cache allows. Anonymous collapse is unaffected: its orders are not capped this way. Link: https://lore.kernel.org/20260919002451.496763-5-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/lib/mm/hugepage_settings.h | 9 +++++++++ tools/testing/selftests/mm/khugepaged.c | 19 +++++++++++++++++++ 2 files changed, 28 insertions(+) diff --git a/tools/lib/mm/hugepage_settings.h b/tools/lib/mm/hugepage_settings.h index 548e9d288d1d16..94d9fc747f4979 100644 --- a/tools/lib/mm/hugepage_settings.h +++ b/tools/lib/mm/hugepage_settings.h @@ -87,6 +87,15 @@ void thp_set_read_ahead_path(char *path); unsigned long thp_supported_orders(void); unsigned long thp_shmem_supported_orders(void); +/* + * The per-order shmem_enabled attribute is created for the orders the page + * cache can hold, not just for shmem, so it answers for regular files too. + */ +static inline unsigned long thp_file_supported_orders(void) +{ + return thp_shmem_supported_orders(); +} + bool thp_available(void); bool thp_is_enabled(void); diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 189cc4fee18ffe..5318f3cfc0d044 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1358,6 +1358,25 @@ int main(int argc, char **argv) setbuf(stdout, NULL); + /* + * Without a PMD-order page cache folio the kernel refuses these + * collapses, so there is nothing to test. + */ + if (!(thp_file_supported_orders() & (1UL << hpage_pmd_order))) { + if (shmem_ops) { + ksft_print_msg("no PMD-order page cache folio: skipping shmem\n"); + shmem_ops = NULL; + } + if (read_only_file_ops) { + ksft_print_msg("no PMD-order page cache folio: skipping file\n"); + read_only_file_ops = NULL; + read_write_file_read_ops = NULL; + read_write_file_write_ops = NULL; + } + if (!anon_ops && !shmem_ops && !read_only_file_ops) + ksft_exit_skip("No mem_type left to run\n"); + } + default_settings.khugepaged.max_ptes_none = hpage_pmd_nr - 1; default_settings.khugepaged.max_ptes_swap = hpage_pmd_nr / 8; default_settings.khugepaged.max_ptes_shared = hpage_pmd_nr / 2; From 42761018d536a265a56520b8db4626a7fbcf629b Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:35 +0100 Subject: [PATCH 0793/1012] selftests/mm: make the swap cases' swapout reliable collapse_swapin_single_pte() and collapse_max_ptes_swap() swap a range out and then require smaps to report exactly the count they asked for. Two things keep that count from arriving. MADV_PAGEOUT is best effort, so the count often turns up a moment late. And wait_for_scan() leaves the range eligible for collapsing, so khugepaged is still working on it. Collapsing reads the swapped-out pages back in, so the daemon empties the swap as fast as the case fills it. On arm64 with 64K pages max_ptes_swap is 1024 pages, which is 64M a step, and the case loses the race: # Swapout 1024 of 8192 pages... Fail not ok 10 collapse_max_ptes_swap Retry for up to two seconds, holding the range out of khugepaged's reach meanwhile. The collapse each case runs next restores MADV_HUGEPAGE, so only the setup is affected. If the pages still won't swap out, skip: no swap, swap too small or full, a memcg cap or busy writeback. None of that is a kernel bug. Link: https://lore.kernel.org/20260919002451.496763-6-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Muhammad Usama Anjum Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 41 +++++++++++++++++-------- 1 file changed, 29 insertions(+), 12 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 5318f3cfc0d044..13a2a47ab1108e 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -221,6 +221,29 @@ static bool check_swap(void *addr, unsigned long size) return swap; } +static bool swapout_range(void *p, unsigned long size) +{ + int i; + + /* keep khugepaged from collapsing the range and swapping it back in */ + if (madvise(p, size, MADV_NOHUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_NOHUGEPAGE)"); + + /* + * Retry several times because MADV_PAGEOUT is best effort. Sleep + * between the retries to give outstanding writeback a chance to + * finish. + */ + for (i = 0; i < 40; i++) { + if (madvise(p, size, MADV_PAGEOUT)) + ksft_exit_fail_perror("madvise(MADV_PAGEOUT)"); + if (check_swap(p, size)) + return true; + usleep(50 * 1000); + } + return false; +} + static void *alloc_mapping(int nr) { void *p; @@ -828,12 +851,10 @@ static void collapse_swapin_single_pte(struct collapse_context *c, struct mem_op ops->fault(p, 0, hpage_pmd_size); ksft_print_msg("Swapout one page..."); - if (madvise(p, page_size, MADV_PAGEOUT)) - ksft_exit_fail_perror("madvise(MADV_PAGEOUT)"); - if (check_swap(p, page_size)) { + if (swapout_range(p, page_size)) { success("OK"); } else { - fail("Fail"); + skip("Could not swap out"); goto out; } @@ -854,12 +875,10 @@ static void collapse_max_ptes_swap(struct collapse_context *c, struct mem_ops *o ops->fault(p, 0, hpage_pmd_size); ksft_print_msg("Swapout %d of %d pages...", max_ptes_swap + 1, hpage_pmd_nr); - if (madvise(p, (max_ptes_swap + 1) * page_size, MADV_PAGEOUT)) - ksft_exit_fail_perror("madvise(MADV_PAGEOUT)"); - if (check_swap(p, (max_ptes_swap + 1) * page_size)) { + if (swapout_range(p, (max_ptes_swap + 1) * page_size)) { success("OK"); } else { - fail("Fail"); + skip("Could not swap out"); goto out; } @@ -871,12 +890,10 @@ static void collapse_max_ptes_swap(struct collapse_context *c, struct mem_ops *o ops->fault(p, 0, hpage_pmd_size); ksft_print_msg("Swapout %d of %d pages...", max_ptes_swap, hpage_pmd_nr); - if (madvise(p, max_ptes_swap * page_size, MADV_PAGEOUT)) - ksft_exit_fail_perror("madvise(MADV_PAGEOUT)"); - if (check_swap(p, max_ptes_swap * page_size)) { + if (swapout_range(p, max_ptes_swap * page_size)) { success("OK"); } else { - fail("Fail"); + skip("Could not swap out"); goto out; } From b1013710108d74f725441af3bb42504c6f0144c2 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:36 +0100 Subject: [PATCH 0794/1012] selftests/mm: stop khugepaged during the MADV_COLLAPSE cases __madvise_collapse() turns THP off before each MADV_COLLAPSE, both to keep khugepaged out of the range and to prove MADV_COLLAPSE ignores the setting. It clears the global controls only, which is no longer enough. A per-order control overrides them, and -s, which makes the cases fault in folios of one order, leaves that order's control at "always". khugepaged then collapses the very range the case is working on, and the case fails on a collapse that was interfered with rather than refused. Clear the per-order controls too. MADV_COLLAPSE does not consult them: anon never did, and shmem stopped with "mm: shmem: ignore sysfs configs for shmem forced collapse". Link: https://lore.kernel.org/20260919002451.496763-7-kirill@shutemov.name Fixes: b7f16963efe7 ("mm/khugepaged: run khugepaged for all orders") Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Muhammad Usama Anjum Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 13a2a47ab1108e..e013eebc7136ed 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -538,8 +538,8 @@ static bool is_anon(struct mem_ops *ops) static void __madvise_collapse(const char *msg, char *p, int nr_hpages, struct mem_ops *ops, bool expect) { - int ret; struct thp_settings settings = *thp_current_settings(); + int ret, i; ksft_print_msg("%s...", msg); @@ -555,6 +555,10 @@ static void __madvise_collapse(const char *msg, char *p, int nr_hpages, */ settings.thp_enabled = THP_NEVER; settings.shmem_enabled = SHMEM_NEVER; + for (i = 0; i < NR_ORDERS; i++) { + settings.hugepages[i].enabled = THP_NEVER; + settings.shmem_hugepages[i].enabled = SHMEM_NEVER; + } thp_push_settings(&settings); /* Clear VM_NOHUGEPAGE */ From 59b1bcded25aeaa0334f011800953ec56e695109 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:37 +0100 Subject: [PATCH 0795/1012] selftests/mm: move is_backed_by_folio() into vm_util Checking that an address range is backed by a folio of a given order is useful to any test that builds or collapses large folios. mTHP collapse coverage in the khugepaged selftest needs exactly that. split_huge_page_test.c already has the building block: is_backed_by_folio() reads the compound head and tail flags from /proc/kpageflags to classify the folio behind a page. Move it into vm_util so other tests can use it. No functional change. Link: https://lore.kernel.org/20260919002451.496763-8-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Mike Rapoport (Microsoft) Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Michal Hocko Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- .../selftests/mm/split_huge_page_test.c | 61 ------------------- tools/testing/selftests/mm/vm_util.c | 61 +++++++++++++++++++ tools/testing/selftests/mm/vm_util.h | 2 + 3 files changed, 63 insertions(+), 61 deletions(-) diff --git a/tools/testing/selftests/mm/split_huge_page_test.c b/tools/testing/selftests/mm/split_huge_page_test.c index 68f508c9a355fc..c5d96a4b1db355 100644 --- a/tools/testing/selftests/mm/split_huge_page_test.c +++ b/tools/testing/selftests/mm/split_huge_page_test.c @@ -41,67 +41,6 @@ const char *kpageflags_proc = "/proc/kpageflags"; int pagemap_fd; int kpageflags_fd; -static bool is_backed_by_folio(char *vaddr, int order, int pagemap_fd, - int kpageflags_fd) -{ - const uint64_t folio_head_flags = KPF_THP | KPF_COMPOUND_HEAD; - const uint64_t folio_tail_flags = KPF_THP | KPF_COMPOUND_TAIL; - const unsigned long nr_pages = 1UL << order; - unsigned long pfn_head; - uint64_t pfn_flags; - unsigned long pfn; - unsigned long i; - - pfn = pagemap_get_pfn(pagemap_fd, vaddr); - - /* non present page */ - if (pfn == -1UL) - return false; - - if (pageflags_get(pfn, kpageflags_fd, &pfn_flags)) - goto fail; - - /* check for order-0 pages */ - if (!order) { - if (pfn_flags & (folio_head_flags | folio_tail_flags)) - return false; - return true; - } - - /* non THP folio */ - if (!(pfn_flags & KPF_THP)) - return false; - - pfn_head = pfn & ~(nr_pages - 1); - - if (pageflags_get(pfn_head, kpageflags_fd, &pfn_flags)) - goto fail; - - /* head PFN has no compound_head flag set */ - if ((pfn_flags & folio_head_flags) != folio_head_flags) - return false; - - /* check all tail PFN flags */ - for (i = 1; i < nr_pages; i++) { - if (pageflags_get(pfn_head + i, kpageflags_fd, &pfn_flags)) - goto fail; - if ((pfn_flags & folio_tail_flags) != folio_tail_flags) - return false; - } - - /* - * check the PFN after this folio, but if its flags cannot be obtained, - * assume this folio has the expected order - */ - if (pageflags_get(pfn_head + nr_pages, kpageflags_fd, &pfn_flags)) - return true; - - /* If we find another tail page, then the folio is larger. */ - return (pfn_flags & folio_tail_flags) != folio_tail_flags; -fail: - ksft_exit_fail_msg("Failed to get folio info\n"); -} - static int check_after_split_folio_orders(char *vaddr_start, size_t len, int pagemap_fd, int kpageflags_fd, int orders[], int nr_orders) { diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index f8916b2abc2efb..d2c5a20d724a29 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -490,6 +490,67 @@ int pageflags_get(unsigned long pfn, int kpageflags_fd, uint64_t *flags) return 0; } +bool is_backed_by_folio(char *vaddr, int order, int pagemap_fd, + int kpageflags_fd) +{ + const uint64_t folio_head_flags = KPF_THP | KPF_COMPOUND_HEAD; + const uint64_t folio_tail_flags = KPF_THP | KPF_COMPOUND_TAIL; + const unsigned long nr_pages = 1UL << order; + unsigned long pfn_head; + uint64_t pfn_flags; + unsigned long pfn; + unsigned long i; + + pfn = pagemap_get_pfn(pagemap_fd, vaddr); + + /* non present page */ + if (pfn == -1UL) + return false; + + if (pageflags_get(pfn, kpageflags_fd, &pfn_flags)) + goto fail; + + /* check for order-0 pages */ + if (!order) { + if (pfn_flags & (folio_head_flags | folio_tail_flags)) + return false; + return true; + } + + /* non THP folio */ + if (!(pfn_flags & KPF_THP)) + return false; + + pfn_head = pfn & ~(nr_pages - 1); + + if (pageflags_get(pfn_head, kpageflags_fd, &pfn_flags)) + goto fail; + + /* head PFN has no compound_head flag set */ + if ((pfn_flags & folio_head_flags) != folio_head_flags) + return false; + + /* check all tail PFN flags */ + for (i = 1; i < nr_pages; i++) { + if (pageflags_get(pfn_head + i, kpageflags_fd, &pfn_flags)) + goto fail; + if ((pfn_flags & folio_tail_flags) != folio_tail_flags) + return false; + } + + /* + * check the PFN after this folio, but if its flags cannot be obtained, + * assume this folio has the expected order + */ + if (pageflags_get(pfn_head + nr_pages, kpageflags_fd, &pfn_flags)) + return true; + + /* If we find another tail page, then the folio is larger. */ + return (pfn_flags & folio_tail_flags) != folio_tail_flags; +fail: + ksft_exit_fail_msg("Failed to get folio info\n"); +} + /* If `ioctls' non-NULL, the allowed ioctls will be returned into the var */ int uffd_register_with_ioctls(int uffd, void *addr, uint64_t len, bool miss, bool wp, bool minor, uint64_t *ioctls) diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index 64a86e8a0c41bf..f12979a70135c6 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -99,6 +99,8 @@ int64_t allocate_transhuge(void *ptr, int pagemap_fd); int pageflags_get(unsigned long pfn, int kpageflags_fd, uint64_t *flags); int gather_folio_orders(char *vaddr_start, size_t len, int pagemap_fd, int kpageflags_fd, int orders[], int nr_orders); +bool is_backed_by_folio(char *vaddr, int order, int pagemap_fd, + int kpageflags_fd); int uffd_register(int uffd, void *addr, uint64_t len, bool miss, bool wp, bool minor); From 62ceaa806137020a08489ca34de37a88cc6a03f2 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:38 +0100 Subject: [PATCH 0796/1012] selftests/mm: add folio-order check for address ranges An mTHP collapse test needs to know that a range is backed by folios of the target order, and that they sit where a collapse would put them. Nothing answers both: is_backed_by_folio() classifies the folio behind a single page, and check_huge_anon() counts the folios of an order in a range without saying where they start. Add is_range_backed_by_order(). It requires every folio-sized, folio- aligned part of the range to map one folio of that order, head to tail, with the head at the start of the part. A part backed by two smaller folios fails, and so does a folio mapped off its natural alignment. The mTHP cases need both to tell a collapsed range from the one beside it. Link: https://lore.kernel.org/20260919002451.496763-9-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/vm_util.c | 47 ++++++++++++++++++++++++++++ tools/testing/selftests/mm/vm_util.h | 2 ++ 2 files changed, 49 insertions(+) diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index d2c5a20d724a29..1f88330fd0274a 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -551,6 +551,53 @@ bool is_backed_by_folio(char *vaddr, int order, int pagemap_fd, ksft_exit_fail_msg("Failed to get folio info\n"); } +/** + * is_range_backed_by_order() - check that a range is backed by @order folios + * @start: start of the range, a multiple of the folio size + * @len: length of the range in bytes, a multiple of the folio size + * @order: the folio order to check for + * @pagemap_fd: open /proc//pagemap of the range's owner + * @kpageflags_fd: open /proc/kpageflags + * + * Every folio-sized, folio-aligned part of the range must map one folio of + * @order, head to tail, with the head at the start of the part. A part + * backed by several smaller folios fails, and so does a folio mapped off + * its natural alignment. + * + * Returns: true if the whole range is backed that way, false otherwise. + */ +bool is_range_backed_by_order(char *start, size_t len, int order, + int pagemap_fd, int kpageflags_fd) +{ + const unsigned long nr_pages = 1UL << order; + const size_t folio_size = nr_pages * psize(); + char *vaddr; + + if ((uintptr_t)start % folio_size || len % folio_size) + return false; + + for (vaddr = start; vaddr < start + len; vaddr += folio_size) { + const unsigned long pfn = pagemap_get_pfn(pagemap_fd, vaddr); + unsigned long i; + + /* Not present, or a tail page */ + if (pfn == -1UL || pfn % nr_pages) + return false; + + for (i = 1; i < nr_pages; i++) { + char *page = vaddr + i * psize(); + + if (pagemap_get_pfn(pagemap_fd, page) != pfn + i) + return false; + } + + if (!is_backed_by_folio(vaddr, order, pagemap_fd, kpageflags_fd)) + return false; + } + + return true; +} + /* If `ioctls' non-NULL, the allowed ioctls will be returned into the var */ int uffd_register_with_ioctls(int uffd, void *addr, uint64_t len, bool miss, bool wp, bool minor, uint64_t *ioctls) diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index f12979a70135c6..0172003c16dcb9 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -101,6 +101,8 @@ int gather_folio_orders(char *vaddr_start, size_t len, int pagemap_fd, int kpageflags_fd, int orders[], int nr_orders); bool is_backed_by_folio(char *vaddr, int order, int pagemap_fd, int kpageflags_fd); +bool is_range_backed_by_order(char *start, size_t len, int order, + int pagemap_fd, int kpageflags_fd); int uffd_register(int uffd, void *addr, uint64_t len, bool miss, bool wp, bool minor); From c1116e6427fa551bc2e7eb9a4c7773755868c4cf Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:39 +0100 Subject: [PATCH 0797/1012] selftests/mm: add folio-order detection self-check The khugepaged mTHP tests detect collapse results with the vm_util folio-order helpers rather than smaps AnonHugePages, which only sees PMD mappings. If those helpers are wrong, every case built on them is wrong the same way, and nothing says so. Check them directly. For every anon THP order the kernel supports, fault memory in with only that order enabled. Require the helpers to classify the backing as exactly that order: not the order below it, and base-page memory as order 0. Run it in the thp category, ahead of ./khugepaged, so a broken helper is reported as itself rather than as a collapse failure. Verified on x86-64 4K (orders 0, 2-9) and arm64 64K (orders 0, 2-13). The test needs ALIGN(), which hmm-tests.c and migration.c each defined privately. Move it to vm_util.h and drop both copies. Link: https://lore.kernel.org/20260919002451.496763-10-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/Makefile | 1 + .../testing/selftests/mm/folio_order_check.c | 122 ++++++++++++++++++ tools/testing/selftests/mm/hmm-tests.c | 1 - tools/testing/selftests/mm/migration.c | 1 - tools/testing/selftests/mm/run_vmtests.sh | 2 + tools/testing/selftests/mm/vm_util.h | 2 + 6 files changed, 127 insertions(+), 2 deletions(-) create mode 100644 tools/testing/selftests/mm/folio_order_check.c diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index 7d69baeb93f430..cb32cf5d867e25 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -105,6 +105,7 @@ TEST_GEN_FILES += guard-regions TEST_GEN_FILES += merge TEST_GEN_FILES += rmap TEST_GEN_FILES += folio_split_race_test +TEST_GEN_FILES += folio_order_check TEST_GEN_FILES += soft-dirty ifeq ($(ARCH),x86_64) diff --git a/tools/testing/selftests/mm/folio_order_check.c b/tools/testing/selftests/mm/folio_order_check.c new file mode 100644 index 00000000000000..5eafbcc1b4f3c5 --- /dev/null +++ b/tools/testing/selftests/mm/folio_order_check.c @@ -0,0 +1,122 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Self-check for the vm_util folio-order helpers, is_backed_by_folio() and + * is_range_backed_by_order(), which the khugepaged mTHP cases use to detect + * collapse results. For every anon THP order the kernel supports, fault + * memory in with only that order enabled and require the helpers to report + * exactly that order. + */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include + +#include "kselftest.h" +#include "vm_util.h" +#include + +static int pagemap_fd; +static int kpageflags_fd; + +static char *alloc_aligned(size_t size) +{ + size_t len = size * 2; + char *p, *aligned; + + p = mmap(NULL, len, PROT_READ | PROT_WRITE, + MAP_ANONYMOUS | MAP_PRIVATE, -1, 0); + if (p == MAP_FAILED) + ksft_exit_fail_perror("mmap()"); + + aligned = (char *)ALIGN((uintptr_t)p, size); + if (aligned != p) + munmap(p, aligned - p); + if (aligned + size != p + len) + munmap(aligned + size, p + len - aligned - size); + + return aligned; +} + +static void check_order(int order) +{ + struct thp_settings settings = *thp_current_settings(); + size_t size = psize() << order; + bool ok = true; + char *p; + int i; + + for (i = 0; i < NR_ORDERS; i++) + settings.hugepages[i].enabled = THP_NEVER; + if (order) + settings.hugepages[order].enabled = THP_ALWAYS; + thp_push_settings(&settings); + + p = alloc_aligned(size); + *p = 1; + + if (!is_range_backed_by_order(p, size, order, pagemap_fd, kpageflags_fd)) { + ksft_print_msg("order %d not detected after fault\n", order); + ok = false; + } + + /* A lower order must be rejected: the folio is larger */ + if (order && is_range_backed_by_order(p, size, order - 1, + pagemap_fd, kpageflags_fd)) { + ksft_print_msg("order %d also reported as order %d\n", + order, order - 1); + ok = false; + } + + /* A large folio must not pass as order 0 */ + if (order && is_range_backed_by_order(p, size, 0, + pagemap_fd, kpageflags_fd)) { + ksft_print_msg("order %d also reported as order 0\n", order); + ok = false; + } + + munmap(p, size); + thp_pop_settings(); + + ksft_test_result(ok, "order %d classified\n", order); +} + +int main(void) +{ + struct thp_settings settings; + unsigned long orders; + int order; + + ksft_print_header(); + + if (!thp_available()) + ksft_exit_skip("Transparent Hugepages not available\n"); + + pagemap_fd = open("/proc/self/pagemap", O_RDONLY); + if (pagemap_fd < 0) + ksft_exit_fail_perror("open(/proc/self/pagemap)"); + kpageflags_fd = open("/proc/kpageflags", O_RDONLY); + if (kpageflags_fd < 0) + ksft_exit_skip("open(/proc/kpageflags) requires root\n"); + + orders = thp_supported_orders(); + if (!orders) + ksft_exit_skip("No supported THP orders\n"); + + ksft_set_plan(__builtin_popcountl(orders) + 1); + + thp_save_settings(); + thp_read_settings(&settings); + /* Base of the settings stack; the bottom entry is never popped */ + thp_push_settings(&settings); + + check_order(0); + for (order = 1; order < NR_ORDERS; order++) { + if (!(orders & (1UL << order))) + continue; + check_order(order); + } + + ksft_finished(); +} diff --git a/tools/testing/selftests/mm/hmm-tests.c b/tools/testing/selftests/mm/hmm-tests.c index fa1a651963fd01..e5f273ca84c107 100644 --- a/tools/testing/selftests/mm/hmm-tests.c +++ b/tools/testing/selftests/mm/hmm-tests.c @@ -65,7 +65,6 @@ enum { #define HMM_PATH_MAX 64 #define NTIMES 10 -#define ALIGN(x, a) (((x) + (a - 1)) & (~((a) - 1))) /* Just the flags we need, copied from mm.h: */ #ifndef FOLL_WRITE diff --git a/tools/testing/selftests/mm/migration.c b/tools/testing/selftests/mm/migration.c index a35e2b57e05b2d..d1d0989ed2cada 100644 --- a/tools/testing/selftests/mm/migration.c +++ b/tools/testing/selftests/mm/migration.c @@ -20,7 +20,6 @@ #define TWOMEG (2<<20) #define RUNTIME (20) -#define ALIGN(x, a) (((x) + (a - 1)) & (~((a) - 1))) HUGETLB_SETUP_DEFAULT_PAGES(1) diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 0d8c93087d0529..6b80cf2eab149c 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -382,6 +382,8 @@ CATEGORY="pfnmap" run_test ./pfnmap # COW tests CATEGORY="cow" run_test ./cow +CATEGORY="thp" run_test ./folio_order_check + CATEGORY="thp" run_test ./khugepaged CATEGORY="thp" run_test ./khugepaged -s 2 diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index 0172003c16dcb9..ea48e6a7527e13 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -12,6 +12,8 @@ #include #define BIT_ULL(nr) (1ULL << (nr)) +#define ALIGN(x, a) (((x) + (a) - 1) & ~((a) - 1)) + #define PM_SOFT_DIRTY BIT_ULL(55) #define PM_MMAP_EXCLUSIVE BIT_ULL(56) #define PM_UFFD_WP BIT_ULL(57) From db031fbab9c063ce05007549110105814e956e88 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:40 +0100 Subject: [PATCH 0798/1012] selftests/mm: add khugepaged completion barrier helper A khugepaged test has to tell "not collapsed" from "not scanned yet", and nothing in the selftests can. wait_for_scan() in khugepaged.c comes closest: it polls full_scans until the counter has advanced by two, since the pass in progress may already have passed the test's mm. But it only returns in time if scan_sleep_millisecs happens to be short, and it is private to that one test. Add khugepaged_full_pass() to hugepage_settings, built on the same advance-by-two wait but driven through sysfs: a store to scan_sleep_millisecs wakes the daemon, so the barrier completes whatever the scan cadence. A store made while the daemon is scanning rather than sleeping is lost, so the helper keeps storing until the pass lands. One wake completes one pass only if pages_to_scan covers every mm on the list, so callers need it large. Settings pushes must not start passes of their own. A store to either sleep knob wakes the daemon, so thp_write_settings() now writes a khugepaged knob only when its value changes. Link: https://lore.kernel.org/20260919002451.496763-11-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/lib/mm/hugepage_settings.c | 60 +++++++++++++++++++++++++++----- tools/lib/mm/hugepage_settings.h | 2 ++ 2 files changed, 53 insertions(+), 9 deletions(-) diff --git a/tools/lib/mm/hugepage_settings.c b/tools/lib/mm/hugepage_settings.c index 656442c8d3954a..77918677a9cd02 100644 --- a/tools/lib/mm/hugepage_settings.c +++ b/tools/lib/mm/hugepage_settings.c @@ -217,6 +217,13 @@ void thp_read_settings(struct thp_settings *settings) } } +/* A store to either sleep knob wakes khugepaged, so write only on change */ +static void thp_update_num(const char *name, unsigned long num) +{ + if (thp_read_num(name) != num) + thp_write_num(name, num); +} + void thp_write_settings(struct thp_settings *settings) { struct khugepaged_settings *khugepaged = &settings->khugepaged; @@ -232,15 +239,15 @@ void thp_write_settings(struct thp_settings *settings) shmem_enabled_strings[settings->shmem_enabled]); thp_write_num("use_zero_page", settings->use_zero_page); - thp_write_num("khugepaged/defrag", khugepaged->defrag); - thp_write_num("khugepaged/alloc_sleep_millisecs", - khugepaged->alloc_sleep_millisecs); - thp_write_num("khugepaged/scan_sleep_millisecs", - khugepaged->scan_sleep_millisecs); - thp_write_num("khugepaged/max_ptes_none", khugepaged->max_ptes_none); - thp_write_num("khugepaged/max_ptes_swap", khugepaged->max_ptes_swap); - thp_write_num("khugepaged/max_ptes_shared", khugepaged->max_ptes_shared); - thp_write_num("khugepaged/pages_to_scan", khugepaged->pages_to_scan); + thp_update_num("khugepaged/defrag", khugepaged->defrag); + thp_update_num("khugepaged/alloc_sleep_millisecs", + khugepaged->alloc_sleep_millisecs); + thp_update_num("khugepaged/scan_sleep_millisecs", + khugepaged->scan_sleep_millisecs); + thp_update_num("khugepaged/max_ptes_none", khugepaged->max_ptes_none); + thp_update_num("khugepaged/max_ptes_swap", khugepaged->max_ptes_swap); + thp_update_num("khugepaged/max_ptes_shared", khugepaged->max_ptes_shared); + thp_update_num("khugepaged/pages_to_scan", khugepaged->pages_to_scan); if (dev_queue_read_ahead_path[0]) { int ret = write_num(dev_queue_read_ahead_path, @@ -271,6 +278,41 @@ void thp_write_settings(struct thp_settings *settings) } } +/* + * Wait for a full khugepaged scan pass that started after this call: the + * pass in progress may already have passed this mm, so full_scans has to + * advance twice. + * + * A store to scan_sleep_millisecs wakes the daemon, but one made while it + * is scanning rather than sleeping is lost, so keep storing until the pass + * lands. + * + * One wake is one pass only if pages_to_scan covers every mm on the list. + */ +bool khugepaged_full_pass(unsigned int timeout_s) +{ + unsigned long deadline_ms = timeout_s * 1000UL; + unsigned long elapsed_ms = 0, poll_ms = 10; + unsigned long sleep_ms; + int pass; + + sleep_ms = thp_read_num("khugepaged/scan_sleep_millisecs"); + for (pass = 0; pass < 2; pass++) { + unsigned long target = + thp_read_num("khugepaged/full_scans") + 1; + + while (thp_read_num("khugepaged/full_scans") < target) { + if (elapsed_ms >= deadline_ms) + return false; + thp_write_num("khugepaged/scan_sleep_millisecs", + sleep_ms); + usleep(poll_ms * 1000); + elapsed_ms += poll_ms; + } + } + return true; +} + struct thp_settings *thp_current_settings(void) { if (!settings_index) { diff --git a/tools/lib/mm/hugepage_settings.h b/tools/lib/mm/hugepage_settings.h index 94d9fc747f4979..8f4581099b7aba 100644 --- a/tools/lib/mm/hugepage_settings.h +++ b/tools/lib/mm/hugepage_settings.h @@ -83,6 +83,8 @@ static inline void thp_save_settings(void) hugepage_save_settings(/* thp = */ true, /* hugetlb = */ false); } +bool khugepaged_full_pass(unsigned int timeout_s); + void thp_set_read_ahead_path(char *path); unsigned long thp_supported_orders(void); unsigned long thp_shmem_supported_orders(void); From a50d2b63d490d38234da0c7cf6f1ca45ea8a1368 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:41 +0100 Subject: [PATCH 0799/1012] selftests/mm: add order-parameterized khugepaged collapse cases The mthp_khugepaged context runs the generic cases at a sub-PMD order, which answers how many folios of that order a range ends up with. It cannot say which order-sized window they landed in, so "the populated window collapsed" and "the empty window next to it collapsed instead" look alike. Add four cases that check each window on its own, with the folio-order helpers in vm_util: - collapse_order_single_window(): only the populated window collapses; - collapse_order_partial_window(): the default max_ptes_none lets a window with one present PTE collapse; - collapse_order_max_ptes_none(): with max_ptes_none=0 a full window collapses and one missing a page does not; - collapse_order_mixed_sources(): sources that are already large folios of a smaller order collapse to the target. Each case faults its region before MADV_HUGEPAGE with only the target order enabled, so the sources are order 0 and the result can only come from khugepaged. They wait for a full pass rather than for the result to appear: without a completed pass, "not collapsed" and "not scanned yet" are the same thing. Link: https://lore.kernel.org/20260919002451.496763-12-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Tested-by: Baolin Wang Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 212 ++++++++++++++++++++++++ 1 file changed, 212 insertions(+) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index e013eebc7136ed..51bda01446cd7e 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -30,6 +30,8 @@ static unsigned long page_size; static int hpage_pmd_nr; static int anon_order; static int collapse_order; +static int pagemap_fd = -1; +static int kpageflags_fd = -1; #define PID_SMAPS "/proc/self/smaps" #define TEST_FILE "collapse_test_file" @@ -1209,6 +1211,198 @@ static void madvise_retracted_page_tables(struct collapse_context *c, ksft_test_result_report(exit_status, "%s\n", __func__); } +/* Smallest order khugepaged will consider for mTHP collapse */ +#define MIN_MTHP_ORDER 2 + +/* Time budget for one khugepaged pass in the collapse_order_* cases */ +#define MTHP_PASS_TIMEOUT_S 30 + +static size_t mthp_window_size(void) +{ + return page_size << collapse_order; +} + +static void mthp_push_target_order(void) +{ + struct thp_settings settings = *thp_current_settings(); + int i; + + /* + * Only the target order, and only for madvise: the cases fault their + * region first, so the sources stay order 0 whatever -s asked for. + */ + settings.thp_enabled = THP_NEVER; + for (i = 0; i < NR_ORDERS; i++) + settings.hugepages[i].enabled = THP_NEVER; + settings.hugepages[collapse_order].enabled = THP_MADVISE; + thp_push_settings(&settings); +} + +static bool all_windows_at_order(void *p, size_t len) +{ + return is_range_backed_by_order(p, len, collapse_order, + pagemap_fd, kpageflags_fd); +} + +static bool any_window_at_order(void *p, size_t len) +{ + size_t window = mthp_window_size(); + char *addr = p; + + for (; len >= window; addr += window, len -= window) { + if (all_windows_at_order(addr, window)) + return true; + } + return false; +} + +static void collapse_order_single_window(struct collapse_context *c, + struct mem_ops *ops) +{ + size_t window = mthp_window_size(); + void *p; + + mthp_push_target_order(); + + p = ops->setup_area(1); + ops->fault(p, window, 2 * window); + if (any_window_at_order(p, hpage_pmd_size)) + ksft_exit_fail_msg("Unexpected large folio after fault\n"); + + if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); + ksft_print_msg("Collapse one fully populated window..."); + if (!khugepaged_full_pass(MTHP_PASS_TIMEOUT_S)) + fail("Timeout"); + else if (all_windows_at_order(p + window, window) && + !any_window_at_order(p, window) && + !any_window_at_order(p + 2 * window, + hpage_pmd_size - 2 * window)) + success("OK"); + else + fail("Fail"); + + validate_memory(p, window, 2 * window); + ops->cleanup_area(p, hpage_pmd_size); + thp_pop_settings(); + ksft_test_result_report(exit_status, "%s\n", __func__); +} + +static void collapse_order_partial_window(struct collapse_context *c, + struct mem_ops *ops) +{ + void *p; + + mthp_push_target_order(); + + p = ops->setup_area(1); + ops->fault(p, 0, page_size); + if (any_window_at_order(p, hpage_pmd_size)) + ksft_exit_fail_msg("Unexpected large folio after fault\n"); + + if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); + ksft_print_msg("Collapse window with single PTE entry present..."); + if (!khugepaged_full_pass(MTHP_PASS_TIMEOUT_S)) + fail("Timeout"); + else if (all_windows_at_order(p, mthp_window_size())) + success("OK"); + else + fail("Fail"); + + validate_memory(p, 0, page_size); + ops->cleanup_area(p, hpage_pmd_size); + thp_pop_settings(); + ksft_test_result_report(exit_status, "%s\n", __func__); +} + +static void collapse_order_max_ptes_none(struct collapse_context *c, + struct mem_ops *ops) +{ + struct thp_settings settings; + size_t window = mthp_window_size(); + void *p; + + mthp_push_target_order(); + settings = *thp_current_settings(); + settings.khugepaged.max_ptes_none = 0; + thp_push_settings(&settings); + + p = ops->setup_area(1); + ops->fault(p, 0, 2 * window - page_size); + if (any_window_at_order(p, hpage_pmd_size)) + ksft_exit_fail_msg("Unexpected large folio after fault\n"); + + if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); + ksft_print_msg("Collapse full window, not the one missing a page..."); + if (!khugepaged_full_pass(MTHP_PASS_TIMEOUT_S)) + fail("Timeout"); + else if (all_windows_at_order(p, window) && + !any_window_at_order(p + window, window)) + success("OK"); + else + fail("Fail"); + + validate_memory(p, 0, 2 * window - page_size); + ops->cleanup_area(p, hpage_pmd_size); + thp_pop_settings(); + thp_pop_settings(); + ksft_test_result_report(exit_status, "%s\n", __func__); +} + +static void collapse_order_mixed_sources(struct collapse_context *c, + struct mem_ops *ops) +{ + struct thp_settings settings; + void *p; + + if (collapse_order <= MIN_MTHP_ORDER) { + ksft_test_result_skip("%s: no source order below target\n", + __func__); + return; + } + + mthp_push_target_order(); + + settings = *thp_current_settings(); + settings.hugepages[MIN_MTHP_ORDER].enabled = THP_ALWAYS; + thp_push_settings(&settings); + p = ops->setup_area(1); + ops->fault(p, 0, hpage_pmd_size); + thp_pop_settings(); + + /* + * The allocator can fall back to smaller folios under fragmentation; + * having nothing to collapse from is not a failure. + */ + if (!is_range_backed_by_order(p, hpage_pmd_size, MIN_MTHP_ORDER, + pagemap_fd, kpageflags_fd)) { + ksft_print_msg("No order-%d sources to collapse...", + MIN_MTHP_ORDER); + skip("Skip"); + ops->cleanup_area(p, hpage_pmd_size); + thp_pop_settings(); + ksft_test_result_report(exit_status, "%s\n", __func__); + return; + } + + if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); + ksft_print_msg("Collapse region backed by smaller large folios..."); + if (!khugepaged_full_pass(MTHP_PASS_TIMEOUT_S)) + fail("Timeout"); + else if (all_windows_at_order(p, hpage_pmd_size)) + success("OK"); + else + fail("Fail"); + + validate_memory(p, 0, hpage_pmd_size); + ops->cleanup_area(p, hpage_pmd_size); + thp_pop_settings(); + ksft_test_result_report(exit_status, "%s\n", __func__); +} + static void usage(void) { fprintf(stderr, "\nUsage: ./khugepaged [OPTIONS] [dir]\n\n"); @@ -1377,6 +1571,20 @@ int main(int argc, char **argv) parse_test_type(argc, argv); + if (mthp_khugepaged_context && + !(thp_supported_orders() & (1UL << collapse_order))) + ksft_exit_skip("Order %d is not a supported anon THP order\n", + collapse_order); + + if (mthp_khugepaged_context) { + pagemap_fd = open("/proc/self/pagemap", O_RDONLY); + if (pagemap_fd < 0) + ksft_exit_fail_perror("open(/proc/self/pagemap)"); + kpageflags_fd = open("/proc/kpageflags", O_RDONLY); + if (kpageflags_fd < 0) + ksft_exit_fail_perror("open(/proc/kpageflags)"); + } + setbuf(stdout, NULL); /* @@ -1427,6 +1635,10 @@ int main(int argc, char **argv) TEST(collapse_empty, madvise_context, anon_ops); TEST(collapse_single_mthp, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_single_window, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_partial_window, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_max_ptes_none, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_mixed_sources, mthp_khugepaged_context, anon_ops); TEST(collapse_single_pte_entry, khugepaged_context, anon_ops); TEST(collapse_single_pte_entry, khugepaged_context, read_only_file_ops); From d34add903ff9aef1faaf08c18a00f308f550d112 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:42 +0100 Subject: [PATCH 0800/1012] selftests/mm: parameterize the mixed-source collapse case by source order collapse_order_mixed_sources() faults its region as order-2 folios and collapses them to the -c target. Order 2 is below the contpte size on every arm64 page size, so nothing in this suite collapses a contpte-mapped source on purpose. Let -s name the source order alongside -c. The case then faults at that order, keeping order 2 when -s is absent, and the source order has to be a supported mTHP order below the target. The other mTHP cases are unaffected: mthp_push_target_order() enables only the target order. A -c at or below -s is refused before any case runs: the sources would already be the size being asked for. Without the check the generic cases fail on that one by one instead of saying why. "-s 5 -c 7" on arm64/64K then collapses contpte-mapped sources into a larger mTHP. Link: https://lore.kernel.org/20260919002451.496763-13-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Acked-by: Lorenzo Stoakes (ARM) Tested-by: Muhammad Usama Anjum Reviewed-by: Baolin Wang Tested-by: Baolin Wang Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 20 +++++++++++++------- 1 file changed, 13 insertions(+), 7 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 51bda01446cd7e..9d1c47bd501349 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1354,11 +1354,13 @@ static void collapse_order_max_ptes_none(struct collapse_context *c, static void collapse_order_mixed_sources(struct collapse_context *c, struct mem_ops *ops) { + int source_order = anon_order ? anon_order : MIN_MTHP_ORDER; struct thp_settings settings; void *p; - if (collapse_order <= MIN_MTHP_ORDER) { - ksft_test_result_skip("%s: no source order below target\n", + if (source_order >= collapse_order || + !(thp_supported_orders() & (1UL << source_order))) { + ksft_test_result_skip("%s: no supported source order below target\n", __func__); return; } @@ -1366,7 +1368,7 @@ static void collapse_order_mixed_sources(struct collapse_context *c, mthp_push_target_order(); settings = *thp_current_settings(); - settings.hugepages[MIN_MTHP_ORDER].enabled = THP_ALWAYS; + settings.hugepages[source_order].enabled = THP_ALWAYS; thp_push_settings(&settings); p = ops->setup_area(1); ops->fault(p, 0, hpage_pmd_size); @@ -1376,10 +1378,9 @@ static void collapse_order_mixed_sources(struct collapse_context *c, * The allocator can fall back to smaller folios under fragmentation; * having nothing to collapse from is not a failure. */ - if (!is_range_backed_by_order(p, hpage_pmd_size, MIN_MTHP_ORDER, + if (!is_range_backed_by_order(p, hpage_pmd_size, source_order, pagemap_fd, kpageflags_fd)) { - ksft_print_msg("No order-%d sources to collapse...", - MIN_MTHP_ORDER); + ksft_print_msg("No order-%d sources to collapse...", source_order); skip("Skip"); ops->cleanup_area(p, hpage_pmd_size); thp_pop_settings(); @@ -1389,7 +1390,8 @@ static void collapse_order_mixed_sources(struct collapse_context *c, if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE)) ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); - ksft_print_msg("Collapse region backed by smaller large folios..."); + ksft_print_msg("Collapse region backed by order-%d sources...", + source_order); if (!khugepaged_full_pass(MTHP_PASS_TIMEOUT_S)) fail("Timeout"); else if (all_windows_at_order(p, hpage_pmd_size)) @@ -1420,6 +1422,7 @@ static void usage(void) fprintf(stderr, "\t\t-s: mTHP size, expressed as page order.\n"); fprintf(stderr, "\t\t Defaults to 0. Use this size for anon or shmem allocations.\n"); fprintf(stderr, "\t\t-c: collapse order for mTHP collapse, expressed as page order.\n"); + fprintf(stderr, "\t\t -s, if set, is the source order for the mixed-source case.\n"); exit(1); } @@ -1575,6 +1578,9 @@ int main(int argc, char **argv) !(thp_supported_orders() & (1UL << collapse_order))) ksft_exit_skip("Order %d is not a supported anon THP order\n", collapse_order); + if (mthp_khugepaged_context && collapse_order <= anon_order) + ksft_exit_skip("-c %d needs a source order below it, -s says %d\n", + collapse_order, anon_order); if (mthp_khugepaged_context) { pagemap_fd = open("/proc/self/pagemap", O_RDONLY); From 16db9fd26d1a8eb349ca29c76510c59ea074f489 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:43 +0100 Subject: [PATCH 0801/1012] selftests/mm: cover a shared-source collapse write race collapse_fork() checks that a fork-shared range collapses in the child while the parent keeps its own pages, but the parent sits still while that happens. Nothing checks that CoW isolation survives a collapse racing with writes to the shared source. Add a case where the parent writes to the shared range throughout the child's collapse. CoW has to keep the two apart: the child must see the content from before the fork, and the parent only its own writes. The parent unshares one page every 10ms, starting only once the child says it is about to collapse. Writing the range in a burst would break CoW on all of it before the collapse begins, leaving the child to collapse pages that are already exclusive to it. Preparation for changing how collapse handles fork-shared sources. Link: https://lore.kernel.org/20260919002451.496763-14-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 106 ++++++++++++++++++++++++ 1 file changed, 106 insertions(+) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 9d1c47bd501349..21ae258bd56eb7 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1166,6 +1166,109 @@ static void collapse_max_ptes_shared(struct collapse_context *c, struct mem_ops ksft_test_result_report(exit_status, "%s\n", __func__); } +/* + * The parent writes to the fork-shared range throughout the child's + * collapse. CoW must keep the two apart: the child sees the pre-fork + * content, the parent only its own writes. + */ +static void collapse_fork_cow_race(struct collapse_context *c, struct mem_ops *ops) +{ + const int stride = page_size / sizeof(int); + int wstatus, child_status, i, n; + unsigned long shared; + volatile int *ip; + pid_t child; + int sync[2]; + char go = 1; + void *p; + + /* At a page per 10 ms, 64 pages spread the writes across the collapse */ + n = 64; + shared = n * page_size; + + p = ops->setup_area(1); + /* Shared prefix, with the pre-fork pattern */ + ops->fault(p, 0, shared); + if (pipe(sync)) + ksft_exit_fail_perror("pipe()"); + + /* A volatile pointer so the stores are not merged or dropped */ + ip = p; + + ksft_print_msg("Fork, collapse in the child while the parent rewrites..."); + child = fork(); + if (!child) { + int collapse_status; + + close(sync[0]); + /* Private remainder */ + ops->fault(p, shared, hpage_pmd_size); + /* Start the parent unsharing, and give it a head start */ + if (write(sync[1], &go, 1) != 1) + _exit(KSFT_FAIL); + usleep(5000); + c->collapse("Collapse a range the parent is writing to", + p, 1, ops, true); + collapse_status = exit_status; + for (i = 0; i < n; i++) + if (ip[i * stride] != i + 0xdead0000) + break; + if (i == n) + success("OK"); + else + fail("Fail: child content"); + /* The content check must not bury a failed collapse */ + if (exit_status != KSFT_FAIL) + exit_status = collapse_status; + ops->cleanup_area(p, hpage_pmd_size); + _exit(exit_status); + } + + close(sync[1]); + if (read(sync[0], &go, 1) != 1) + ksft_exit_fail_msg("child never reached the collapse\n"); + close(sync[0]); + + /* + * Unshare one page at a time: a burst would break CoW on the whole + * range before the collapse starts, leaving nothing shared to collapse. + */ + i = 0; + for (;;) { + pid_t ret; + + if (i < n) + ip[i * stride] = i + 0xbeef0000; + i++; + usleep(10 * 1000); + ret = waitpid(child, &wstatus, WNOHANG); + if (ret == child) + break; + if (ret < 0) + ksft_exit_fail_perror("waitpid()"); + } + + /* Finish whatever the paced sweep did not reach */ + for (; i < n; i++) + ip[i * stride] = i + 0xbeef0000; + /* A child that died reading the racing pages is a failure, not a zero */ + child_status = WIFEXITED(wstatus) ? WEXITSTATUS(wstatus) : KSFT_FAIL; + + ksft_print_msg("Check the parent sees only its own writes..."); + for (i = 0; i < n; i++) + if (ip[i * stride] != i + 0xbeef0000) + break; + if (i == n) + success("OK"); + else + fail("Fail: parent content"); + ops->cleanup_area(p, hpage_pmd_size); + /* The parent's check must not bury the child's verdict */ + if (exit_status != KSFT_FAIL) + exit_status = child_status; + ksft_test_result_report(exit_status, "%s\n", __func__); +} + static void madvise_collapse_existing_thps(struct collapse_context *c, struct mem_ops *ops) { @@ -1700,6 +1803,9 @@ int main(int argc, char **argv) TEST(collapse_max_ptes_shared, khugepaged_context, anon_ops); TEST(collapse_max_ptes_shared, madvise_context, anon_ops); + TEST(collapse_fork_cow_race, khugepaged_context, anon_ops); + TEST(collapse_fork_cow_race, madvise_context, anon_ops); + TEST(madvise_collapse_existing_thps, madvise_context, anon_ops); TEST(madvise_collapse_existing_thps, madvise_context, read_only_file_ops); TEST(madvise_collapse_existing_thps, madvise_context, read_write_file_read_ops); From cdfe056700dc1006fbc9541cfbd89eb90697d68a Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:44 +0100 Subject: [PATCH 0802/1012] selftests/mm: run every supported collapse order by default The mTHP collapse cases only run when the caller names both the context and an order, so a plain ./khugepaged covers the PMD contexts on anon and nothing else. run_vmtests.sh pinned order 4 and covered no other. Run the mTHP cases once per supported anon THP order below the PMD when -c is absent, and pull that context into both the no-argument invocation and "all". Also: - -c still pins one order, and now says what is wrong instead of printing the usage text. - Both orders end up as array indices and shift counts, so -s and -c are range-checked before they get there. - The mTHP context has only anon cases, so a run that names a different mem_type -- "all:shmem", say -- drops it again rather than refusing to start. Naming both explicitly still refuses. - A case carries the order it was registered at, so a result names it: # Run test: collapse_single_mthp (mthp_khugepaged:anon, order 6) On x86-64 with 4K pages that is orders 2 through 8, and ./khugepaged goes from 29 results in 17 seconds to 77 in 29, so run_vmtests.sh can drop its pinned order-4 line. Link: https://lore.kernel.org/20260919002451.496763-15-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Tested-by: Muhammad Usama Anjum Tested-by: Baolin Wang Assisted-by: LLM Cc: Alexander Gordeev Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged.c | 101 +++++++++++++++++----- tools/testing/selftests/mm/run_vmtests.sh | 2 - 2 files changed, 78 insertions(+), 25 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 21ae258bd56eb7..a2ac3b3ca5def3 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -30,6 +30,9 @@ static unsigned long page_size; static int hpage_pmd_nr; static int anon_order; static int collapse_order; +static bool collapse_order_set; +static int collapse_orders[NR_ORDERS]; +static int nr_collapse_orders; static int pagemap_fd = -1; static int kpageflags_fd = -1; @@ -1525,12 +1528,14 @@ static void usage(void) fprintf(stderr, "\t\t-s: mTHP size, expressed as page order.\n"); fprintf(stderr, "\t\t Defaults to 0. Use this size for anon or shmem allocations.\n"); fprintf(stderr, "\t\t-c: collapse order for mTHP collapse, expressed as page order.\n"); + fprintf(stderr, "\t\t Defaults to every supported order below the PMD.\n"); fprintf(stderr, "\t\t -s, if set, is the source order for the mixed-source case.\n"); exit(1); } static void parse_test_type(int argc, char **argv) { + bool mthp_context_implied = false; int opt; char *buf; const char *token; @@ -1542,6 +1547,7 @@ static void parse_test_type(int argc, char **argv) break; case 'c': collapse_order = atoi(optarg); + collapse_order_set = true; break; case 'h': default: @@ -1549,12 +1555,25 @@ static void parse_test_type(int argc, char **argv) } } + /* + * Both orders end up as array indices and shift counts, so neither + * can be negative, and a zero collapse order asks for base pages. + */ + if (anon_order < 0 || anon_order > hpage_pmd_order) + ksft_exit_fail_msg("-s takes an order in 0..%d, not %d\n", + hpage_pmd_order, anon_order); + if (collapse_order_set && + (collapse_order <= 0 || collapse_order >= hpage_pmd_order)) + ksft_exit_fail_msg("-c takes an order in 1..%d, not %d\n", + hpage_pmd_order - 1, collapse_order); + argv += optind; argc -= optind; if (argc == 0) { - /* Backwards compatibility */ + /* No arguments: anon under every context */ khugepaged_context = &__khugepaged_context; + mthp_khugepaged_context = &__mthp_khugepaged_context; madvise_context = &__madvise_context; anon_ops = &__anon_ops; return; @@ -1565,13 +1584,14 @@ static void parse_test_type(int argc, char **argv) if (!strcmp(token, "all")) { khugepaged_context = &__khugepaged_context; + mthp_khugepaged_context = &__mthp_khugepaged_context; madvise_context = &__madvise_context; + /* The mTHP context has only anon cases; let other mem_types drop it */ + mthp_context_implied = true; } else if (!strcmp(token, "khugepaged")) { khugepaged_context = &__khugepaged_context; } else if (!strcmp(token, "mthp_khugepaged")) { mthp_khugepaged_context = &__mthp_khugepaged_context; - if (collapse_order <= 0 || collapse_order >= hpage_pmd_order) - usage(); } else if (!strcmp(token, "madvise")) { madvise_context = &__madvise_context; } else { @@ -1587,20 +1607,20 @@ static void parse_test_type(int argc, char **argv) read_write_file_write_ops = &__read_write_file_write_ops; anon_ops = &__anon_ops; shmem_ops = &__shmem_ops; - if (mthp_khugepaged_context) - usage(); } else if (!strcmp(buf, "anon")) { anon_ops = &__anon_ops; } else if (!strcmp(buf, "file")) { read_only_file_ops = &__read_only_file_ops; read_write_file_read_ops = &__read_write_file_read_ops; read_write_file_write_ops = &__read_write_file_write_ops; - if (mthp_khugepaged_context) + if (mthp_khugepaged_context && !mthp_context_implied) usage(); + mthp_khugepaged_context = NULL; } else if (!strcmp(buf, "shmem")) { shmem_ops = &__shmem_ops; - if (mthp_khugepaged_context) + if (mthp_khugepaged_context && !mthp_context_implied) usage(); + mthp_khugepaged_context = NULL; } else { usage(); } @@ -1622,6 +1642,7 @@ struct test_case { struct mem_ops *ops; const char *desc; test_fn fn; + int order; /* mTHP contexts: the collapse order */ }; #define MAX_TEST_CASES 256 @@ -1637,6 +1658,7 @@ static int nr_test_cases; .ops = o, \ .desc = #t, \ .fn = t, \ + .order = collapse_order, \ }; \ } \ } while (0) @@ -1677,13 +1699,35 @@ int main(int argc, char **argv) parse_test_type(argc, argv); - if (mthp_khugepaged_context && - !(thp_supported_orders() & (1UL << collapse_order))) - ksft_exit_skip("Order %d is not a supported anon THP order\n", - collapse_order); - if (mthp_khugepaged_context && collapse_order <= anon_order) - ksft_exit_skip("-c %d needs a source order below it, -s says %d\n", - collapse_order, anon_order); + if (mthp_khugepaged_context) { + unsigned long orders = thp_supported_orders(); + + if (collapse_order_set) { + if (!(orders & (1UL << collapse_order))) + ksft_exit_skip("Order %d is not a supported anon THP order\n", + collapse_order); + if (collapse_order <= anon_order) + ksft_exit_skip("-c %d needs a source order below it, -s says %d\n", + collapse_order, anon_order); + collapse_orders[nr_collapse_orders++] = collapse_order; + } else { + /* + * Every supported order above the source: -s makes the + * fault path hand out folios of that order, so a target + * at or below it has nothing to collapse. + */ + int first = anon_order + 1; + + if (first < MIN_MTHP_ORDER) + first = MIN_MTHP_ORDER; + for (int i = first; i < hpage_pmd_order; i++) { + if (orders & (1UL << i)) + collapse_orders[nr_collapse_orders++] = i; + } + if (!nr_collapse_orders) + ksft_print_msg("mTHP cases skipped: no order above the source\n"); + } + } if (mthp_khugepaged_context) { pagemap_fd = open("/proc/self/pagemap", O_RDONLY); @@ -1732,7 +1776,17 @@ int main(int argc, char **argv) TEST(collapse_full, khugepaged_context, read_write_file_read_ops); TEST(collapse_full, khugepaged_context, read_write_file_write_ops); TEST(collapse_full, khugepaged_context, shmem_ops); - TEST(collapse_full, mthp_khugepaged_context, anon_ops); + for (int i = 0; i < nr_collapse_orders; i++) { + collapse_order = collapse_orders[i]; + TEST(collapse_full, mthp_khugepaged_context, anon_ops); + TEST(collapse_empty, mthp_khugepaged_context, anon_ops); + TEST(collapse_single_mthp, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_single_window, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_partial_window, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_max_ptes_none, mthp_khugepaged_context, anon_ops); + TEST(collapse_order_mixed_sources, mthp_khugepaged_context, anon_ops); + } + TEST(collapse_full, madvise_context, anon_ops); TEST(collapse_full, madvise_context, read_only_file_ops); TEST(collapse_full, madvise_context, read_write_file_read_ops); @@ -1740,15 +1794,8 @@ int main(int argc, char **argv) TEST(collapse_full, madvise_context, shmem_ops); TEST(collapse_empty, khugepaged_context, anon_ops); - TEST(collapse_empty, mthp_khugepaged_context, anon_ops); TEST(collapse_empty, madvise_context, anon_ops); - TEST(collapse_single_mthp, mthp_khugepaged_context, anon_ops); - TEST(collapse_order_single_window, mthp_khugepaged_context, anon_ops); - TEST(collapse_order_partial_window, mthp_khugepaged_context, anon_ops); - TEST(collapse_order_max_ptes_none, mthp_khugepaged_context, anon_ops); - TEST(collapse_order_mixed_sources, mthp_khugepaged_context, anon_ops); - TEST(collapse_single_pte_entry, khugepaged_context, anon_ops); TEST(collapse_single_pte_entry, khugepaged_context, read_only_file_ops); TEST(collapse_single_pte_entry, khugepaged_context, read_write_file_read_ops); @@ -1821,7 +1868,15 @@ int main(int argc, char **argv) for (int i = 0; i < nr_test_cases; i++) { struct test_case *t = &test_cases[i]; - ksft_print_msg("\n# Run test: %s (%s:%s)\n", t->desc, t->ctx->name, t->ops->name); + if (t->ctx == &__mthp_khugepaged_context) { + collapse_order = t->order; + ksft_print_msg("\n# Run test: %s (%s:%s, order %d)\n", + t->desc, t->ctx->name, t->ops->name, + t->order); + } else { + ksft_print_msg("\n# Run test: %s (%s:%s)\n", t->desc, + t->ctx->name, t->ops->name); + } t->fn(t->ctx, t->ops); } diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 6b80cf2eab149c..f1138412788574 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -392,8 +392,6 @@ CATEGORY="thp" run_test ./khugepaged all:shmem CATEGORY="thp" run_test ./khugepaged -s 4 all:shmem -CATEGORY="thp" run_test ./khugepaged -c 4 mthp_khugepaged:anon - # Try to create XFS if not provided if [ -z "${SPLIT_HUGE_PAGE_TEST_XFS_PATH}" ]; then if test_selected "thp"; then From fe56f04dd267cd94b3b02dde60411400404b2a60 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:45 +0100 Subject: [PATCH 0803/1012] selftests/mm: check that one khugepaged pass collapses one window khugepaged_full_pass() drives the daemon through sysfs: a store to scan_sleep_millisecs wakes it, and full_scans advancing by two marks one pass that started after setup. Every mTHP collapse result in the suite rests on that pair, and nothing checks it. Add khugepaged_sync_check. Each step: - prepare one aligned window - record its source PFNs from pagemap - run one khugepaged_full_pass() barrier - require the window came out collapsed, with exactly one collapse attempt attributed to it The anon events carry no virtual address, so an attempt is matched by the source folio PFN and order that the mm_collapse_huge_page_isolate tracepoint reports. Reading the trace buffer takes four small helpers in vm_util: open an event subsystem's enable file, flip it, clear the buffer, and open it for reading. scan_sleep_millisecs is set to a minute, so a step that took a sleep instead of a wake would blow the budget. Passes 5/5 on x86-64 4K and arm64 64K. Link: https://lore.kernel.org/20260919002451.496763-16-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/Makefile | 1 + .../selftests/mm/khugepaged_sync_check.c | 179 ++++++++++++++++++ tools/testing/selftests/mm/run_vmtests.sh | 2 + tools/testing/selftests/mm/vm_util.c | 38 ++++ tools/testing/selftests/mm/vm_util.h | 4 + 5 files changed, 224 insertions(+) create mode 100644 tools/testing/selftests/mm/khugepaged_sync_check.c diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index cb32cf5d867e25..2ed88132491164 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -106,6 +106,7 @@ TEST_GEN_FILES += merge TEST_GEN_FILES += rmap TEST_GEN_FILES += folio_split_race_test TEST_GEN_FILES += folio_order_check +TEST_GEN_FILES += khugepaged_sync_check TEST_GEN_FILES += soft-dirty ifeq ($(ARCH),x86_64) diff --git a/tools/testing/selftests/mm/khugepaged_sync_check.c b/tools/testing/selftests/mm/khugepaged_sync_check.c new file mode 100644 index 00000000000000..28a9b1ff5d4474 --- /dev/null +++ b/tools/testing/selftests/mm/khugepaged_sync_check.c @@ -0,0 +1,179 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Check that khugepaged_full_pass() drives khugepaged in step: one barrier + * over one prepared window must collapse it with exactly one collapse + * attempt attributed to its source pages, step after step. + * + * scan_sleep_millisecs is a minute so that a step which slept instead of + * being woken blows the budget. + */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include + +#include "kselftest.h" +#include "vm_util.h" +#include + +#define BASE_ADDR ((void *)(1UL << 30)) +/* Smallest order khugepaged considers */ +#define TARGET_ORDER 2 +#define NR_ITERATIONS 5 +#define PASS_TIMEOUT_S 30 + +static int pagemap_fd; +static int kpageflags_fd; +static int trace_events_fd = -1; +static unsigned long hpage_pmd_size; + +/* + * The events are system-wide: switch them off however the test ends, + * including from inside a helper that gives up. + */ +static void trace_events_off(void) +{ + if (trace_events_fd >= 0) + tracing_events_enable(trace_events_fd, false); +} + +/* Count the isolate events whose scan_pfn is one of the window's source PFNs */ +static int count_attributed(unsigned long *pfns, int nr_pfns, + unsigned int order) +{ + char line[1024]; + int count = 0; + FILE *fp; + + fp = tracing_open_trace(); + if (!fp) + ksft_exit_fail_msg("Cannot open trace buffer\n"); + + while (fgets(line, sizeof(line), fp)) { + unsigned long val; + unsigned int ord; + char *s, *o; + int i; + + s = strstr(line, "mm_collapse_huge_page_isolate:"); + if (!s) + continue; + if (sscanf(s, "mm_collapse_huge_page_isolate: scan_pfn=0x%lx", + &val) != 1) + continue; + o = strstr(s, "order="); + if (!o || sscanf(o, "order=%u", &ord) != 1 || ord != order) + continue; + for (i = 0; i < nr_pfns; i++) { + if (val == pfns[i]) { + count++; + break; + } + } + } + fclose(fp); + return count; +} + +static void one_step(int iteration) +{ + const size_t window = getpagesize() << TARGET_ORDER; + const int nr_pages = 1 << TARGET_ORDER; + unsigned long pfns[1 << TARGET_ORDER]; + bool collapsed, passed; + int attributed; + char *p; + int i; + + p = mmap(BASE_ADDR, hpage_pmd_size, PROT_READ | PROT_WRITE, + MAP_ANONYMOUS | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0); + if (p != BASE_ADDR) + ksft_exit_fail_perror("mmap() window"); + + for (i = 0; i < nr_pages; i++) { + p[i * getpagesize()] = i + 1; + pfns[i] = pagemap_get_pfn(pagemap_fd, p + i * getpagesize()); + if (pfns[i] == -1UL) + ksft_exit_fail_msg("Source page not present\n"); + } + + /* Clear before enabling so the buffer holds only this step's events */ + if (tracing_clear_trace()) + ksft_exit_fail_msg("Cannot clear the trace buffer\n"); + if (tracing_events_enable(trace_events_fd, true)) + ksft_exit_fail_msg("Cannot enable huge_memory events\n"); + + if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); + passed = khugepaged_full_pass(PASS_TIMEOUT_S); + + /* Off before anything that can give up: the events are system-wide */ + if (tracing_events_enable(trace_events_fd, false)) + ksft_exit_fail_msg("Cannot disable huge_memory events\n"); + if (!passed) + ksft_exit_fail_msg("khugepaged did not complete a full pass\n"); + + collapsed = is_range_backed_by_order(p, window, TARGET_ORDER, + pagemap_fd, kpageflags_fd); + attributed = count_attributed(pfns, nr_pages, TARGET_ORDER); + + ksft_test_result(collapsed && attributed == 1, + "step %d: window collapsed, %d attributed result(s)\n", + iteration, attributed); + + munmap(p, hpage_pmd_size); +} + +int main(void) +{ + struct thp_settings settings; + int i; + + ksft_print_header(); + + if (!thp_available()) + ksft_exit_skip("Transparent Hugepages not available\n"); + if (!(thp_supported_orders() & (1UL << TARGET_ORDER))) + ksft_exit_skip("Order %d is not a supported anon THP order\n", + TARGET_ORDER); + + hpage_pmd_size = read_pmd_pagesize(); + if (!hpage_pmd_size) + ksft_exit_fail_msg("Reading PMD pagesize failed\n"); + pagemap_fd = open("/proc/self/pagemap", O_RDONLY); + if (pagemap_fd < 0) + ksft_exit_fail_perror("open(/proc/self/pagemap)"); + kpageflags_fd = open("/proc/kpageflags", O_RDONLY); + if (kpageflags_fd < 0) + ksft_exit_skip("open(/proc/kpageflags) requires root\n"); + trace_events_fd = tracing_events_open("huge_memory"); + if (trace_events_fd < 0) + ksft_exit_skip("huge_memory events require tracefs and root\n"); + atexit(trace_events_off); + + ksft_set_plan(NR_ITERATIONS); + + thp_save_settings(); + thp_read_settings(&settings); + settings.thp_enabled = THP_MADVISE; + settings.thp_defrag = THP_DEFRAG_ALWAYS; + settings.khugepaged.defrag = 1; + settings.khugepaged.scan_sleep_millisecs = 60 * 1000; + settings.khugepaged.alloc_sleep_millisecs = 60 * 1000; + settings.khugepaged.max_ptes_none = (hpage_pmd_size / getpagesize()) - 1; + /* One wake must complete one full pass; see khugepaged_full_pass() */ + settings.khugepaged.pages_to_scan = 1UL << 24; + for (i = 0; i < NR_ORDERS; i++) + settings.hugepages[i].enabled = THP_NEVER; + settings.hugepages[TARGET_ORDER].enabled = THP_INHERIT; + /* Base of the settings stack; the bottom entry is never popped */ + thp_push_settings(&settings); + + for (i = 0; i < NR_ITERATIONS; i++) + one_step(i); + + ksft_finished(); +} diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index f1138412788574..3e8c76cf4948b0 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -384,6 +384,8 @@ CATEGORY="cow" run_test ./cow CATEGORY="thp" run_test ./folio_order_check +CATEGORY="thp" run_test ./khugepaged_sync_check + CATEGORY="thp" run_test ./khugepaged CATEGORY="thp" run_test ./khugepaged -s 2 diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index 1f88330fd0274a..a0ab78ceedb130 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -598,6 +598,44 @@ bool is_range_backed_by_order(char *start, size_t len, int order, return true; } +#define TRACEFS_ROOT "/sys/kernel/tracing" + +/* + * Returns -1 without tracefs or the subsystem. The events are system-wide: + * whoever switches them on has to switch them off again, on every exit path. + */ +int tracing_events_open(const char *subsys) +{ + char path[256]; + + snprintf(path, sizeof(path), TRACEFS_ROOT "/events/%s/enable", + subsys); + return open(path, O_WRONLY); +} + +int tracing_events_enable(int fd, bool enable) +{ + if (pwrite(fd, enable ? "1" : "0", 1, 0) != 1) + return -1; + return 0; +} + +/* Drop what the trace buffer holds so far */ +int tracing_clear_trace(void) +{ + int fd = open(TRACEFS_ROOT "/trace", O_WRONLY | O_TRUNC); + + if (fd < 0) + return -1; + close(fd); + return 0; +} + +FILE *tracing_open_trace(void) +{ + return fopen(TRACEFS_ROOT "/trace", "r"); +} + /* If `ioctls' non-NULL, the allowed ioctls will be returned into the var */ int uffd_register_with_ioctls(int uffd, void *addr, uint64_t len, bool miss, bool wp, bool minor, uint64_t *ioctls) diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index ea48e6a7527e13..072a6c756c5170 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -121,6 +121,10 @@ int close_procmap(struct procmap_fd *procmap); int write_sysfs(const char *file_path, unsigned long val); int read_sysfs(const char *file_path, unsigned long *val); bool softdirty_supported(void); +int tracing_events_open(const char *subsys); +int tracing_events_enable(int fd, bool enable); +int tracing_clear_trace(void); +FILE *tracing_open_trace(void); static inline int open_self_procmap(struct procmap_fd *procmap_out) { From becb11ed566223a40328260e35b87394cb4adad4 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:46 +0100 Subject: [PATCH 0804/1012] selftests/mm: add khugepaged race harness Collapse serialises against faults, GUP, fork, mremap and zapping through a protocol of locks, TLB flushes and refcount checks. No khugepaged selftest exercises any of it under contention. Add khugepaged_race. Six racing threads work the same address space: - two faulters - an MADV_DONTNEED thread - a transient FOLL_PIN thread (gup_test) - a forker - an mremap thread One of three drivers collapses under them: stepped khugepaged, one full pass at a time via khugepaged_full_pass(), so each step covers a known extent; free khugepaged left to run (scan_sleep_millisecs=0), for soak; madvise an MADV_COLLAPSE and MADV_DONTNEED loop. Every mode runs in turn unless -m names one, five seconds each. Every supported anon THP order is set to inherit and max_ptes_none is 0, so a window collapses only once fully populated and the racing MADV_DONTNEED steers selection across orders. The rule is that a racing page reads as its pattern or as zero, never anything else. The faulters and fork children check it throughout, and a final sweep checks it again. The other half of the check is the kernel's own assertions, so read dmesg too. The pin thread goes through gup_test, so the harness skips without CONFIG_GUP_TEST or root. The default playground is three shared PMD-sized areas plus the mremap thread's, over two gigabytes at a 512M PMD; -a shrinks it. Link: https://lore.kernel.org/20260919002451.496763-17-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/Makefile | 1 + tools/testing/selftests/mm/khugepaged_race.c | 415 +++++++++++++++++++ tools/testing/selftests/mm/run_vmtests.sh | 2 + 3 files changed, 418 insertions(+) create mode 100644 tools/testing/selftests/mm/khugepaged_race.c diff --git a/tools/testing/selftests/mm/Makefile b/tools/testing/selftests/mm/Makefile index 2ed88132491164..beacc0f873049c 100644 --- a/tools/testing/selftests/mm/Makefile +++ b/tools/testing/selftests/mm/Makefile @@ -107,6 +107,7 @@ TEST_GEN_FILES += rmap TEST_GEN_FILES += folio_split_race_test TEST_GEN_FILES += folio_order_check TEST_GEN_FILES += khugepaged_sync_check +TEST_GEN_FILES += khugepaged_race TEST_GEN_FILES += soft-dirty ifeq ($(ARCH),x86_64) diff --git a/tools/testing/selftests/mm/khugepaged_race.c b/tools/testing/selftests/mm/khugepaged_race.c new file mode 100644 index 00000000000000..66c9e2c10f6601 --- /dev/null +++ b/tools/testing/selftests/mm/khugepaged_race.c @@ -0,0 +1,415 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * Race collapse against faults, GUP pins, fork, mremap and MADV_DONTNEED + * over the same ranges. A racing page must read as its pattern or as + * zero, never anything else; the kernel's own assertions in dmesg are the + * other half of the check. + */ +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "kselftest.h" +#include "vm_util.h" +#include +#include "../../../../mm/gup_test.h" + +#ifndef FOLL_WRITE +#define FOLL_WRITE 0x01 +#endif + +#define BASE_ADDR ((void *)(1UL << 30)) +#define PASS_TIMEOUT_S 30 + +/* + * PMD-sized areas the racing threads share, plus one for the mremap + * thread. -a shrinks it where a PMD is 512M. + */ +#define DEFAULT_SHARED_AREAS 3 +static int nr_shared_areas; +static int nr_areas; + +static unsigned long hpage_pmd_size; +static unsigned long page_size; +/* nr_areas PMD-sized areas; the last one belongs to the mremap thread */ +static char *region; +static char *mremap_area; +static char *mremap_scratch; +static int gup_fd = -1; +static volatile int stop; +static volatile int corrupted; + +static unsigned int pattern(unsigned long page_idx) +{ + unsigned int val = (unsigned int)page_idx * 2654435761U; + + return val ? val : 1; /* never collides with the zero-fill */ +} + +/* Zero means never written; anything else must be this page's pattern */ +static bool page_is_corrupt(unsigned long page_idx, unsigned int *val) +{ + *val = *(unsigned int *)(region + page_idx * page_size); + + return *val && *val != pattern(page_idx); +} + +static void check_page(unsigned long page_idx) +{ + unsigned int val; + + if (page_is_corrupt(page_idx, &val)) { + corrupted = 1; + ksft_print_msg("Corruption at page %lu: %#x != %#x\n", + page_idx, val, pattern(page_idx)); + } +} + +static unsigned long shared_pages(void) +{ + return nr_shared_areas * hpage_pmd_size / page_size; +} + +static unsigned long rand_page(unsigned int *seed) +{ + return (unsigned long)rand_r(seed) % shared_pages(); +} + +/* Clamp so a range never reaches the mremap thread's area */ +static unsigned long room_from(unsigned long page_idx, unsigned long want) +{ + unsigned long left = shared_pages() - page_idx; + + return want < left ? want : left; +} + +static void *faulter_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + + while (!stop) { + unsigned long page_idx = rand_page(&seed); + + if (rand_r(&seed) & 1) + *(unsigned int *)(region + page_idx * page_size) = + pattern(page_idx); + else + check_page(page_idx); + } + return NULL; +} + +static void *dontneed_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + + while (!stop) { + unsigned long page_idx = rand_page(&seed); + unsigned long nr = 1UL << (rand_r(&seed) % 6); /* 1..32 pages */ + + madvise(region + page_idx * page_size, + room_from(page_idx, nr) * page_size, MADV_DONTNEED); + usleep(rand_r(&seed) % 500); + } + return NULL; +} + +static void *pinner_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + + while (!stop) { + struct gup_test gup = {}; + unsigned long page_idx = rand_page(&seed); + unsigned long nr = room_from(page_idx, 16); + + gup.addr = (unsigned long)(region + page_idx * page_size); + gup.size = nr * page_size; + gup.nr_pages_per_call = nr; + gup.gup_flags = FOLL_WRITE; + /* Racing MADV_DONTNEED makes transient failures expected */ + ioctl(gup_fd, PIN_FAST_BENCHMARK, &gup); + usleep(rand_r(&seed) % 200); + } + return NULL; +} + +static void *forker_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + + while (!stop) { + pid_t pid = fork(); + + if (pid == 0) { + unsigned int val; + int bad = 0; + + /* + * No stdio in the child: a thread may hold stdout's + * lock across the fork, and printing under it hangs. + */ + for (int i = 0; i < 16; i++) + bad |= page_is_corrupt(rand_page(&seed), &val); + _exit(bad); + } + if (pid > 0) { + int wstatus; + + if (waitpid(pid, &wstatus, 0) < 0) + ksft_exit_fail_perror("waitpid()"); + /* A child killed on the read counts too */ + if (!WIFEXITED(wstatus) || WEXITSTATUS(wstatus)) + corrupted = 1; + } + usleep(rand_r(&seed) % 2000); + } + return NULL; +} + +static void *mremapper_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + + while (!stop) { + void *p; + + p = mremap(mremap_area, hpage_pmd_size, hpage_pmd_size, + MREMAP_MAYMOVE | MREMAP_FIXED, mremap_scratch); + if (p == MAP_FAILED) + ksft_exit_fail_perror("mremap() away"); + for (int i = 0; i < 8; i++) + mremap_scratch[(rand_r(&seed) % + (hpage_pmd_size / page_size)) * page_size] = 1; + p = mremap(mremap_scratch, hpage_pmd_size, hpage_pmd_size, + MREMAP_MAYMOVE | MREMAP_FIXED, mremap_area); + if (p == MAP_FAILED) + ksft_exit_fail_perror("mremap() back"); + /* The move back unmapped the scratch address: claim it again */ + if (mmap(mremap_scratch, hpage_pmd_size, PROT_NONE, + MAP_ANONYMOUS | MAP_PRIVATE | MAP_FIXED_NOREPLACE, + -1, 0) != (void *)mremap_scratch) + ksft_exit_fail_perror("mmap() mremap scratch"); + usleep(rand_r(&seed) % 2000); + } + return NULL; +} + +static unsigned long now_ms(void) +{ + struct timeval tv; + + gettimeofday(&tv, NULL); + return tv.tv_sec * 1000UL + tv.tv_usec / 1000; +} + +static void usage(void) +{ + fprintf(stderr, + "Usage: khugepaged_race [-d seconds] [-m stepped|free|madvise] [-a areas] [-t mask]\n" + "\tWithout -m, every mode runs in turn.\n" + "\t-d: seconds per mode (default 5)\n" + "\t-a: number of shared PMD-sized playground areas (default 3)\n" + "\t-t: bitmask of racing threads to start, for bisecting a failure\n"); + exit(1); +} + +int main(int argc, char **argv) +{ + static const char * const thread_names[] = { + "faulter", "faulter2", "dontneed", "pinner", "forker", + "mremapper", + }; + void *(*const thread_fns[])(void *) = { + faulter_fn, faulter_fn, dontneed_fn, pinner_fn, forker_fn, + mremapper_fn, + }; + const int nr_threads = ARRAY_SIZE(thread_names); + pthread_t threads[ARRAY_SIZE(thread_names)]; + static const char * const all_modes[] = { "stepped", "free", "madvise" }; + const char *one_mode[1]; + const char * const *modes = all_modes; + int nr_modes = ARRAY_SIZE(all_modes); + const char *mode_arg = NULL; + struct thp_settings settings; + unsigned long end_ms; + int duration_s = 5; + unsigned long thread_mask = ~0UL; + int nr_areas_arg = 0; + unsigned long i; + int steps = 0; + int opt; + + while ((opt = getopt(argc, argv, "a:d:m:t:h")) != -1) { + switch (opt) { + case 'a': + nr_areas_arg = atoi(optarg); + break; + case 'd': + duration_s = atoi(optarg); + break; + case 'm': + mode_arg = optarg; + break; + case 't': + thread_mask = strtoul(optarg, NULL, 0); + break; + default: + usage(); + } + } + + if (mode_arg) { + if (strcmp(mode_arg, "stepped") && strcmp(mode_arg, "free") && + strcmp(mode_arg, "madvise")) + usage(); + one_mode[0] = mode_arg; + modes = one_mode; + nr_modes = 1; + } + + ksft_print_header(); + if (!thp_available()) + ksft_exit_skip("Transparent Hugepages not available\n"); + + page_size = getpagesize(); + hpage_pmd_size = read_pmd_pagesize(); + if (!hpage_pmd_size) + ksft_exit_fail_msg("Reading PMD pagesize failed\n"); + + gup_fd = open("/sys/kernel/debug/gup_test", O_RDWR); + if (gup_fd < 0) + ksft_exit_skip("/sys/kernel/debug/gup_test requires CONFIG_GUP_TEST and root\n"); + + nr_shared_areas = nr_areas_arg > 0 ? nr_areas_arg : DEFAULT_SHARED_AREAS; + nr_areas = nr_shared_areas + 1; + + /* + * MREMAP_FIXED unmaps whatever is in the way without saying so, so + * claim the mremap thread's scratch address up front. + */ + mremap_scratch = (char *)BASE_ADDR + 2 * nr_areas * hpage_pmd_size; + if (mmap(mremap_scratch, hpage_pmd_size, PROT_NONE, + MAP_ANONYMOUS | MAP_PRIVATE | MAP_FIXED_NOREPLACE, + -1, 0) != (void *)mremap_scratch) + ksft_exit_fail_perror("mmap() mremap scratch"); + + ksft_set_plan(nr_modes); + + thp_save_settings(); + thp_read_settings(&settings); + + /* Base of the settings stack; the bottom entry is never popped */ + thp_push_settings(&settings); + + for (int m = 0; m < nr_modes; m++) { + const char *mode = modes[m]; + + thp_read_settings(&settings); + settings.thp_enabled = THP_MADVISE; + settings.thp_defrag = THP_DEFRAG_ALWAYS; + settings.shmem_enabled = SHMEM_NEVER; + settings.khugepaged.defrag = 1; + settings.khugepaged.scan_sleep_millisecs = + strcmp(mode, "free") ? 1000 : 0; + settings.khugepaged.alloc_sleep_millisecs = 10; + /* + * mTHP collapse honours only 0 or HPAGE_PMD_NR - 1 here, and 0 + * keeps a step from being spent on PMD allocations that racing + * MADV_DONTNEED will not let succeed. + */ + settings.khugepaged.max_ptes_none = 0; + /* One wake, one pass: the playground plus the forked children's copies */ + settings.khugepaged.pages_to_scan = + nr_areas * (hpage_pmd_size / page_size) * 8; + for (i = 0; i < NR_ORDERS; i++) { + if (thp_supported_orders() & (1UL << i)) + settings.hugepages[i].enabled = THP_INHERIT; + } + thp_push_settings(&settings); + + region = mmap(BASE_ADDR, nr_areas * hpage_pmd_size, + PROT_READ | PROT_WRITE, MAP_ANONYMOUS | + MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0); + if (region != BASE_ADDR) + ksft_exit_fail_perror("mmap() playground"); + mremap_area = region + nr_shared_areas * hpage_pmd_size; + + /* Populate so the first pass has something to collapse */ + for (i = 0; i < nr_shared_areas * hpage_pmd_size / page_size; i++) + *(unsigned int *)(region + i * page_size) = pattern(i); + memset(mremap_area, 1, hpage_pmd_size); + if (madvise(region, nr_areas * hpage_pmd_size, MADV_HUGEPAGE)) + ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)"); + + for (i = 0; i < nr_threads; i++) { + if (!(thread_mask & (1UL << i))) { + threads[i] = 0; + continue; + } + if (pthread_create(&threads[i], NULL, thread_fns[i], + (void *)(i + 1))) + ksft_exit_fail_perror(thread_names[i]); + } + + end_ms = now_ms() + duration_s * 1000UL; + if (!strcmp(mode, "stepped")) { + while (now_ms() < end_ms && !corrupted) { + if (!khugepaged_full_pass(PASS_TIMEOUT_S)) + ksft_exit_fail_msg("khugepaged pass timed out\n"); + steps++; + } + } else if (!strcmp(mode, "free")) { + while (now_ms() < end_ms && !corrupted) + usleep(100 * 1000); + } else { /* madvise */ + while (now_ms() < end_ms && !corrupted) { + for (i = 0; i < nr_shared_areas; i++) { + madvise(region + i * hpage_pmd_size, + hpage_pmd_size, MADV_COLLAPSE); + } + madvise(region, nr_shared_areas * hpage_pmd_size, + MADV_DONTNEED); + steps++; + } + } + + stop = 1; + for (i = 0; i < nr_threads; i++) { + if (threads[i]) + pthread_join(threads[i], NULL); + } + + for (i = 0; i < nr_shared_areas * hpage_pmd_size / page_size; i++) + check_page(i); + + ksft_test_result(!corrupted, + "%s: %ds, %d steps, no corruption\n", + mode, duration_s, steps); + + /* The next mode maps the same fixed address with its own settings */ + munmap(region, nr_areas * hpage_pmd_size); + thp_pop_settings(); + stop = 0; + steps = 0; + + if (corrupted) { + /* Memory is suspect; the rest would prove nothing */ + while (++m < nr_modes) + ksft_test_result_skip("%s: skipped after corruption\n", + modes[m]); + break; + } + } + + ksft_finished(); +} diff --git a/tools/testing/selftests/mm/run_vmtests.sh b/tools/testing/selftests/mm/run_vmtests.sh index 3e8c76cf4948b0..a1b45a3dedaeea 100755 --- a/tools/testing/selftests/mm/run_vmtests.sh +++ b/tools/testing/selftests/mm/run_vmtests.sh @@ -386,6 +386,8 @@ CATEGORY="thp" run_test ./folio_order_check CATEGORY="thp" run_test ./khugepaged_sync_check +CATEGORY="thp" run_test ./khugepaged_race + CATEGORY="thp" run_test ./khugepaged CATEGORY="thp" run_test ./khugepaged -s 2 From 495d6995c80c8e46a141271de666014337dee932 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:47 +0100 Subject: [PATCH 0805/1012] selftests/mm: race the collapse of windows with holes The harness pins max_ptes_none to 0, so khugepaged only collapses a window once every PTE in it is present. A window with holes takes a different route, and never gets raced. A hole is zero-filled in the new folio rather than copied. Which slots count as holes keeps moving under the racing MADV_DONTNEED, right up to the moment the PMD is detached. Run both ends of the occupancy scale for every driver mode, one after the other. mTHP collapse supports only those two, 0 and HPAGE_PMD_NR - 1, and coerces anything between them to 0. Each result says which end it ran: ok 1 stepped/strict: 5s, 231 steps, no corruption ok 2 stepped/holes: 5s, 194 steps, no corruption Link: https://lore.kernel.org/20260919002451.496763-18-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged_race.c | 32 ++++++++++++-------- 1 file changed, 20 insertions(+), 12 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged_race.c b/tools/testing/selftests/mm/khugepaged_race.c index 66c9e2c10f6601..6575045c048262 100644 --- a/tools/testing/selftests/mm/khugepaged_race.c +++ b/tools/testing/selftests/mm/khugepaged_race.c @@ -236,6 +236,8 @@ int main(int argc, char **argv) const int nr_threads = ARRAY_SIZE(thread_names); pthread_t threads[ARRAY_SIZE(thread_names)]; static const char * const all_modes[] = { "stepped", "free", "madvise" }; + static const bool occupancies[] = { false, true }; /* strict, holes */ + const int nr_occupancies = ARRAY_SIZE(occupancies); const char *one_mode[1]; const char * const *modes = all_modes; int nr_modes = ARRAY_SIZE(all_modes); @@ -303,7 +305,7 @@ int main(int argc, char **argv) -1, 0) != (void *)mremap_scratch) ksft_exit_fail_perror("mmap() mremap scratch"); - ksft_set_plan(nr_modes); + ksft_set_plan(nr_modes * nr_occupancies); thp_save_settings(); thp_read_settings(&settings); @@ -311,8 +313,9 @@ int main(int argc, char **argv) /* Base of the settings stack; the bottom entry is never popped */ thp_push_settings(&settings); - for (int m = 0; m < nr_modes; m++) { - const char *mode = modes[m]; + for (int run = 0; run < nr_modes * nr_occupancies; run++) { + const char *mode = modes[run / nr_occupancies]; + bool holes = occupancies[run % nr_occupancies]; thp_read_settings(&settings); settings.thp_enabled = THP_MADVISE; @@ -322,12 +325,14 @@ int main(int argc, char **argv) settings.khugepaged.scan_sleep_millisecs = strcmp(mode, "free") ? 1000 : 0; settings.khugepaged.alloc_sleep_millisecs = 10; + /* - * mTHP collapse honours only 0 or HPAGE_PMD_NR - 1 here, and 0 - * keeps a step from being spent on PMD allocations that racing - * MADV_DONTNEED will not let succeed. + * mTHP collapse honours only 0 or HPAGE_PMD_NR - 1 here. The two + * ends race different paths: a strict window has every PTE + * present, a hole-heavy one is mostly zero-filled. */ - settings.khugepaged.max_ptes_none = 0; + settings.khugepaged.max_ptes_none = holes ? + (hpage_pmd_size / page_size) - 1 : 0; /* One wake, one pass: the playground plus the forked children's copies */ settings.khugepaged.pages_to_scan = nr_areas * (hpage_pmd_size / page_size) * 8; @@ -393,8 +398,9 @@ int main(int argc, char **argv) check_page(i); ksft_test_result(!corrupted, - "%s: %ds, %d steps, no corruption\n", - mode, duration_s, steps); + "%s/%s: %ds, %d steps, no corruption\n", + mode, holes ? "holes" : "strict", + duration_s, steps); /* The next mode maps the same fixed address with its own settings */ munmap(region, nr_areas * hpage_pmd_size); @@ -404,9 +410,11 @@ int main(int argc, char **argv) if (corrupted) { /* Memory is suspect; the rest would prove nothing */ - while (++m < nr_modes) - ksft_test_result_skip("%s: skipped after corruption\n", - modes[m]); + while (++run < nr_modes * nr_occupancies) + ksft_test_result_skip("%s/%s: skipped after corruption\n", + modes[run / nr_occupancies], + occupancies[run % nr_occupancies] ? + "holes" : "strict"); break; } } From a34aa616610fdb59782959ee605c3b74c6e6fca2 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:48 +0100 Subject: [PATCH 0806/1012] selftests/mm: add memory-pressure threads to the khugepaged race harness The harness races collapse against faults, pins, fork, mremap and MADV_DONTNEED, but nothing in it runs reclaim or compaction against the collapse. Add two more threads, and run every mode and occupancy limit both with and without them: - pageout: cycles MADV_PAGEOUT over a region of its own, faults it back in and checks the content each round, since a page's pattern must survive the trip through swap. Left out when the host has no swap, because then there is no anon reclaim to drive. - compactor: writes /proc/sys/vm/compact_memory in a loop. Compaction isolates and migrates folios, so it competes with a collapse for the pages it is gathering, with refcount elevations and migration entries of its own. Each result says whether it ran under pressure: ok 2 stepped/strict/pressure: 5s, 88 steps, no corruption A full run is now twelve combinations; -m picks one mode, -d shortens each run. Link: https://lore.kernel.org/20260919002451.496763-19-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged_race.c | 142 ++++++++++++++++--- 1 file changed, 122 insertions(+), 20 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged_race.c b/tools/testing/selftests/mm/khugepaged_race.c index 6575045c048262..68046f0bf6d318 100644 --- a/tools/testing/selftests/mm/khugepaged_race.c +++ b/tools/testing/selftests/mm/khugepaged_race.c @@ -44,6 +44,8 @@ static unsigned long page_size; static char *region; static char *mremap_area; static char *mremap_scratch; +static char *pageout_area; +static size_t pageout_size; static int gup_fd = -1; static volatile int stop; static volatile int corrupted; @@ -204,6 +206,69 @@ static void *mremapper_fn(void *arg) return NULL; } +/* + * Swap traffic and LRU churn on a region nothing else writes, so a page's + * pattern must survive the trip through swap exactly. + */ +static void *pageout_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + unsigned long nr = pageout_size / page_size; + unsigned long i; + + for (i = 0; i < nr; i++) + *(unsigned int *)(pageout_area + i * page_size) = pattern(i); + + while (!stop) { + madvise(pageout_area, pageout_size, MADV_PAGEOUT); + for (i = 0; i < nr && !stop; i++) { + unsigned int val = *(unsigned int *)(pageout_area + + i * page_size); + + if (val != pattern(i)) { + corrupted = 1; + ksft_print_msg("Pageout corruption at page %lu: %#x != %#x\n", + i, val, pattern(i)); + } + } + usleep(rand_r(&seed) % 2000); + } + return NULL; +} + +/* Compaction migrates the collapse sources while they are being gathered */ +static void *compactor_fn(void *arg) +{ + unsigned int seed = (unsigned long)arg; + int fd = open("/proc/sys/vm/compact_memory", O_WRONLY); + + if (fd < 0) { + ksft_print_msg("No compact_memory; compactor idle\n"); + return NULL; + } + while (!stop) { + if (write(fd, "1", 1) < 0) + break; + usleep(10000 + rand_r(&seed) % 100000); + } + close(fd); + return NULL; +} + +static bool swap_available(void) +{ + char line[256]; + int lines = 0; + FILE *fp = fopen("/proc/swaps", "r"); + + if (!fp) + return false; + while (fgets(line, sizeof(line), fp)) + lines++; + fclose(fp); + return lines > 1; +} + static unsigned long now_ms(void) { struct timeval tv; @@ -227,17 +292,23 @@ int main(int argc, char **argv) { static const char * const thread_names[] = { "faulter", "faulter2", "dontneed", "pinner", "forker", - "mremapper", + "mremapper", "pageout", "compactor", }; void *(*const thread_fns[])(void *) = { faulter_fn, faulter_fn, dontneed_fn, pinner_fn, forker_fn, - mremapper_fn, + mremapper_fn, pageout_fn, compactor_fn, }; + enum { T_FAULTER, T_FAULTER2, T_DONTNEED, T_PINNER, T_FORKER, + T_MREMAPPER, T_PAGEOUT, T_COMPACTOR }; + const unsigned long pageout_bit = 1UL << T_PAGEOUT; + const unsigned long compactor_bit = 1UL << T_COMPACTOR; const int nr_threads = ARRAY_SIZE(thread_names); pthread_t threads[ARRAY_SIZE(thread_names)]; static const char * const all_modes[] = { "stepped", "free", "madvise" }; static const bool occupancies[] = { false, true }; /* strict, holes */ + static const bool pressures[] = { false, true }; /* quiet, under pressure */ const int nr_occupancies = ARRAY_SIZE(occupancies); + const int nr_pressures = ARRAY_SIZE(pressures); const char *one_mode[1]; const char * const *modes = all_modes; int nr_modes = ARRAY_SIZE(all_modes); @@ -246,6 +317,9 @@ int main(int argc, char **argv) unsigned long end_ms; int duration_s = 5; unsigned long thread_mask = ~0UL; + unsigned long base_mask; + bool have_swap; + char label[64]; int nr_areas_arg = 0; unsigned long i; int steps = 0; @@ -305,7 +379,13 @@ int main(int argc, char **argv) -1, 0) != (void *)mremap_scratch) ksft_exit_fail_perror("mmap() mremap scratch"); - ksft_set_plan(nr_modes * nr_occupancies); + base_mask = thread_mask; + have_swap = swap_available(); + if (!have_swap) + /* No swap, no anon reclaim: compaction-only pressure */ + ksft_print_msg("no swap: the pageout thread is not started\n"); + + ksft_set_plan(nr_modes * nr_occupancies * nr_pressures); thp_save_settings(); thp_read_settings(&settings); @@ -313,9 +393,25 @@ int main(int argc, char **argv) /* Base of the settings stack; the bottom entry is never popped */ thp_push_settings(&settings); - for (int run = 0; run < nr_modes * nr_occupancies; run++) { - const char *mode = modes[run / nr_occupancies]; - bool holes = occupancies[run % nr_occupancies]; + for (int run = 0; run < nr_modes * nr_occupancies * nr_pressures; run++) { + int rem = run % (nr_occupancies * nr_pressures); + const char *mode = modes[run / (nr_occupancies * nr_pressures)]; + bool holes = occupancies[rem / nr_pressures]; + bool pressure = pressures[rem % nr_pressures]; + + snprintf(label, sizeof(label), "%s/%s%s", mode, + holes ? "holes" : "strict", pressure ? "/pressure" : ""); + if (corrupted) { + /* Memory is suspect; the rest would prove nothing */ + ksft_test_result_skip("%s: skipped after corruption\n", label); + continue; + } + + thread_mask = base_mask; + if (!pressure) + thread_mask &= ~(pageout_bit | compactor_bit); + else if (!have_swap) + thread_mask &= ~pageout_bit; thp_read_settings(&settings); settings.thp_enabled = THP_MADVISE; @@ -349,6 +445,20 @@ int main(int argc, char **argv) ksft_exit_fail_perror("mmap() playground"); mremap_area = region + nr_shared_areas * hpage_pmd_size; + if (thread_mask & pageout_bit) { + /* Enough to drive real reclaim without swamping a small guest */ + pageout_size = 4 * hpage_pmd_size; + if (pageout_size < 16UL << 20) + pageout_size = 16UL << 20; + if (pageout_size > 64UL << 20) + pageout_size = 64UL << 20; + pageout_area = mmap(NULL, pageout_size, + PROT_READ | PROT_WRITE, + MAP_ANONYMOUS | MAP_PRIVATE, -1, 0); + if (pageout_area == MAP_FAILED) + ksft_exit_fail_perror("mmap() pageout area"); + } + /* Populate so the first pass has something to collapse */ for (i = 0; i < nr_shared_areas * hpage_pmd_size / page_size; i++) *(unsigned int *)(region + i * page_size) = pattern(i); @@ -397,26 +507,18 @@ int main(int argc, char **argv) for (i = 0; i < nr_shared_areas * hpage_pmd_size / page_size; i++) check_page(i); - ksft_test_result(!corrupted, - "%s/%s: %ds, %d steps, no corruption\n", - mode, holes ? "holes" : "strict", - duration_s, steps); + ksft_test_result(!corrupted, "%s: %ds, %d steps, no corruption\n", + label, duration_s, steps); /* The next mode maps the same fixed address with its own settings */ munmap(region, nr_areas * hpage_pmd_size); + if (pageout_area) { + munmap(pageout_area, pageout_size); + pageout_area = NULL; + } thp_pop_settings(); stop = 0; steps = 0; - - if (corrupted) { - /* Memory is suspect; the rest would prove nothing */ - while (++run < nr_modes * nr_occupancies) - ksft_test_result_skip("%s/%s: skipped after corruption\n", - modes[run / nr_occupancies], - occupancies[run % nr_occupancies] ? - "holes" : "strict"); - break; - } } ksft_finished(); From e518c1ce0509c51638cce78bba4b3c0edff34085 Mon Sep 17 00:00:00 2001 From: "Kiryl Shutsemau (Meta)" Date: Sat, 19 Sep 2026 01:24:49 +0100 Subject: [PATCH 0807/1012] selftests/mm: zap whole PTE tables in the khugepaged race harness The harness's MADV_DONTNEED thread zaps 1 to 32 pages at a time, never a whole PMD-aligned area, and only a zap that covers a full table frees the table itself (CONFIG_PT_RECLAIM). Make the thread zap a whole PMD-aligned area about one iteration in 64, and keep the fine-grained zaps as the common case. The new case frees page tables, racing that against a collapse walking the same table. Link: https://lore.kernel.org/20260919002451.496763-20-kirill@shutemov.name Signed-off-by: Kiryl Shutsemau (Meta) Signed-off-by: Andrew Morton Tested-by: Muhammad Usama Anjum Assisted-by: LLM Cc: Alexander Gordeev Cc: Baolin Wang Cc: Barry Song Cc: "David Hildenbrand (arm)" Cc: Dev Jain Cc: Hugh Dickens Cc: Jason Gunthorpe Cc: kernel-team@meta.com Cc: Lance Yang Cc: Leon Romanovsky Cc: Liam Howlett Cc: Lorenzo Stoakes (ARM) Cc: Michal Hocko Cc: Mike Rapoport (Microsoft) Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Shuah Khan (Samsung OSG) Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" Cc: Zi Yan --- tools/testing/selftests/mm/khugepaged_race.c | 17 +++++++++++++++-- 1 file changed, 15 insertions(+), 2 deletions(-) diff --git a/tools/testing/selftests/mm/khugepaged_race.c b/tools/testing/selftests/mm/khugepaged_race.c index 68046f0bf6d318..abb0678522cd9f 100644 --- a/tools/testing/selftests/mm/khugepaged_race.c +++ b/tools/testing/selftests/mm/khugepaged_race.c @@ -118,8 +118,21 @@ static void *dontneed_fn(void *arg) unsigned long page_idx = rand_page(&seed); unsigned long nr = 1UL << (rand_r(&seed) % 6); /* 1..32 pages */ - madvise(region + page_idx * page_size, - room_from(page_idx, nr) * page_size, MADV_DONTNEED); + /* + * Now and then zap a whole PMD-aligned area: only a zap that + * covers the full table frees the table itself (CONFIG_PT_RECLAIM). + */ + if (!(rand_r(&seed) % 64)) { + unsigned long area = page_idx / + (hpage_pmd_size / page_size); + + madvise(region + area * hpage_pmd_size, + hpage_pmd_size, MADV_DONTNEED); + } else { + madvise(region + page_idx * page_size, + room_from(page_idx, nr) * page_size, + MADV_DONTNEED); + } usleep(rand_r(&seed) % 500); } return NULL; From f353a5bea00874594d990edb547e099fcecf0260 Mon Sep 17 00:00:00 2001 From: Lance Yang Date: Mon, 21 Sep 2026 13:42:25 +0800 Subject: [PATCH 0808/1012] mm: disallow raw PFN mappings of huge/shared zeropage Handling the huge/shared zeropage correctly in vmf_insert_pfn_pmd() and vmf_insert_pfn_prot() is more involved. We would need to check whether the VMA allows it and keep the mapping read-only, similar to the checks in vm_mixed_ok(). No in-tree user needs that support, so reject these mappings with VM_FAULT_SIGBUS rather than complicate the code for now. Link: https://lore.kernel.org/20260921054225.28537-1-lance.yang@linux.dev Link: https://lore.kernel.org/all/20260917121010.60966-1-lance.yang@linux.dev/ Signed-off-by: Lance Yang Signed-off-by: Andrew Morton Suggested-by: Kiryl Shutsemau (Meta) Suggested-by: David Hildenbrand (Arm) Reviewed-by: Kiryl Shutsemau (Meta) Acked-by: David Hildenbrand (Arm) Acked-by: Zi Yan Reviewed-by: Lorenzo Stoakes (ARM) Cc: Baolin Wang Cc: Barry Song Cc: Dev Jain Cc: Liam Howlett Cc: Michal Hocko Cc: "Mike Rapoport (IBM)" Cc: Nico Pache (Red Hat) Cc: Ryan Roberts Cc: Suren Baghdasaryan Cc: Usama Arif Cc: "Vlastimil Babka (SUSE)" --- mm/huge_memory.c | 3 +++ mm/memory.c | 3 +++ 2 files changed, 6 insertions(+) diff --git a/mm/huge_memory.c b/mm/huge_memory.c index ba5e20bbfd3bc3..c5c210b3ebc646 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -1722,6 +1722,9 @@ vm_fault_t vmf_insert_pfn_pmd(struct vm_fault *vmf, unsigned long pfn, (VM_PFNMAP|VM_MIXEDMAP)); BUG_ON((vma->vm_flags & VM_PFNMAP) && vma_is_cow_mapping(vma)); + if (unlikely(is_huge_zero_pfn(pfn))) + return VM_FAULT_SIGBUS; + pfnmap_setup_cachemode_pfn(pfn, &pgprot); return insert_pmd(vma, addr, vmf->pmd, fop, pgprot, write); diff --git a/mm/memory.c b/mm/memory.c index 6349ef676549a8..79fa57a381ce00 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -2956,6 +2956,9 @@ vm_fault_t vmf_insert_pfn_prot(struct vm_area_struct *vma, unsigned long addr, BUG_ON((vma->vm_flags & VM_PFNMAP) && vma_is_cow_mapping(vma)); BUG_ON((vma->vm_flags & VM_MIXEDMAP) && pfn_valid(pfn)); + if (unlikely(is_zero_pfn(pfn))) + return VM_FAULT_SIGBUS; + if (addr < vma->vm_start || addr >= vma->vm_end) return VM_FAULT_SIGBUS; From ef50de60e1ac129d3a3b7faa2e24cd7cc7e5fca4 Mon Sep 17 00:00:00 2001 From: Donggeun Yoo Date: Mon, 21 Sep 2026 08:24:41 -0700 Subject: [PATCH 0809/1012] mm/damon/core: charge only the part of a region the filter left Patch series "mm/damon/core: fix the size charged for a filter-trimmed region", v2. An address range DAMOS filter that partially overlaps a monitoring region splits it, so the scheme action reaches only one side of the boundary. damos_apply_scheme() reads the region size before that split and never re-reads it, so the quota charge and schemes//stats/sz_tried account for the whole original region. Patch 1 re-reads the size after the core filters have run. Patch 2 adds KUnit coverage for what the function charges and reports, for the trimmed cases and for the cases that must not change. This patch (of 2): An address range DAMOS filter that partially overlaps a monitoring region splits the region at the filter boundary, so the scheme action is applied to only one side of it. damos_apply_scheme() reads the region size once on entry, before damos_core_filter_out() performs that split. The quota charge and the statistics update therefore account for the region as it was before the trim. schemes//stats/sz_tried counts memory the filter excluded, which Documentation/mm/damon/design.rst says is not counted as tried, and with quotas/bytes set the excluded part is charged against the budget, throttling the scheme to a fraction of what was configured. User impact is that DAMOS could unexpectedly slowly run, due to over-charged quota. DAMOS stat could also be confusing. No critical events such as crashes or leask happen. Re-read the region size after the core filters have run. Link: https://lore.kernel.org/20260921152443.80132-1-sj@kernel.org Link: https://lore.kernel.org/20260921152443.80132-2-sj@kernel.org Fixes: ab9bda001b68 ("mm/damon/core: introduce address range type damos filter") Signed-off-by: Donggeun Yoo Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Brendan Higgins Cc: David Gow Cc: --- mm/damon/core.c | 1 + 1 file changed, 1 insertion(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index add1b7afb957ac..8d8b0cba1a56c9 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2784,6 +2784,7 @@ static void damos_apply_scheme(struct damon_ctx *c, struct damon_target *t, } if (damos_core_filter_out(c, t, r, s)) return; + sz = damon_sz_region(r); ktime_get_coarse_ts64(&begin); trace_damos_before_apply(cidx, sidx, tidx, r, nr_accesses, damon_nr_regions(t), do_trace); From 9755b3f485c0bead8cefffc9df201abfd4a260e2 Mon Sep 17 00:00:00 2001 From: Donggeun Yoo Date: Mon, 21 Sep 2026 08:24:42 -0700 Subject: [PATCH 0810/1012] mm/damon/tests/core-kunit: test the size charged for a filter-trimmed region damos_test_filter_out() checks that an address range filter splits a region at the filter boundary, and stops there. Nothing checks what damos_apply_scheme() then charges to the quota and reports as tried. damos_test_apply_scheme_filtered_sz() covers the two ways such a filter trims a region: one that starts before the filter's range, and one that starts inside it. damos_test_apply_scheme_filter_sz_unchanged() and damos_test_apply_scheme_quota_sz() cover the cases whose accounting must not change: a region wholly inside a reject range, one wholly inside an allow range, two filters trimming in a single call, a filter type that never splits, no filter at all, a region the quota trims, and a quota remainder too small for one region. Link: https://lore.kernel.org/20260921152443.80132-3-sj@kernel.org Signed-off-by: Donggeun Yoo Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Brendan Higgins Cc: David Gow Cc: --- mm/damon/tests/core-kunit.h | 316 ++++++++++++++++++++++++++++++++++++ 1 file changed, 316 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index 5ff0436c58441f..df84d9cc7d204a 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -1706,6 +1706,319 @@ static void damos_test_filter_out(struct kunit *test) damos_free_filter(f); } +static unsigned long damos_test_apply_scheme_stub(struct damon_ctx *c, + struct damon_target *t, struct damon_region *r, + struct damos *s, unsigned long *sz_filter_passed) +{ + return 0; +} + +static void damos_test_apply_scheme_filtered_sz(struct kunit *test) +{ + struct damos_access_pattern pattern = { + .min_sz_region = 0, + .max_sz_region = ULONG_MAX, + .min_nr_accesses = 0, + .max_nr_accesses = UINT_MAX, + .min_age_region = 0, + .max_age_region = UINT_MAX, + }; + unsigned long min_sz = DAMON_MIN_REGION_SZ; + struct damos_watermarks wmarks = {}; + struct damos_quota quota = {}; + struct damon_ctx *c; + struct damon_target *t; + struct damon_region *r; + struct damos_filter *f; + struct damos *s; + + c = damon_new_ctx(); + if (!c) + kunit_skip(test, "ctx alloc fail"); + c->ops.apply_scheme = damos_test_apply_scheme_stub; + + s = damon_new_scheme(&pattern, DAMOS_STAT, 0, "a, &wmarks, + NUMA_NO_NODE); + if (!s) { + damon_destroy_ctx(c); + kunit_skip(test, "scheme alloc fail"); + } + damon_add_scheme(c, s); + + t = damon_new_target(); + if (!t) { + damon_destroy_ctx(c); + kunit_skip(test, "target alloc fail"); + } + damon_add_target(c, t); + + f = damos_new_filter(DAMOS_FILTER_TYPE_ADDR, true, false); + if (!f) { + damon_destroy_ctx(c); + kunit_skip(test, "filter alloc fail"); + } + f->addr_range = (struct damon_addr_range){ + .start = 2 * min_sz, + .end = 8 * min_sz + }; + damos_add_filter(s, f); + damos_set_filters_default_reject(s); + + /* reject filter, region starting before the range */ + r = damon_new_region(0, 4 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 2); + KUNIT_EXPECT_EQ(test, r->ar.end, 2 * min_sz); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 2 * min_sz); + + if (damon_nr_regions(t) != 2) + goto out; + damon_destroy_region(damon_next_region(r), t); + damon_destroy_region(r, t); + s->stat = (struct damos_stat){}; + + /* allow filter, region starting inside the range */ + f->allow = true; + damos_set_filters_default_reject(s); + r = damon_new_region(2 * min_sz, 10 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 2); + KUNIT_EXPECT_EQ(test, r->ar.end, 8 * min_sz); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 6 * min_sz); + +out: + damon_destroy_ctx(c); +} + +static void damos_test_apply_scheme_filter_sz_unchanged(struct kunit *test) +{ + struct damos_access_pattern pattern = { + .min_sz_region = 0, + .max_sz_region = ULONG_MAX, + .min_nr_accesses = 0, + .max_nr_accesses = UINT_MAX, + .min_age_region = 0, + .max_age_region = UINT_MAX, + }; + unsigned long min_sz = DAMON_MIN_REGION_SZ; + struct damos_watermarks wmarks = {}; + struct damos_quota quota = {}; + struct damos_filter *f, *f2; + struct damon_ctx *c; + struct damon_target *t; + struct damon_region *r; + struct damos *s; + + c = damon_new_ctx(); + if (!c) + kunit_skip(test, "ctx alloc fail"); + c->ops.apply_scheme = damos_test_apply_scheme_stub; + + s = damon_new_scheme(&pattern, DAMOS_STAT, 0, "a, &wmarks, + NUMA_NO_NODE); + if (!s) { + damon_destroy_ctx(c); + kunit_skip(test, "scheme alloc fail"); + } + damon_add_scheme(c, s); + + t = damon_new_target(); + if (!t) { + damon_destroy_ctx(c); + kunit_skip(test, "target alloc fail"); + } + damon_add_target(c, t); + + f = damos_new_filter(DAMOS_FILTER_TYPE_ADDR, true, false); + if (!f) { + damon_destroy_ctx(c); + kunit_skip(test, "filter alloc fail"); + } + f->addr_range = (struct damon_addr_range){ + .start = 2 * min_sz, .end = 8 * min_sz}; + damos_add_filter(s, f); + damos_set_filters_default_reject(s); + + /* wholly inside a reject range: not counted at all */ + r = damon_new_region(4 * min_sz, 6 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, s->stat.nr_tried, 0); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 0); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 1); + + /* wholly inside an allow range: counted whole, not split */ + f->allow = true; + damos_set_filters_default_reject(s); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 2 * min_sz); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 1); + + /* two filters, each trimming: the size left after both is counted */ + damon_destroy_region(r, t); + s->stat = (struct damos_stat){}; + f->allow = false; + f->addr_range = (struct damon_addr_range){ + .start = 4 * min_sz, .end = 12 * min_sz}; + f2 = damos_new_filter(DAMOS_FILTER_TYPE_ADDR, true, false); + if (!f2) { + damon_destroy_ctx(c); + kunit_skip(test, "filter alloc fail"); + } + f2->addr_range = (struct damon_addr_range){ + .start = 2 * min_sz, .end = 3 * min_sz}; + damos_add_filter(s, f2); + damos_set_filters_default_reject(s); + + r = damon_new_region(0, 8 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 3); + KUNIT_EXPECT_EQ(test, r->ar.end, 2 * min_sz); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 2 * min_sz); + + /* a core filter that never splits: counted whole */ + damos_destroy_filter(f2); + damos_destroy_filter(f); + f = damos_new_filter(DAMOS_FILTER_TYPE_TARGET, true, true); + if (!f) { + damon_destroy_ctx(c); + kunit_skip(test, "filter alloc fail"); + } + f->target_idx = 0; + damos_add_filter(s, f); + damos_set_filters_default_reject(s); + s->stat = (struct damos_stat){}; + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 3); + KUNIT_EXPECT_EQ(test, r->ar.end, 2 * min_sz); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 2 * min_sz); + + damon_destroy_ctx(c); +} + +static void damos_test_apply_scheme_quota_sz(struct kunit *test) +{ + struct damos_access_pattern pattern = { + .min_sz_region = 0, + .max_sz_region = ULONG_MAX, + .min_nr_accesses = 0, + .max_nr_accesses = UINT_MAX, + .min_age_region = 0, + .max_age_region = UINT_MAX, + }; + unsigned long min_sz = DAMON_MIN_REGION_SZ; + struct damos_watermarks wmarks = {}; + struct damos_quota quota = {}; + struct damon_region *r, *next; + struct damon_ctx *c; + struct damon_target *t; + struct damos *s; + + c = damon_new_ctx(); + if (!c) + kunit_skip(test, "ctx alloc fail"); + + s = damon_new_scheme(&pattern, DAMOS_STAT, 0, "a, &wmarks, + NUMA_NO_NODE); + if (!s) { + damon_destroy_ctx(c); + kunit_skip(test, "scheme alloc fail"); + } + damon_add_scheme(c, s); + damos_set_filters_default_reject(s); + + t = damon_new_target(); + if (!t) { + damon_destroy_ctx(c); + kunit_skip(test, "target alloc fail"); + } + damon_add_target(c, t); + + /* no apply_scheme operation: the whole region is counted */ + r = damon_new_region(0, 4 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 4 * min_sz); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 1); + + /* no filter and no quota: the whole region is counted */ + c->ops.apply_scheme = damos_test_apply_scheme_stub; + s->stat = (struct damos_stat){}; + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 4 * min_sz); + KUNIT_EXPECT_EQ(test, damon_nr_regions(t), 1); + + /* the quota trims the region: the trimmed size is counted */ + damon_for_each_region_safe(r, next, t) + damon_destroy_region(r, t); + s->stat = (struct damos_stat){}; + s->quota.esz = 2 * min_sz; + s->quota.charged_sz = 0; + + r = damon_new_region(0, 8 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, r->ar.end, 2 * min_sz); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 2 * min_sz); + + /* quota remainder below one region: tried, but nothing counted */ + damon_for_each_region_safe(r, next, t) + damon_destroy_region(r, t); + s->stat = (struct damos_stat){}; + s->quota.esz = 1; + s->quota.charged_sz = 0; + + r = damon_new_region(0, 4 * min_sz); + if (!r) { + damon_destroy_ctx(c); + kunit_skip(test, "region alloc fail"); + } + damon_add_region(r, t); + + damos_apply_scheme(c, t, r, s); + KUNIT_EXPECT_EQ(test, s->stat.nr_tried, 1); + KUNIT_EXPECT_EQ(test, s->stat.sz_tried, 0); + KUNIT_EXPECT_EQ(test, r->ar.end, 4 * min_sz); + + damon_destroy_ctx(c); +} + static void damon_test_feed_loop_next_input(struct kunit *test) { unsigned long last_input = 900000, current_score = 200; @@ -1959,6 +2272,9 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damon_test_commit_ctx), KUNIT_CASE(damon_test_valid_probe_params), KUNIT_CASE(damos_test_filter_out), + KUNIT_CASE(damos_test_apply_scheme_filtered_sz), + KUNIT_CASE(damos_test_apply_scheme_filter_sz_unchanged), + KUNIT_CASE(damos_test_apply_scheme_quota_sz), KUNIT_CASE(damon_test_feed_loop_next_input), KUNIT_CASE(damon_test_set_filters_default_reject), KUNIT_CASE(damon_test_apply_min_nr_regions), From ef006be5b431853fde8634f5ded91d9c3b407482 Mon Sep 17 00:00:00 2001 From: Liew Rui Yan Date: Mon, 21 Sep 2026 08:15:43 -0700 Subject: [PATCH 0811/1012] mm/damon/core: skip quota score setup when the quota is full Patch series "mm/damon: improvements in efficiency, error handling, documents". Four various improvements. Patch 1 from Liew Rui Yan skips unnecessary quota setup work when relevant. Patch 2 from Xuewen Wang propagates ignored damon_call() error to sysfs users. Patch 3 from Adrian Huang (Lenovo) fixes typos in comments. Patch 4 from Karthikeyan KS clarifies zero sample interval acceptance on kernel-doc comment. This patch (of 4): In damos_adjust_quota(), the quota could already be full. In this situation, damos_adjust_quota() will still calculates quota->min_score. However, this min_score will not be used in this window, because in damon_do_apply_schemes(), damos_quota_is_full() will always returns true, preventing the scheme from being applied to any region. Therefore, add a short circuit for 'esz < min_region_sz' schemes to early return from damos_adjust_quota() before calculating min_score. Link: https://lore.kernel.org/20260921151547.78472-1-sj@kernel.org Link: https://lore.kernel.org/20260921151547.78472-2-sj@kernel.org Signed-off-by: Liew Rui Yan Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Adrian Huang Cc: Karthikeyan KS Cc: Xuewen Wang Cc: Zenghui Yu (Huawei) --- mm/damon/core.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/mm/damon/core.c b/mm/damon/core.c index 8d8b0cba1a56c9..749897846b4c2b 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3385,6 +3385,8 @@ static void damos_adjust_quota(struct damon_ctx *c, struct damos *s) damos_trace_esz(c, s, quota); } + if (damos_quota_is_full(quota, c->min_region_sz)) + return; if (!c->ops.get_scheme_score) return; From 13ff2fb9bf62ae6742e6b112e94a7b80f3b42258 Mon Sep 17 00:00:00 2001 From: Xuewen Wang Date: Mon, 21 Sep 2026 08:15:44 -0700 Subject: [PATCH 0812/1012] mm/damon/sysfs: propagate damon_call() error in turn_damon_on damon_sysfs_turn_damon_on() ignored the return value of damon_call() for the repeat call control registration and always returned the stale result of damon_start() (0 at that point). When damon_call() fails, e.g., the kdamond is already exiting, the user still gets success from the state file write while monitoring is not actually on. Save and return the damon_call() result instead. No rollback of damon_start() is needed since a failed damon_call() guarantees the context is stopped. Link: https://lore.kernel.org/20260921151547.78472-3-sj@kernel.org Signed-off-by: Xuewen Wang Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Adrian Huang Cc: Karthikeyan KS Cc: Liew Rui Yan Cc: Zenghui Yu (Huawei) --- mm/damon/sysfs.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/mm/damon/sysfs.c b/mm/damon/sysfs.c index 43519afb9eb7f3..70a4b64fb8ea9f 100644 --- a/mm/damon/sysfs.c +++ b/mm/damon/sysfs.c @@ -2607,7 +2607,8 @@ static int damon_sysfs_turn_damon_on(struct damon_sysfs_kdamond *kdamond) repeat_call_control->data = kdamond; repeat_call_control->repeat = true; repeat_call_control->dealloc_on_cancel = true; - if (damon_call(ctx, repeat_call_control)) + err = damon_call(ctx, repeat_call_control); + if (err) kfree(repeat_call_control); return err; } From db1a7184b610a988730e7fc4b6c037ce2bea69a8 Mon Sep 17 00:00:00 2001 From: "Adrian Huang (Lenovo)" Date: Mon, 21 Sep 2026 08:15:45 -0700 Subject: [PATCH 0813/1012] mm/damon: fix typos in comments Correct spelling mistakes. No functional changes. Link: https://lore.kernel.org/20260921151547.78472-4-sj@kernel.org Signed-off-by: Adrian Huang (Lenovo) Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: Zenghui Yu (Huawei) Reviewed-by: SJ Park Cc: Karthikeyan KS Cc: Liew Rui Yan Cc: Xuewen Wang --- mm/damon/core.c | 6 +++--- mm/damon/lru_sort.c | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 749897846b4c2b..733025b367457b 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -2486,7 +2486,7 @@ static bool damos_valid_target(struct damon_ctx *c, struct damon_region *r, * This function checks if a given region should be skipped or not for the * reason. If only the starting part of the region has previously charged, * this function splits the region into two so that the second one covers the - * area that not charged in the previous charge widnow, and return true. The + * area that not charged in the previous charge window, and return true. The * caller can see the second one on the next iteration of the region walk. * Note that this means the caller should use damon_for_each_region() instead * of damon_for_each_region_safe(). If damon_for_each_region_safe() is used, @@ -2652,7 +2652,7 @@ static void damos_walk_call_walk(struct damon_ctx *ctx, struct damon_target *t, * This function is called when kdamond finished applying the action of a DAMOS * scheme to all regions that eligible for the given &damos->apply_interval_us. * If every scheme of @ctx including @s now finished walking for at least one - * &damos->apply_interval_us, this function makrs the handling of the given + * &damos->apply_interval_us, this function marks the handling of the given * DAMOS walk request is done, so that damos_walk() can wake up and return. */ static void damos_walk_complete(struct damon_ctx *ctx, struct damos *s) @@ -4194,7 +4194,7 @@ static bool damon_find_system_rams_range(unsigned long *start, * This function sets the region of @t as requested by @start and @end. If the * values of @start and @end are zero, however, this function finds 'System * RAM' resources and sets the region to cover all the resource. In the latter - * case, this function saves the start and the end addresseses of the first and + * case, this function saves the start and the end addresses of the first and * the last resources in @start and @end, respectively. * * Return: 0 on success, negative error code otherwise. diff --git a/mm/damon/lru_sort.c b/mm/damon/lru_sort.c index ad8e86dd3a93e1..273efa3c913ed4 100644 --- a/mm/damon/lru_sort.c +++ b/mm/damon/lru_sort.c @@ -260,7 +260,7 @@ static int damon_lru_sort_add_filters(struct damos *hot_scheme, return -ENOMEM; damos_add_filter(hot_scheme, filter); - /* disabllow de-prioritizing young pages */ + /* disallow de-prioritizing young pages */ filter = damos_new_filter(DAMOS_FILTER_TYPE_YOUNG, true, false); if (!filter) return -ENOMEM; From 52913f549c4d84a17a987cf876289c214a9d802f Mon Sep 17 00:00:00 2001 From: Karthikeyan KS Date: Mon, 21 Sep 2026 08:15:46 -0700 Subject: [PATCH 0814/1012] mm/damon: document that a zero sample_interval is accepted damon_set_attrs() accepts sample_interval == 0. This was reported as a bug in v1 of this patch (rejecting it in damon_set_attrs()). A similar patch was already declined for the same reason: a zero interval is intentionally supported [1]. Document the behavior instead of changing it. Link: https://lore.kernel.org/20260921151547.78472-5-sj@kernel.org Link: https://lore.kernel.org/all/20260722094304.3132750-1-dayou5941@163.com/ [1] Signed-off-by: Karthikeyan KS Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Adrian Huang Cc: Liew Rui Yan Cc: Xuewen Wang Cc: Zenghui Yu (Huawei) --- include/linux/damon.h | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 836353c4ab9aab..844b175120f09b 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -836,7 +836,8 @@ struct damon_probe { /** * struct damon_attrs - Monitoring attributes for accuracy/overhead control. * - * @sample_interval: The time between access samplings. + * @sample_interval: The time between access samplings. Zero is + * accepted. * @aggr_interval: The time between monitor results aggregations. * @ops_update_interval: The time between monitoring operations updates. * @intervals_goal: Intervals auto-tuning goal. From aef0a6dc9b755fe7f251947046b9927df68f71ab Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Mon, 21 Sep 2026 05:30:26 -0700 Subject: [PATCH 0815/1012] mm: kmemleak: move the struct page scan into a helper Patch series "mm: kmemleak: batch the struct page scan". kmemleak walks the struct page array as a scan root and hands it to scan_block() one page at a time. scan_block() takes kmemleak_lock with interrupts disabled for the duration of the call, so the scanner acquires the lock once per online PFN in order to look at a single 64-byte struct page. Batching adjacent pages up to that size keeps exactly the same pages under the same maximum lock hold time, with MAX_SCAN_SIZE / pagesize fewer acquisitions. Patch 1 lifts the loop out of __kmemleak_scan() into scan_zone_pages() with no functional change. Patch 2 does the batching, which is then contained in that one function. This improves the performance due to less atomic operations, which is not a big deal on a regular machine, but, given kmemleak usually comes with extra debug options, such as PROVE_LOCKING, DEBUG_SPINLOCK, etc. For instance, measuring Meta's "debug kernel flavor" on an arm64 hosts, this improve the scan time by 20%. It is safe to get more work into scan_block(), given it has the protections, added by commit eb11f56eeca560 ("mm/kmemleak: stop the task stack scan early when interrupted") MAX_SCAN_SIZE is also not a new maximum for a single scan_block() call. kmemleak_scan_task_stacks() already hands it a whole task stack in one go. On arm64 and x86_64 THREAD_SIZE is never below 16 KiB, four times MAX_SCAN_SIZE, and it is 64 KiB on arm64 with 64K pages. This patch (of 2): The struct page scanning loop sits inline in __kmemleak_scan(), nested three levels deep and sharing the caller's "stop" variable with the zone walk around it. Move it into scan_zone_pages(), which scans one zone and returns 1 if scanning should stop. No functional change; this only makes room for changing how the pages are handed to scan_block(). Link: https://lore.kernel.org/20260921-b4-kmemleak-page-scan-v1-1-fb97d4801b3a@debian.org Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: Catalin Marinas --- mm/kmemleak.c | 57 ++++++++++++++++++++++++++++++--------------------- 1 file changed, 34 insertions(+), 23 deletions(-) diff --git a/mm/kmemleak.c b/mm/kmemleak.c index 5d0daea93c471f..fe34c5650f42fc 100644 --- a/mm/kmemleak.c +++ b/mm/kmemleak.c @@ -1855,6 +1855,39 @@ static void dedup_flush(struct xarray *dedup) } } +/* + * Scan the struct pages of a zone, skipping memory holes, pages that belong to + * another zone and pages that are not in use. Returns 1 if the scan should be + * stopped. + */ +static int scan_zone_pages(struct zone *zone) +{ + unsigned long start_pfn = zone->zone_start_pfn; + unsigned long end_pfn = zone_end_pfn(zone); + unsigned long pfn; + + for (pfn = start_pfn; pfn < end_pfn; pfn++) { + struct page *page = pfn_to_online_page(pfn); + + if (!(pfn & 63)) + cond_resched_tasks_rcu_qs(); + + if (!page) + continue; + + /* only scan pages belonging to this zone */ + if (page_zone(page) != zone) + continue; + /* only scan if page is in use */ + if (page_count(page) == 0) + continue; + if (scan_block(page, page + 1, NULL)) + return 1; + } + + return 0; +} + /* * Scan data sections and all the referenced memory blocks allocated via the * kernel's standard allocators. This function must be called with the @@ -1928,29 +1961,7 @@ static int __kmemleak_scan(bool full) */ get_online_mems(); for_each_populated_zone(zone) { - unsigned long start_pfn = zone->zone_start_pfn; - unsigned long end_pfn = zone_end_pfn(zone); - unsigned long pfn; - - for (pfn = start_pfn; pfn < end_pfn; pfn++) { - struct page *page = pfn_to_online_page(pfn); - - if (!(pfn & 63)) - cond_resched_tasks_rcu_qs(); - - if (!page) - continue; - - /* only scan pages belonging to this zone */ - if (page_zone(page) != zone) - continue; - /* only scan if page is in use */ - if (page_count(page) == 0) - continue; - stop = scan_block(page, page + 1, NULL); - if (stop) - break; - } + stop = scan_zone_pages(zone); if (stop) break; } From ec0ed48ff81f9515486a75a1d5115358f2b278ed Mon Sep 17 00:00:00 2001 From: Breno Leitao Date: Mon, 21 Sep 2026 05:30:27 -0700 Subject: [PATCH 0816/1012] mm: kmemleak: scan the struct page array in MAX_SCAN_SIZE batches scan_zone_pages() scans the struct page array one page per scan_block() call: if (scan_block(page, page + 1, NULL)) scan_block() takes kmemleak_lock with interrupts disabled for the duration of the call, so this acquires the lock once per online PFN to scan a single struct page. Gather runs of adjacent eligible struct pages and pass each run to scan_block() in one call, capped at MAX_SCAN_SIZE. The longest kmemleak_lock hold duration is now MAX_SCAN_SIZE worth of bytes. The same bound scan_large_block() already applies to the data sections and the per-CPU areas. This can make the scan faster. On an arm64 VM with 24 GiB and ~17 GiB in use, the struct page phase goes from 4390k scan_block() calls down to 69k for the same 4390k pages scanned, and the whole scan about 20% faster (on debug kernel). The win is in the per-acquisition cost, so on a kernel built without the lock debugging options the scan time is unchanged. I don't think it will be a problem doing more on scan_block(), given we have the scan_should_stop() protection. Coverage is unchanged: a temporary assertion comparing the number of eligible pages against the number actually passed to scan_block() matched on every zone of every scan, including while memory was being freed underneath the scan. Link: https://lore.kernel.org/20260921-b4-kmemleak-page-scan-v1-2-fb97d4801b3a@debian.org Signed-off-by: Breno Leitao Signed-off-by: Andrew Morton Reviewed-by: Catalin Marinas --- mm/kmemleak.c | 29 +++++++++++++++++++++-------- 1 file changed, 21 insertions(+), 8 deletions(-) diff --git a/mm/kmemleak.c b/mm/kmemleak.c index fe34c5650f42fc..20c0eb41c8730b 100644 --- a/mm/kmemleak.c +++ b/mm/kmemleak.c @@ -1862,8 +1862,11 @@ static void dedup_flush(struct xarray *dedup) */ static int scan_zone_pages(struct zone *zone) { + const unsigned int max_batch = MAX_SCAN_SIZE / sizeof(struct page); unsigned long start_pfn = zone->zone_start_pfn; unsigned long end_pfn = zone_end_pfn(zone); + struct page *first = NULL, *last = NULL; + unsigned int batch = 0; unsigned long pfn; for (pfn = start_pfn; pfn < end_pfn; pfn++) { @@ -1872,19 +1875,29 @@ static int scan_zone_pages(struct zone *zone) if (!(pfn & 63)) cond_resched_tasks_rcu_qs(); - if (!page) - continue; + /* only scan in-use pages belonging to this zone */ + if (page && (page_zone(page) != zone || + page_count(page) == 0)) + page = NULL; - /* only scan pages belonging to this zone */ - if (page_zone(page) != zone) - continue; - /* only scan if page is in use */ - if (page_count(page) == 0) + if (page && first && page == last + 1 && + batch < max_batch) { + last = page; + batch++; continue; - if (scan_block(page, page + 1, NULL)) + } + + if (first && scan_block(first, last + 1, NULL)) return 1; + + first = page; + last = page; + batch = page ? 1 : 0; } + if (first && scan_block(first, last + 1, NULL)) + return 1; + return 0; } From db96aed9b6c3012299da23b2e51e503d9f8cb2ba Mon Sep 17 00:00:00 2001 From: Ridong Chen Date: Sun, 20 Sep 2026 21:25:19 +0800 Subject: [PATCH 0817/1012] mm: vmscan: put rotation-missed folios at the LRU tail The page reclaim isolates a batch of folios from the tail of an LRU list and works on them one by one. For a suitable swap-backed folio on an async swap device, it queues the folio for writeback and, after finishing the batch, puts the folio back to the head of the original LRU list. Meanwhile the page writeback flushes the queued folios in its own, independent batches. For each folio it writes back it calls folio_rotate_reclaimable(), which tries to rotate the folio to the LRU tail. But folio_rotate_reclaimable() only takes effect once the folio has been put back by reclaim. If the async swap device is fast enough, the writeback can complete a folio while reclaim is still working on the rest of the batch that contains it. In that case the folio stays near the head and reclaim will not revisit it before wrapping around, causing a cold/hot inversion: a clean, written-back folio that should be a prime reclaim candidate is kept ahead of hotter folios. commit 359a5e1416ca ("mm: multi-gen LRU: retry folios written back while isolated") addressed this for MGLRU only. The traditional active/inactive LRU has the same problem, reported at [1]. A reproducer is available at [2]. Rather than re-reclaiming those folios (which would drop the swap cache that may still be useful for a future hit [4]), restore the rotation that was missed: when move_folios_to_lru() puts a folio back, add it to the LRU tail if it looks like it missed folio_rotate_reclaimable() (inactive, not mapped, not dirty and not under writeback). A referenced folio is left at the head so it still gets a second chance, and a folio with an unexpected reference (e.g. a GUP or speculative pin) is left at the head because it cannot be reclaimed yet anyway. A new do_rotate parameter gates this so it only applies on the reclaim put-back path (shrink_inactive_list()), not on shrink_active_list() where the list order is already deliberate. This approach was suggested by Barry Song [3]. Only the traditional LRU is handled here. MGLRU already retries such folios via its own clean-list retry pass in evict_folios(), so it is left unchanged. The same do_rotate scheme could later replace that retry pass to unify both LRUs, which is left for a follow-up. Test result with [2]: Without patch: cat memory.usage_in_bytes 1073700864 cat memory.memsw.usage_in_bytes 1413124096 free -h total used free Mem: 1.6Gi 1.2Gi 299Mi Swap: 1.0Gi 678Mi 346Mi With patch: cat memory.usage_in_bytes 1071140864 cat memory.memsw.usage_in_bytes 1413423104 free -h total used free Mem: 1.6Gi 1.2Gi 322Mi Swap: 1.0Gi 328Mi 695Mi After applying the patch, the difference between memory.memsw.usage_in_bytes and memory.usage_in_bytes is close to the swap "used" value reported by 'free -h'. Link: https://lore.kernel.org/20260920132519.3369946-1-ridong.chen@linux.dev Link: https://lore.kernel.org/linux-kernel/20241010081802.290893-1-chenridong@huaweicloud.com/ [1] Link: https://lore.kernel.org/lkml/46037a37-4cf6-448e-a94b-30a4d16e8814@linux.dev/ [2] Link: https://lore.kernel.org/linux-mm/20260911121341.178028-1-alex@ghiti.fr/ [4] Signed-off-by: Ridong Chen Signed-off-by: Andrew Morton Suggested-by: Barry Song Link: https://lore.kernel.org/lkml/CAGsJ_4zwP3_+EYY5Ug9EJ+yD1UdxsBSGr25u8s1K3u_i7LH3Zg@mail.gmail.com/ [3] Reviewed-by: Barry Song Reviewed-by: Baolin Wang Cc: Johannes Weiner Cc: David Hildenbrand Cc: Michal Hocko Cc: Qi Zheng Cc: Shakeel Butt Cc: Lorenzo Stoakes Cc: Kairui Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He --- mm/vmscan.c | 24 ++++++++++++++++++------ 1 file changed, 18 insertions(+), 6 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index e200ce3eb056b1..91295070ca336c 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -1971,7 +1971,7 @@ static bool too_many_isolated(struct pglist_data *pgdat, int file, * * Note: The caller must not hold any lruvec lock. */ -static unsigned int move_folios_to_lru(struct list_head *list) +static unsigned int move_folios_to_lru(struct list_head *list, bool do_rotate) { int nr_pages, nr_moved = 0; struct lruvec *lruvec = NULL; @@ -2018,7 +2018,19 @@ static unsigned int move_folios_to_lru(struct list_head *list) continue; } - lruvec_add_folio(lruvec, folio); + /* + * Put clean, unreferenced and unpinned folios that may have + * missed folio_rotate_reclaimable() at the tail to avoid + * cold/hot inversion. + */ + if (do_rotate && !folio_test_active(folio) && !folio_mapped(folio) && + !folio_test_dirty(folio) && !folio_test_writeback(folio) && + !folio_test_referenced(folio) && + folio_ref_count(folio) == folio_expected_ref_count(folio)) + lruvec_add_folio_tail(lruvec, folio); + else + lruvec_add_folio(lruvec, folio); + nr_pages = folio_nr_pages(folio); nr_moved += nr_pages; if (folio_test_active(folio)) @@ -2135,7 +2147,7 @@ static unsigned long shrink_inactive_list(unsigned long nr_to_scan, nr_reclaimed = shrink_folio_list(&folio_list, pgdat, sc, &stat, false, lruvec_memcg(lruvec)); - move_folios_to_lru(&folio_list); + move_folios_to_lru(&folio_list, true); mod_lruvec_state(lruvec, PGDEMOTE_KSWAPD + reclaimer_offset(sc), stat.nr_demoted); @@ -2246,8 +2258,8 @@ static void shrink_active_list(unsigned long nr_to_scan, /* * Move folios back to the lru list. */ - nr_activate = move_folios_to_lru(&l_active); - nr_deactivate = move_folios_to_lru(&l_inactive); + nr_activate = move_folios_to_lru(&l_active, false); + nr_deactivate = move_folios_to_lru(&l_inactive, false); count_vm_events(PGDEACTIVATE, nr_deactivate); count_memcg_events(lruvec_memcg(lruvec), PGDEACTIVATE, nr_deactivate); @@ -5115,7 +5127,7 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec, folio_set_active(folio); } - move_folios_to_lru(&list); + move_folios_to_lru(&list, false); walk = current->reclaim_state->mm_walk; if (walk && walk->batched) { From 27a398b2406bfd3f721cd338147ff26d1e2002e2 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Wed, 23 Sep 2026 10:05:15 +0800 Subject: [PATCH 0818/1012] mm/hugetlb: account migration target folio in per-node NR_HUGETLB vmstat Patch series "mm: fix hugetlb NR_HUGETLB accounting on folio migration", v2. The hugeTLB counters added by 05d4532b60e3 ("memcg/hugetlb: add hugeTLB counters to memcg") are maintained in two per-node places: - the per-node vmstat counter NR_HUGETLB, exposed as nr_hugetlb in /proc/vmstat; - the per-node memcg lruvec stat, exposed via memory.numa_stat. Both are accounted against the folio's node, and both drift when a hugetlb folio is migrated, though in different ways. A migration target folio is allocated by alloc_hugetlb_folio_nodemask() and inherits the old folio's state without ever being accounted, while the old folio is freed right after and its free is accounted. That alone loses vmstat accounting: the target node has no matching increment for the decrement on the old node, so /proc/vmstat's nr_hugetlb shrinks by nr_pages per migration. Patch 1 accounts the folio where it is obtained, so the increment pairs with the free in free_huge_folio() on the successful as well as the failed migration path. The memfd page cache preallocation helper has the same asymmetry and is fixed in the same patch. The per-node lruvec stat breaks differently. mem_cgroup_migrate() moves the charge to the new folio and drops the old folio's memcg data, so the old folio's free right after migration skips the memcg per-node decrement; the count stays attributed to the old node for the rest of the charge's life, and the target folio never gets an increment on its new node. Patch 2 moves that per-node accounting along with the charge, in move_hugetlb_state(). This patch (of 2): The NR_HUGETLB vmstat counter is maintained per folio's node: incremented when a huge page is handed to a user via hugetlb_alloc_folio() and decremented when it is returned to the pool via free_huge_folio(). A folio obtained by alloc_hugetlb_folio_nodemask() never goes through hugetlb_alloc_folio(), so it is never accounted, while its free always is. For a migration target this means the target node gets no matching increment for the decrement on the old node, so the global nr_hugetlb in /proc/vmstat drops by nr_pages for each migration. The same asymmetry affects the failed migration path, which frees the target again right away, and the temporary folio hugetlb_mfill_atomic_pte() takes from the same helper. alloc_hugetlb_folio_reserve(), used to preallocate the memfd page cache folios, has the same asymmetry: the folio is handed to a user without being accounted, while its free is accounted through free_huge_folio(). Account the folio where it is obtained, in alloc_hugetlb_folio_nodemask() and alloc_hugetlb_folio_reserve(), so that the increment pairs with the decrement in free_huge_folio(): a successful migration hands the folio to a user, a failed one frees it again. Link: https://lore.kernel.org/20260923-for-hugetlb_state3-v2-0-e8a36245bfab@kylinos.cn Link: https://lore.kernel.org/20260923-for-hugetlb_state3-v2-1-e8a36245bfab@kylinos.cn Fixes: 05d4532b60e3 ("memcg/hugetlb: add hugeTLB counters to memcg") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Tested-by: Joshua Hahn Reviewed-by: Joshua Hahn Acked-by: Muchun Song Acked-by: Oscar Salvador Cc: David Hildenbrand Cc: Shakeel Butt Cc: Michal Hocko Cc: Johannes Weiner Cc: Nhat Pham Cc: Michal Hocko Cc: Roman Gushchin Cc: Chris Down Cc: --- mm/hugetlb.c | 35 +++++++++++++++++++++++------------ 1 file changed, 23 insertions(+), 12 deletions(-) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index da980377d35339..519c30b338a8b0 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -2206,6 +2206,11 @@ struct folio *alloc_hugetlb_folio_reserve(struct hstate *h, int preferred_nid, } spin_unlock_irq(&hugetlb_lock); + + if (folio) + lruvec_stat_mod_folio(folio, NR_HUGETLB, + folio_nr_pages(folio)); + return folio; } @@ -2213,24 +2218,30 @@ struct folio *alloc_hugetlb_folio_reserve(struct hstate *h, int preferred_nid, struct folio *alloc_hugetlb_folio_nodemask(struct hstate *h, int preferred_nid, nodemask_t *nmask, gfp_t gfp_mask, bool allow_alloc_fallback) { - spin_lock_irq(&hugetlb_lock); - if (available_huge_pages(h)) { - struct folio *folio; + struct folio *folio = NULL; + spin_lock_irq(&hugetlb_lock); + if (available_huge_pages(h)) folio = dequeue_hugetlb_folio_nodemask(h, gfp_mask, preferred_nid, nmask); - if (folio) { - spin_unlock_irq(&hugetlb_lock); - return folio; - } - } spin_unlock_irq(&hugetlb_lock); - /* We cannot fallback to other nodes, as we could break the per-node pool. */ - if (!allow_alloc_fallback) - gfp_mask |= __GFP_THISNODE; + if (!folio) { + /* + * We cannot fallback to other nodes, as we could break the + * per-node pool. + */ + if (!allow_alloc_fallback) + gfp_mask |= __GFP_THISNODE; - return alloc_migrate_hugetlb_folio(h, gfp_mask, preferred_nid, nmask); + folio = alloc_migrate_hugetlb_folio(h, gfp_mask, preferred_nid, + nmask); + } + + if (folio) + lruvec_stat_mod_folio(folio, NR_HUGETLB, folio_nr_pages(folio)); + + return folio; } static nodemask_t *policy_mbind_nodemask(gfp_t gfp) From 60bf0e29b02ab973014d5477843e53444ccc1b23 Mon Sep 17 00:00:00 2001 From: Hongfu Li Date: Wed, 23 Sep 2026 10:05:16 +0800 Subject: [PATCH 0819/1012] mm/memcg: migrate per-node hugetlb lruvec stat together with hugetlb folio memory.numa_stat exposes per-node hugetlb counters from per-node lruvec stats. These stats are accounted against folio_nid(): incremented on the folio's node when handed to a user, decremented when the folio is returned to the pool. During hugetlb folio migration, mem_cgroup_migrate() moves the charge to the new folio and drops the memcg data of the old one, so the free of the old folio right after the migration skips the memcg per-node lruvec decrement. The hugetlb count stays attributed to the old node for the rest of the life of the charge, while the target folio gets no increment on the new node; its later free decrements a counter that was never incremented. Migrate the per-node lruvec accounting alongside migration. Global memcg totals remain balanced because they track resource consumption, not node placement. Link: https://lore.kernel.org/20260923-for-hugetlb_state3-v2-2-e8a36245bfab@kylinos.cn Fixes: 05d4532b60e3 ("memcg/hugetlb: add hugeTLB counters to memcg") Signed-off-by: Hongfu Li Signed-off-by: Andrew Morton Tested-by: Joshua Hahn Reviewed-by: Joshua Hahn Reviewed-by: Oscar Salvador Acked-by: Muchun Song Cc: David Hildenbrand Cc: Shakeel Butt Cc: Michal Hocko Cc: Johannes Weiner Cc: Nhat Pham Cc: Michal Hocko Cc: Roman Gushchin Cc: Chris Down Cc: --- include/linux/memcontrol.h | 8 ++++++++ mm/hugetlb.c | 25 +++++++++++++++++++++++++ mm/memcontrol.c | 5 ++--- 3 files changed, 35 insertions(+), 3 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 64c183be8cbfe7..cd223f60b3a245 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -983,6 +983,9 @@ unsigned long lruvec_page_state_monotonic(const struct lruvec *lruvec, unsigned long lruvec_page_state_local(const struct lruvec *lruvec, enum node_stat_item idx); +void mod_memcg_lruvec_state(struct lruvec *lruvec, + enum node_stat_item idx, int val); + void mem_cgroup_flush_stats(struct mem_cgroup *memcg); void mem_cgroup_flush_stats_ratelimited(struct mem_cgroup *memcg); @@ -1451,6 +1454,11 @@ static inline unsigned long lruvec_page_state_local(const struct lruvec *lruvec, return node_page_state(lruvec_pgdat(lruvec), idx); } +static inline void mod_memcg_lruvec_state(struct lruvec *lruvec, + enum node_stat_item idx, int val) +{ +} + static inline void mem_cgroup_flush_stats(struct mem_cgroup *memcg) { } diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 519c30b338a8b0..76d019594b39c3 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -23,6 +23,7 @@ #include #include #include +#include #include #include #include @@ -7378,12 +7379,36 @@ void folio_putback_hugetlb(struct folio *folio) folio_put(folio); } +static void move_hugetlb_lruvec_stat(struct folio *old_folio, + struct folio *new_folio) +{ + struct mem_cgroup *memcg; + long nr_pages = folio_nr_pages(old_folio); + int old_nid = folio_nid(old_folio); + int new_nid = folio_nid(new_folio); + + if (old_nid == new_nid) + return; + + guard(rcu)(); + + memcg = folio_memcg(new_folio); + if (!memcg) + return; + + mod_memcg_lruvec_state(mem_cgroup_lruvec(memcg, NODE_DATA(old_nid)), + NR_HUGETLB, -nr_pages); + mod_memcg_lruvec_state(mem_cgroup_lruvec(memcg, NODE_DATA(new_nid)), + NR_HUGETLB, nr_pages); +} + void move_hugetlb_state(struct folio *old_folio, struct folio *new_folio, enum migrate_reason reason) { struct hstate *h = folio_hstate(old_folio); hugetlb_cgroup_migrate(old_folio, new_folio); + move_hugetlb_lruvec_stat(old_folio, new_folio); folio_set_owner_migrate_reason(new_folio, reason); /* diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 4d00748c8a5b87..f2ad4de8d80569 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -1015,9 +1015,8 @@ static void __mod_memcg_lruvec_state(struct mem_cgroup_per_node *pn, put_cpu(); } -static void mod_memcg_lruvec_state(struct lruvec *lruvec, - enum node_stat_item idx, - int val) +void mod_memcg_lruvec_state(struct lruvec *lruvec, + enum node_stat_item idx, int val) { struct pglist_data *pgdat = lruvec_pgdat(lruvec); struct mem_cgroup_per_node *pn; From 144a9ee963a9816d451ed01d2c1635ae7c07c1b7 Mon Sep 17 00:00:00 2001 From: Zhenghui Hao Date: Wed, 23 Sep 2026 10:28:58 +0800 Subject: [PATCH 0820/1012] hugetlbfs: fix stale comment in hugetlbfs_file_mmap() The comment above the VMA flag setup in hugetlbfs_file_mmap() says the VMA address alignment has already been checked by prepare_hugepage_range(). That function no longer exists: it was removed by commit eff41389d824 ("mm/hugetlb: remove prepare_hugepage_range()"), and the check now lives in hugetlb_get_unmapped_area(). The comment also mentions ia64, which was removed from the tree by commit cf8e8658100d ("arch: Remove Itanium (IA-64) architecture"). Point the comment at the current function and drop the ia64 reference. The rest of the comment, which tells future edits to add any error return only after VM_HUGETLB has been set, is still accurate and is left untouched. No functional change. Link: https://lore.kernel.org/tencent_AA61551D1F50E61F46C8542F5C8874512C05@qq.com Signed-off-by: Zhenghui Hao Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Oscar Salvador Cc: David Hildenbrand --- fs/hugetlbfs/inode.c | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/fs/hugetlbfs/inode.c b/fs/hugetlbfs/inode.c index ba7097d5720c07..f643855d2acc70 100644 --- a/fs/hugetlbfs/inode.c +++ b/fs/hugetlbfs/inode.c @@ -106,11 +106,10 @@ static int hugetlbfs_file_mmap(struct file *file, struct vm_area_struct *vma) /* * vma address alignment (but not the pgoff alignment) has - * already been checked by prepare_hugepage_range. If you add - * any error returns here, do so after setting VM_HUGETLB, so - * vma_is_hugetlb tests below unmap_region go the right - * way when do_mmap unwinds (may be important on powerpc - * and ia64). + * already been checked by hugetlb_get_unmapped_area(). If you + * add any error returns here, do so after setting VM_HUGETLB, + * so vma_is_hugetlb tests below unmap_region go the right + * way when do_mmap unwinds (may be important on powerpc). */ vma_set_flags(vma, VMA_HUGETLB_BIT, VMA_DONTEXPAND_BIT); vma->vm_ops = &hugetlb_vm_ops; From 9a783a39568eea1708143c0d2b2bcb9947f34cf0 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Tue, 29 Sep 2026 11:33:18 +0100 Subject: [PATCH 0821/1012] kselftest: mm: return fail when child test result is fail in khugepaged Patch series "kselftest: mm: fix intermittent failure khugepaged test", v4. There are intermittent failures in collapse_max_ptes_swap() and collapse_max_ptes_shared() when using the khugepaged_context: # Run test: collapse_max_ptes_shared (khugepaged:anon) # Allocate huge page... OK # Share huge page over fork()... OK # Trigger CoW on page 1023 of 2048... OK # Maybe collapse with max_ptes_shared exceeded.... OK # Trigger CoW on page 1024 of 2048... Fail Bail out! Unexpected huge page # Planned tests != run tests (26 != 23) # Totals: pass:23 fail:0 xfail:0 xpass:0 skip:0 error:0 # Run test: collapse_max_ptes_swap (khugepaged:anon) # Swapout 257 of 2048 pages... OK # Maybe collapse with max_ptes_swap exceeded.... OK # Swapout 256 of 2048 pages... OK Bail out! Unexpected huge page # Planned tests != run tests (26 != 17) # Totals: pass:17 fail:0 xfail:0 xpass:0 skip:0 error:0 This happens because khugepaged may collapse the pages before wait_for_scan() is called, causing a sanity check that expects uncollapsed pages to fail. For example, in collapse_max_ptes_swap(), after faulting the pages back in and paging out up to max_ptes_swap pages, khugepaged may collapse them again before c->collapse() is called. To prevent this, mark the VMA with MADV_NOHUGEPAGE after it has been collapsed by wait_for_scan() for anon. This prevents khugepaged from collapsing it again before c->collapse() is called. Also, fix false-positive results when a child process fails in tests such as collapse_fork*() or collapse_max_ptes_shared(): # ------------------------- # running ./khugepaged -s 2 # ------------------------- # # Run test: collapse_max_ptes_shared (khugepaged:anon) # Allocate huge page... OK # Share huge page over fork()... OK # Trigger CoW on page 1023 of 2048... OK # Maybe collapse with max_ptes_shared exceeded.... OK # Trigger CoW on page 1024 of 2048... Fail Bail out! Unexpected huge page # Planned tests != run tests (26 != 23) # Totals: pass:23 fail:0 xfail:0 xpass:0 skip:0 error:0 // child failed. # Check if parent still has huge page... OK // parent hpage success ok 24 collapse_max_ptes_shared // considered as success ... # Totals: pass:26 fail:0 xfail:0 xpass:0 skip:0 error:0 This failure was observed on NVIDIA Spark with 16KB page. This patch (of 2): Although the child process in collapse_fork*() or collapse_max_ptes_shared() reports `KSFT_FAIL`, the result is ignored because the test only checks whether the parent's page was collapsed into a huge page. As a result, the test is considered successful whenever the parent's page is a huge page, even if the child test fails, as shown below: # # Run test: collapse_max_ptes_shared (khugepaged:anon) # Allocate huge page... OK # Share huge page over fork()... OK # Trigger CoW on page 1023 of 2048... OK # Maybe collapse with max_ptes_shared exceeded.... OK # Trigger CoW on page 1024 of 2048... Fail Bail out! Unexpected huge page # Planned tests != run tests (26 != 23) # Totals: pass:23 fail:0 xfail:0 xpass:0 skip:0 error:0 // child failed. # Check if parent still has huge page... OK // parent hpage success ok 24 collapse_max_ptes_shared // considered as success ... # Totals: pass:26 fail:0 xfail:0 xpass:0 skip:0 error:0 To address this, propagate the child's failure and skip the subsequent check in the parent. Link: https://lore.kernel.org/20260929-fix_khugepagd_fail-v4-0-2169c18f2576@arm.com Link: https://lore.kernel.org/20260929-fix_khugepagd_fail-v4-1-2169c18f2576@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Reviewed-by: Baolin Wang Reviewed-by: Gregory Price (Meta) Reviewed-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Acked-by: Zi Yan Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Kiryl Shutsemau Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan --- tools/testing/selftests/mm/khugepaged.c | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index a2ac3b3ca5def3..2aa7c919715848 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -1053,6 +1053,8 @@ static void collapse_fork(struct collapse_context *c, struct mem_ops *ops) wait(&wstatus); exit_status = WEXITSTATUS(wstatus); + if (exit_status == KSFT_FAIL) + goto out; ksft_print_msg("Check if parent still has small page..."); if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size)) @@ -1060,6 +1062,7 @@ static void collapse_fork(struct collapse_context *c, struct mem_ops *ops) else fail("Fail"); validate_memory(p, 0, page_size); +out: ops->cleanup_area(p, hpage_pmd_size); ksft_test_result_report(exit_status, "%s\n", __func__); } @@ -1100,6 +1103,8 @@ static void collapse_fork_compound(struct collapse_context *c, struct mem_ops *o wait(&wstatus); exit_status = WEXITSTATUS(wstatus); + if (exit_status == KSFT_FAIL) + goto out; ksft_print_msg("Check if parent still has huge page..."); if (ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size)) @@ -1107,6 +1112,7 @@ static void collapse_fork_compound(struct collapse_context *c, struct mem_ops *o else fail("Fail"); validate_memory(p, 0, hpage_pmd_size); +out: ops->cleanup_area(p, hpage_pmd_size); ksft_test_result_report(exit_status, "%s\n", __func__); } @@ -1158,6 +1164,8 @@ static void collapse_max_ptes_shared(struct collapse_context *c, struct mem_ops wait(&wstatus); exit_status = WEXITSTATUS(wstatus); + if (exit_status == KSFT_FAIL) + goto out; ksft_print_msg("Check if parent still has huge page..."); if (ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size)) @@ -1165,6 +1173,7 @@ static void collapse_max_ptes_shared(struct collapse_context *c, struct mem_ops else fail("Fail"); validate_memory(p, 0, hpage_pmd_size); +out: ops->cleanup_area(p, hpage_pmd_size); ksft_test_result_report(exit_status, "%s\n", __func__); } From d0adc317188052d8eeebc0f67e3c8933afb3d247 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Tue, 29 Sep 2026 11:33:19 +0100 Subject: [PATCH 0822/1012] kselftest: mm: fix intermittent failure khugepaged test There are intermittent failures in collapse_max_ptes_swap() and collapse_max_ptes_shared() when using the khugepaged_context: // while running ./khugepaged -s 2 # Run test: collapse_max_ptes_shared (khugepaged:anon) # Allocate huge page... OK # Share huge page over fork()... OK # Trigger CoW on page 1023 of 2048... OK # Maybe collapse with max_ptes_shared exceeded.... OK # Trigger CoW on page 1024 of 2048... Fail Bail out! Unexpected huge page # Planned tests != run tests (26 != 23) # Totals: pass:23 fail:0 xfail:0 xpass:0 skip:0 error:0 # Run test: collapse_max_ptes_swap (khugepaged:anon) # Swapout 257 of 2048 pages... OK # Maybe collapse with max_ptes_swap exceeded.... OK # Swapout 256 of 2048 pages... OK Bail out! Unexpected huge page # Planned tests != run tests (26 != 17) # Totals: pass:17 fail:0 xfail:0 xpass:0 skip:0 error:0 This happens because khugepaged may collapse the pages before wait_for_scan() is called, causing a sanity check that expects uncollapsed pages to fail. For example, in collapse_max_ptes_swap(), after faulting the pages back in and paging out up to max_ptes_swap pages, khugepaged may collapse them again before c->collapse() is called. To prevent this, mark the VMA with MADV_NOHUGEPAGE after it has been collapsed by wait_for_scan() for anon. This prevents khugepaged from collapsing it again before c->collapse() is called. This failure was observed on NVIDIA Spark with 16KB page. Link: https://lore.kernel.org/20260929-fix_khugepagd_fail-v4-2-2169c18f2576@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Reviewed-by: Gregory Price (Meta) Reviewed-by: Baolin Wang Tested-by: Baolin Wang Acked-by: Lorenzo Stoakes (ARM) Cc: Zi Yan Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Kiryl Shutsemau Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: David Hildenbrand Cc: Shuah Khan --- tools/testing/selftests/mm/khugepaged.c | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 2aa7c919715848..4e888b7bf31007 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -618,6 +618,16 @@ static bool wait_for_scan(const char *msg, char *p, size_t len, usleep(TICK); } + /* + * The file and shmem tests rely on refaults to install PMD mappings + * after collapse. MADV_NOHUGEPAGE would prevent those mappings. + * + * Apply MADV_NOHUGEPAGE only to anonymous VMAs to prevent khugepaged + * from unexpectedly collapsing pages during the test. + */ + if (is_anon(ops)) + madvise(p, len, MADV_NOHUGEPAGE); + return timeout == -1; } From 9c485130701a464663ca0c4b5cd0e5eda8aff842 Mon Sep 17 00:00:00 2001 From: Bingfang Guo Date: Mon, 21 Sep 2026 14:16:24 +0800 Subject: [PATCH 0823/1012] memcg: keep swap charging under RCU protection Patch series "memcg: move memcgid refcount to objcg to unpin dying memcgs", v2. Although the dying memcg problem caused by LRU pages is fixed, I can still see many dying memcgs on some workloads that use shmem and those pages are swapped out. For example, programs populating logs to tmpfs or containers sharing data using shmem. This series binds the memcgid refcount to objcgs so dying memcgs can be freed normally in this case. The memcg private ID identifies memcgs for objects that can outlive the cgroup itself: swap entries and workingset shadows. Today the ID's refcount is embedded in the css, and every outstanding ID reference (mostly swap entries) pins the css, keeping the entire memcg alive. This causes a problem: A swapped-out page holds a memcgid reference that pins the css, so the memcg cannot be freed until the page is swapped back in and charged back to its online parent. The work done by Muchun Song and Qi Zheng already charges folios to the objcg, which is reparented to its parent when the memcg offlines. This series applies similar idea to the memcg private ID: the ID's refcount moves from the css into the objcg, and the memcgid xarray holds a reference to an objcg instead of pinning the css. When the memcg offlines, the objcg is reparented and any remaining memcgid references resolve to the ancestor, so swapped-out pages no longer pin the dying memcg and get the online parent naturally on swapin. Unbinding the ID from the memcg has three consequences the series has to deal with: 1. The ID stops pinning the memcg, so the paths that relied on the ID reference to keep the memcg alive have to hold the RCU read lock instead. (Patch 1) 2. Charge and uncharge no longer necessarily happen on the same memcg: swapout charges the folio's memcg, while the slot free resolves the nearest live ancestor. The counters are hierarchical, and the MEMCG_SWAP stat is either reparented at offline (v1) or not visible (v2), so nothing leaks. But "does this entry carry a counter charge at all" can no longer be answered from the resolved memcg: root is skipped only because root's swap is not accounted, and a non-root ID whose memcg was reparented into root still carries a charge that must be released. Patch 2 adds mem_cgroup_private_id_is_root() and makes all three swap paths decide on the ID's root status. 3. An ID can now outlive the memcg it was allocated to, so the memcg resolved from an ID is not necessarily the memcg the ID was handed out for. Callers that need exactly that memcg (list_lru, the workingset and MGLRU shadow tests) now get NULL and skip the entry, while the swap paths, which only need something to account to, get the nearest live ancestor. (Patch 4) The series is four patches. The first three are preparation that keeps today's semantics while the ID is still bound to the css. Only the last patch changes behavior. This patch (of 4): This is a preparatory work for unbinding memcgid from memcg. No functional change. The swap charging path currently drops its RCU read lock after acquiring a private ID reference. This is safe because the ID reference pins the memcg's CSS. After ID references are moved to pin objcgs, that lifetime guarantee will no longer hold. Keep the RCU read lock held while accessing the memcg for counter charging, statistics and failure handling. (This matches what __memcg1_swapout() already does). Save the private ID for swap memcg recording before dropping the RCU read lock. So the swap cluster locking remains outside the RCU read-side critical section. Link: https://lore.kernel.org/20260921-bingfangguo-memcgid-rework-v2-0-6c0637dc0edb@tencent.com Link: https://lore.kernel.org/20260921-bingfangguo-memcgid-rework-v2-1-6c0637dc0edb@tencent.com Signed-off-by: Bingfang Guo Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Johannes Weiner Cc: Michal Hocko Cc: Roman Gushchin Cc: Shakeel Butt Cc: Dave Chinner Cc: Qi Zheng Cc: Kairui Song Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Bingfang Guo --- mm/memcontrol.c | 38 +++++++++++++++++++------------------- 1 file changed, 19 insertions(+), 19 deletions(-) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index f2ad4de8d80569..059eaca173bb20 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5954,6 +5954,7 @@ int __mem_cgroup_try_charge_swap(struct folio *folio) struct page_counter *counter; struct mem_cgroup *memcg; struct obj_cgroup *objcg; + unsigned short private_id; if (do_memsw_account()) return 0; @@ -5963,30 +5964,29 @@ int __mem_cgroup_try_charge_swap(struct folio *folio) if (!objcg) return 0; - rcu_read_lock(); - memcg = obj_cgroup_memcg(objcg); - if (!folio_test_swapcache(folio)) { - memcg_memory_event(memcg, MEMCG_SWAP_FAIL); - rcu_read_unlock(); - return 0; - } + scoped_guard(rcu) { + memcg = obj_cgroup_memcg(objcg); + if (!folio_test_swapcache(folio)) { + memcg_memory_event(memcg, MEMCG_SWAP_FAIL); + return 0; + } - memcg = mem_cgroup_private_id_get_online(memcg, nr_pages); - /* memcg is pined by memcg ID. */ - rcu_read_unlock(); + memcg = mem_cgroup_private_id_get_online(memcg, nr_pages); + /* memcg is pined by memcg ID. */ + private_id = mem_cgroup_private_id(memcg); - if (!mem_cgroup_is_root(memcg) && - !page_counter_try_charge(&memcg->swap, nr_pages, &counter)) { - memcg_memory_event(memcg, MEMCG_SWAP_MAX); - memcg_memory_event(memcg, MEMCG_SWAP_FAIL); - mem_cgroup_private_id_put(memcg, nr_pages); - return -ENOMEM; + if (!mem_cgroup_is_root(memcg) && + !page_counter_try_charge(&memcg->swap, nr_pages, &counter)) { + memcg_memory_event(memcg, MEMCG_SWAP_MAX); + memcg_memory_event(memcg, MEMCG_SWAP_FAIL); + mem_cgroup_private_id_put(memcg, nr_pages); + return -ENOMEM; + } + mod_memcg_state(memcg, MEMCG_SWAP, nr_pages); } - mod_memcg_state(memcg, MEMCG_SWAP, nr_pages); ci = swap_cluster_get_and_lock(folio); - __swap_cgroup_set(ci, swp_cluster_offset(folio->swap), nr_pages, - mem_cgroup_private_id(memcg)); + __swap_cgroup_set(ci, swp_cluster_offset(folio->swap), nr_pages, private_id); swap_cluster_unlock(ci); return 0; From ba78e3729825a7323deff30bfefa095bd1d935e6 Mon Sep 17 00:00:00 2001 From: Bingfang Guo Date: Mon, 21 Sep 2026 14:16:25 +0800 Subject: [PATCH 0824/1012] memcg: base swap charge accounting on memcgid root status A private ID currently pins its original memcg, so testing whether the ID belongs to root is equivalent to testing whether the resolved memcg is root. That equivalence will no longer hold when IDs refer to objcgs. A non-root ID may resolve to root after reparenting, but its swap entries still carry counter charges inherited by root. Skipping their uncharge based on the resolved memcg would leave those charges behind. Use the private ID's root status to decide whether a swap entry carries a counter charge. A root-ID entry carries none, while a non-root-ID entry must release its charge even if its current accounting memcg has become root. This patch adds a new helper to check if the memcgid equals to that of the root memcg. For swap uncharging and v2 swap charging, simply decide whether to charge/uncharge memsw or swap counter based on the swap memcgid is root or not. For v1 swapout, don't recharge the memsw counter, just cancel the charge if the swap memcg ID refers to the root memcg. This makes memory and swap charging have similar semantics: one relys on the objcg's root status, and the other relys on the memcgid's root status. Link: https://lore.kernel.org/20260921-bingfangguo-memcgid-rework-v2-2-6c0637dc0edb@tencent.com Signed-off-by: Bingfang Guo Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Johannes Weiner Cc: Michal Hocko Cc: Roman Gushchin Cc: Shakeel Butt Cc: Dave Chinner Cc: Qi Zheng Cc: Kairui Song Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Bingfang Guo --- mm/memcontrol-v1.c | 16 ++++++++-------- mm/memcontrol-v1.h | 5 +++++ mm/memcontrol.c | 4 ++-- 3 files changed, 15 insertions(+), 10 deletions(-) diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index bf2c7d53b01b1c..ed015fdd951233 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -271,6 +271,7 @@ void __memcg1_swapout(struct folio *folio, struct swap_cluster_info *ci) struct mem_cgroup *memcg, *swap_memcg; struct obj_cgroup *objcg; unsigned int nr_entries; + unsigned short private_id; VM_WARN_ON_ONCE_FOLIO(!folio_test_swapcache(folio), folio); VM_WARN_ON_ONCE_FOLIO(!folio_test_locked(folio), folio); @@ -293,25 +294,24 @@ void __memcg1_swapout(struct folio *folio, struct swap_cluster_info *ci) /* * In case the memcg owning these pages has been offlined and doesn't * have an ID allocated to it anymore, charge the closest online - * ancestor for the swap instead and transfer the memory+swap charge. + * ancestor for the swap instead and cancel the memory+swap charge + * if the ID refers to the root memcg. */ nr_entries = folio_nr_pages(folio); swap_memcg = mem_cgroup_private_id_get_online(memcg, nr_entries); + private_id = mem_cgroup_private_id(swap_memcg); mod_memcg_state(swap_memcg, MEMCG_SWAP, nr_entries); - __swap_cgroup_set(ci, swp_cluster_offset(folio->swap), nr_entries, - mem_cgroup_private_id(swap_memcg)); + __swap_cgroup_set(ci, swp_cluster_offset(folio->swap), nr_entries, private_id); folio_unqueue_deferred_split(folio); folio->memcg_data = 0; - if (!obj_cgroup_is_root(objcg)) + if (!obj_cgroup_is_root(objcg)) { page_counter_uncharge(&memcg->memory, nr_entries); - if (memcg != swap_memcg) { - if (!mem_cgroup_is_root(swap_memcg)) - page_counter_charge(&swap_memcg->memsw, nr_entries); - page_counter_uncharge(&memcg->memsw, nr_entries); + if (mem_cgroup_private_id_is_root(private_id)) + page_counter_uncharge(&memcg->memsw, nr_entries); } /* diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index 0952b2a783e524..7286456125dbef 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -22,6 +22,11 @@ void drain_all_stock(struct mem_cgroup *root_memcg); int memory_stat_show(struct seq_file *m, void *v); +static inline bool mem_cgroup_private_id_is_root(unsigned short id) +{ + return id == mem_cgroup_private_id(root_mem_cgroup); +} + struct mem_cgroup *mem_cgroup_private_id_get_online(struct mem_cgroup *memcg, unsigned int n); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 059eaca173bb20..c43b0917851248 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -5975,7 +5975,7 @@ int __mem_cgroup_try_charge_swap(struct folio *folio) /* memcg is pined by memcg ID. */ private_id = mem_cgroup_private_id(memcg); - if (!mem_cgroup_is_root(memcg) && + if (!mem_cgroup_private_id_is_root(private_id) && !page_counter_try_charge(&memcg->swap, nr_pages, &counter)) { memcg_memory_event(memcg, MEMCG_SWAP_MAX); memcg_memory_event(memcg, MEMCG_SWAP_FAIL); @@ -6004,7 +6004,7 @@ void __mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_pages) rcu_read_lock(); memcg = mem_cgroup_from_private_id(id); if (memcg) { - if (!mem_cgroup_is_root(memcg)) { + if (!mem_cgroup_private_id_is_root(id)) { if (do_memsw_account()) page_counter_uncharge(&memcg->memsw, nr_pages); else From f9a19056a1d6f67a69b12d0bec1ec170faa26769 Mon Sep 17 00:00:00 2001 From: Bingfang Guo Date: Mon, 21 Sep 2026 14:16:26 +0800 Subject: [PATCH 0825/1012] memcg: manipulate memcg private ID references by ID This is a preparatory work for moving memcgid from memcg to objcg. Swap entries retain a private ID rather than a memcg pointer. Once private ID references are moved to objcgs, the ID can also outlive the memcg to which it was originally assigned. So it's better to make the get and put functions accept the ID itself instead of the memcg. Rename mem_cgroup_private_id_get_online() to mem_cgroup_private_id_get(), and make it return the ID only. If the memcg is already dying, the dying memcg will still be used for charging and stats accounting in v2 swap charging path. But they are hierarchical and will be reparented after offlining so it doesn't matter. Make mem_cgroup_private_id_put() take the ID and resolve the reference holder internally. Convert swap uncharge and charge rollback to release the reference using that ID. This introduces an extra xarray lookup for now, which will be removed in the final patch. Separate the online-state reference release into mem_cgroup_private_id_kill(). The offline path already has the memcg pointer and can call the underlying put helper directly. Link: https://lore.kernel.org/20260921-bingfangguo-memcgid-rework-v2-3-6c0637dc0edb@tencent.com Signed-off-by: Bingfang Guo Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Johannes Weiner Cc: Michal Hocko Cc: Roman Gushchin Cc: Shakeel Butt Cc: Dave Chinner Cc: Qi Zheng Cc: Kairui Song Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Bingfang Guo --- mm/memcontrol-v1.c | 7 +++---- mm/memcontrol-v1.h | 3 +-- mm/memcontrol.c | 32 +++++++++++++++++++++++--------- 3 files changed, 27 insertions(+), 15 deletions(-) diff --git a/mm/memcontrol-v1.c b/mm/memcontrol-v1.c index ed015fdd951233..b7f28688850715 100644 --- a/mm/memcontrol-v1.c +++ b/mm/memcontrol-v1.c @@ -268,7 +268,7 @@ void memcg1_commit_charge(struct folio *folio, struct mem_cgroup *memcg) */ void __memcg1_swapout(struct folio *folio, struct swap_cluster_info *ci) { - struct mem_cgroup *memcg, *swap_memcg; + struct mem_cgroup *memcg; struct obj_cgroup *objcg; unsigned int nr_entries; unsigned short private_id; @@ -298,9 +298,8 @@ void __memcg1_swapout(struct folio *folio, struct swap_cluster_info *ci) * if the ID refers to the root memcg. */ nr_entries = folio_nr_pages(folio); - swap_memcg = mem_cgroup_private_id_get_online(memcg, nr_entries); - private_id = mem_cgroup_private_id(swap_memcg); - mod_memcg_state(swap_memcg, MEMCG_SWAP, nr_entries); + private_id = mem_cgroup_private_id_get(memcg, nr_entries); + mod_memcg_state(memcg, MEMCG_SWAP, nr_entries); __swap_cgroup_set(ci, swp_cluster_offset(folio->swap), nr_entries, private_id); diff --git a/mm/memcontrol-v1.h b/mm/memcontrol-v1.h index 7286456125dbef..8201396571842a 100644 --- a/mm/memcontrol-v1.h +++ b/mm/memcontrol-v1.h @@ -27,8 +27,7 @@ static inline bool mem_cgroup_private_id_is_root(unsigned short id) return id == mem_cgroup_private_id(root_mem_cgroup); } -struct mem_cgroup *mem_cgroup_private_id_get_online(struct mem_cgroup *memcg, - unsigned int n); +unsigned short mem_cgroup_private_id_get(struct mem_cgroup *memcg, unsigned int n); void reparent_memcg_lruvec_state_local(struct mem_cgroup *memcg, struct mem_cgroup *parent, int idx); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index c43b0917851248..53c09a0e500e99 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4082,7 +4082,7 @@ static void mem_cgroup_private_id_remove(struct mem_cgroup *memcg) } } -static inline void mem_cgroup_private_id_put(struct mem_cgroup *memcg, unsigned int n) +static void __mem_cgroup_private_id_put(struct mem_cgroup *memcg, unsigned int n) { if (refcount_sub_and_test(n, &memcg->private_id_ref)) { mem_cgroup_private_id_remove(memcg); @@ -4092,7 +4092,22 @@ static inline void mem_cgroup_private_id_put(struct mem_cgroup *memcg, unsigned } } -struct mem_cgroup *mem_cgroup_private_id_get_online(struct mem_cgroup *memcg, unsigned int n) +static inline void mem_cgroup_private_id_put(unsigned short id, unsigned int n) +{ + struct mem_cgroup *memcg; + + lockdep_assert_in_rcu_read_lock(); + + memcg = mem_cgroup_from_private_id(id); + __mem_cgroup_private_id_put(memcg, n); +} + +static void mem_cgroup_private_id_kill(struct mem_cgroup *memcg) +{ + __mem_cgroup_private_id_put(memcg, 1); +} + +unsigned short mem_cgroup_private_id_get(struct mem_cgroup *memcg, unsigned int n) { while (!refcount_add_not_zero(n, &memcg->private_id_ref)) { /* @@ -4105,7 +4120,8 @@ struct mem_cgroup *mem_cgroup_private_id_get_online(struct mem_cgroup *memcg, un } memcg = parent_mem_cgroup(memcg); } - return memcg; + + return mem_cgroup_private_id(memcg); } /** @@ -4430,7 +4446,7 @@ static void mem_cgroup_css_offline(struct cgroup_subsys_state *css) drain_all_stock(memcg); - mem_cgroup_private_id_put(memcg, 1); + mem_cgroup_private_id_kill(memcg); } static void mem_cgroup_css_released(struct cgroup_subsys_state *css) @@ -5971,15 +5987,13 @@ int __mem_cgroup_try_charge_swap(struct folio *folio) return 0; } - memcg = mem_cgroup_private_id_get_online(memcg, nr_pages); - /* memcg is pined by memcg ID. */ - private_id = mem_cgroup_private_id(memcg); + private_id = mem_cgroup_private_id_get(memcg, nr_pages); if (!mem_cgroup_private_id_is_root(private_id) && !page_counter_try_charge(&memcg->swap, nr_pages, &counter)) { memcg_memory_event(memcg, MEMCG_SWAP_MAX); memcg_memory_event(memcg, MEMCG_SWAP_FAIL); - mem_cgroup_private_id_put(memcg, nr_pages); + mem_cgroup_private_id_put(private_id, nr_pages); return -ENOMEM; } mod_memcg_state(memcg, MEMCG_SWAP, nr_pages); @@ -6011,7 +6025,7 @@ void __mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_pages) page_counter_uncharge(&memcg->swap, nr_pages); } mod_memcg_state(memcg, MEMCG_SWAP, -nr_pages); - mem_cgroup_private_id_put(memcg, nr_pages); + mem_cgroup_private_id_put(id, nr_pages); } rcu_read_unlock(); } From da4f0be2366c1c8a6a53220b2f0983731591cf9c Mon Sep 17 00:00:00 2001 From: Bingfang Guo Date: Mon, 21 Sep 2026 14:16:27 +0800 Subject: [PATCH 0826/1012] memcg: move memcg private ID refcount to objcg The memcg private ID is used by objects that can't afford storing a whole pointer and can outlive memcgs to track the memcg (notably swap entries). The current design holds a refcount to the css, preventing the memcg from being freed. This patch unbinds the lifetime of memcgid from the memcg so it can be freed. The idea is to move the refcount of memcgid to one of the memcg's objcg. The objcg is stored in the global memcgid xarray instead and used for retrieving the online memcg from it. So swapped out pages no longer pin the dying memcg. After the change, a memcgid can refer to a non present memcg. To handle this situation, when trying to get the original memcg from the id, compare the memcgid passed in with that of the memcg, and return NULL to indicate its death if they differ. NULL checks are added for list_lru_walk_node(), workingset_test_recent() to skip dead memcgs. For MGLRU recency test, mem_cgroup_lruvec() will substitute NULL with root_mem_cgroup. In the earlier patch, an extra xarray lookup was introduced in swap uncharging path. Now that we have the objcg pointer in the function, the extra overhead can be removed by using it for putting directly. Link: https://lore.kernel.org/20260921-bingfangguo-memcgid-rework-v2-4-6c0637dc0edb@tencent.com Signed-off-by: Bingfang Guo Signed-off-by: Andrew Morton Acked-by: Muchun Song Cc: Johannes Weiner Cc: Michal Hocko Cc: Roman Gushchin Cc: Shakeel Butt Cc: Dave Chinner Cc: Qi Zheng Cc: Kairui Song Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Bingfang Guo --- include/linux/memcontrol.h | 7 ++-- mm/list_lru.c | 2 +- mm/memcontrol.c | 73 ++++++++++++++++++++++++++++---------- mm/workingset.c | 2 +- 4 files changed, 60 insertions(+), 24 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index cd223f60b3a245..74110a324f9e22 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -180,6 +180,7 @@ struct obj_cgroup { struct percpu_ref refcnt; struct mem_cgroup *memcg; atomic_t nr_charged_bytes; + refcount_t private_id_ref; union { struct list_head list; /* protected by objcg_lock */ struct rcu_head rcu; @@ -225,9 +226,6 @@ struct mem_cgroup { /* vmpressure notifications. Written on every reclaim iteration. */ struct vmpressure vmpressure; - /* Written on every swap charge and uncharge. */ - refcount_t private_id_ref; - #ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC /* MEMCG_KMEM for nmi context */ atomic_t kmem_stat; @@ -324,6 +322,9 @@ struct mem_cgroup { unsigned long zswap_max; #endif + /* The objcg holding private memcg ID. */ + struct obj_cgroup *private_id_objcg; + /* Private memcg ID. Used to ID objects that outlive the cgroup */ int private_id; diff --git a/mm/list_lru.c b/mm/list_lru.c index 8a6dd0a489e12c..7edd79113cc56d 100644 --- a/mm/list_lru.c +++ b/mm/list_lru.c @@ -428,7 +428,7 @@ unsigned long list_lru_walk_node(struct list_lru *lru, int nid, xa_for_each(&lru->xa, index, mlru) { rcu_read_lock(); memcg = mem_cgroup_from_private_id(index); - if (!mem_cgroup_tryget(memcg)) { + if (!memcg || !mem_cgroup_tryget(memcg)) { rcu_read_unlock(); continue; } diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 53c09a0e500e99..a5335da5d4257a 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -4074,6 +4074,13 @@ static void memcg_wb_domain_size_changed(struct mem_cgroup *memcg) #define MEM_CGROUP_ID_MAX ((1UL << MEM_CGROUP_ID_SHIFT) - 1) static DEFINE_XARRAY_ALLOC1(mem_cgroup_private_ids); +static inline struct obj_cgroup *obj_cgroup_from_private_id(unsigned short id) +{ + lockdep_assert_once(rcu_read_lock_held()); + + return xa_load(&mem_cgroup_private_ids, id); +} + static void mem_cgroup_private_id_remove(struct mem_cgroup *memcg) { if (memcg->private_id > 0) { @@ -4082,34 +4089,44 @@ static void mem_cgroup_private_id_remove(struct mem_cgroup *memcg) } } -static void __mem_cgroup_private_id_put(struct mem_cgroup *memcg, unsigned int n) +static void __mem_cgroup_private_id_put(struct obj_cgroup *objcg, + unsigned short id, unsigned int n) { - if (refcount_sub_and_test(n, &memcg->private_id_ref)) { - mem_cgroup_private_id_remove(memcg); + struct obj_cgroup *objcg_free; - /* Memcg ID pins CSS */ - css_put(&memcg->css); + if (refcount_sub_and_test(n, &objcg->private_id_ref)) { + objcg_free = xa_erase(&mem_cgroup_private_ids, id); + VM_WARN_ON(objcg_free != objcg); + + /* Memcg ID pins the objcg */ + obj_cgroup_put(objcg); } } static inline void mem_cgroup_private_id_put(unsigned short id, unsigned int n) { - struct mem_cgroup *memcg; + struct obj_cgroup *objcg; lockdep_assert_in_rcu_read_lock(); - memcg = mem_cgroup_from_private_id(id); - __mem_cgroup_private_id_put(memcg, n); + objcg = obj_cgroup_from_private_id(id); + __mem_cgroup_private_id_put(objcg, id, n); } static void mem_cgroup_private_id_kill(struct mem_cgroup *memcg) { - __mem_cgroup_private_id_put(memcg, 1); + __mem_cgroup_private_id_put(memcg->private_id_objcg, memcg->private_id, 1); } unsigned short mem_cgroup_private_id_get(struct mem_cgroup *memcg, unsigned int n) { - while (!refcount_add_not_zero(n, &memcg->private_id_ref)) { + struct obj_cgroup *objcg; + + lockdep_assert_once(rcu_read_lock_held()); + + objcg = memcg->private_id_objcg; + + while (!refcount_add_not_zero(n, &objcg->private_id_ref)) { /* * The root cgroup cannot be destroyed, so it's refcount must * always be >= 1. @@ -4119,6 +4136,7 @@ unsigned short mem_cgroup_private_id_get(struct mem_cgroup *memcg, unsigned int break; } memcg = parent_mem_cgroup(memcg); + objcg = memcg->private_id_objcg; } return mem_cgroup_private_id(memcg); @@ -4129,11 +4147,25 @@ unsigned short mem_cgroup_private_id_get(struct mem_cgroup *memcg, unsigned int * @id: the memcg id to look up * * Caller must hold rcu_read_lock(). + * + * @return: the memcg, or NULL if the memcg referred to is already dead. */ struct mem_cgroup *mem_cgroup_from_private_id(unsigned short id) { + struct obj_cgroup *objcg; + struct mem_cgroup *memcg; + WARN_ON_ONCE(!rcu_read_lock_held()); - return xa_load(&mem_cgroup_private_ids, id); + + objcg = obj_cgroup_from_private_id(id); + if (!objcg) + return NULL; + + memcg = obj_cgroup_memcg(objcg); + if (mem_cgroup_private_id(memcg) != id) + return NULL; + + return memcg; } struct mem_cgroup *mem_cgroup_get_from_id(u64 id) @@ -4380,9 +4412,10 @@ static int mem_cgroup_css_online(struct cgroup_subsys_state *css) FLUSH_TIME); lru_gen_online_memcg(memcg); - /* Online state pins memcg ID, memcg ID pins CSS */ - refcount_set(&memcg->private_id_ref, 1); - css_get(css); + /* CSS pins memcg ID, memcg ID pins obj cgroup */ + memcg->private_id_objcg = objcg; + refcount_set(&memcg->private_id_objcg->private_id_ref, 1); + obj_cgroup_get(memcg->private_id_objcg); /* * Ensure mem_cgroup_from_private_id() works once we're fully online. @@ -4394,7 +4427,7 @@ static int mem_cgroup_css_online(struct cgroup_subsys_state *css) * publish it here at the end of onlining. This matches the * regular ID destruction during offlining. */ - xa_store(&mem_cgroup_private_ids, memcg->private_id, memcg, GFP_KERNEL); + xa_store(&mem_cgroup_private_ids, memcg->private_id, memcg->private_id_objcg, GFP_KERNEL); return 0; free_objcg: @@ -5832,8 +5865,6 @@ static void __init memcg_struct_check(void) memory_events_local); CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, vmpressure); - CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, - private_id_ref); #ifdef CONFIG_MEMCG_NMI_SAFETY_REQUIRES_ATOMIC CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_write_hot, kmem_stat); @@ -5878,6 +5909,8 @@ static void __init memcg_struct_check(void) CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, zswap_writeback); #endif + CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, + private_id_objcg); CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, private_id); CACHELINE_ASSERT_GROUP_MEMBER(struct mem_cgroup, memcg_read_mostly, @@ -6013,10 +6046,12 @@ int __mem_cgroup_try_charge_swap(struct folio *folio) */ void __mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_pages) { + struct obj_cgroup *objcg; struct mem_cgroup *memcg; rcu_read_lock(); - memcg = mem_cgroup_from_private_id(id); + objcg = obj_cgroup_from_private_id(id); + memcg = obj_cgroup_memcg(objcg); if (memcg) { if (!mem_cgroup_private_id_is_root(id)) { if (do_memsw_account()) @@ -6025,7 +6060,7 @@ void __mem_cgroup_uncharge_swap(unsigned short id, unsigned int nr_pages) page_counter_uncharge(&memcg->swap, nr_pages); } mod_memcg_state(memcg, MEMCG_SWAP, -nr_pages); - mem_cgroup_private_id_put(id, nr_pages); + __mem_cgroup_private_id_put(objcg, id, nr_pages); } rcu_read_unlock(); } diff --git a/mm/workingset.c b/mm/workingset.c index 8412f4840ae35c..1504f91cdca5f6 100644 --- a/mm/workingset.c +++ b/mm/workingset.c @@ -470,7 +470,7 @@ bool workingset_test_recent(void *shadow, bool file, bool *workingset, * configurations instead. */ eviction_memcg = mem_cgroup_from_private_id(memcgid); - if (!mem_cgroup_tryget(eviction_memcg)) + if (eviction_memcg && !mem_cgroup_tryget(eviction_memcg)) eviction_memcg = NULL; rcu_read_unlock(); From 613a0b55b6c41c0e8ae50106e560f08cf4da6751 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:53 +0200 Subject: [PATCH 0827/1012] mm/sparse: move mem_section init to sparse_extreme_init() Patch series "mm/sparse: remove SECTION_MARKED_PRESENT and further cleanups", v2. SECTION_MARKED_PRESENT is really only needed during boot, where we can instead just rely on SECTION_IS_EARLY_BIT by setting that flag earlier. Some preparations for doing that conversion and some cleanups in the same (sparse) area. This patch (of 13): Let's just avoid another pair of ifdef inside a function. While at it, switch to INTERNODE_CACHE_BYTES by just defining a fallback in cache.h as well, given that the x86 variant already provides one. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-0-54d81d65e125@kernel.org Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-1-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Reviewed-by: Zi Yan Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- include/linux/cache.h | 1 + mm/sparse.c | 19 ++++++++++++------- 2 files changed, 13 insertions(+), 7 deletions(-) diff --git a/include/linux/cache.h b/include/linux/cache.h index e69768f50d5327..b6e857b985bcaa 100644 --- a/include/linux/cache.h +++ b/include/linux/cache.h @@ -89,6 +89,7 @@ */ #ifndef INTERNODE_CACHE_SHIFT #define INTERNODE_CACHE_SHIFT L1_CACHE_SHIFT +#define INTERNODE_CACHE_BYTES (1 << INTERNODE_CACHE_SHIFT) #endif #if !defined(____cacheline_internodealigned_in_smp) diff --git a/mm/sparse.c b/mm/sparse.c index b75921c622edef..8cd2b06b231ac0 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -103,11 +103,22 @@ int __meminit sparse_index_init(unsigned long section_nr, int nid) return 0; } + +static void __init sparse_extreme_init(void) +{ + const unsigned long size = sizeof(struct mem_section *) * NR_SECTION_ROOTS; + + mem_section = memblock_alloc_or_panic(size, INTERNODE_CACHE_BYTES); +} #else /* !SPARSEMEM_EXTREME */ int __meminit sparse_index_init(unsigned long section_nr, int nid) { return 0; } + +static void __init sparse_extreme_init(void) +{ +} #endif /* @@ -196,13 +207,7 @@ void __init sparse_sections_init(void) unsigned long start, end; int i, nid; -#ifdef CONFIG_SPARSEMEM_EXTREME - unsigned long size, align; - - size = sizeof(struct mem_section *) * NR_SECTION_ROOTS; - align = 1 << (INTERNODE_CACHE_SHIFT); - mem_section = memblock_alloc_or_panic(size, align); -#endif + sparse_extreme_init(); for_each_mem_pfn_range(i, MAX_NUMNODES, &start, &end, &nid) memory_present(nid, start, end); From df2323530e5288dbac3da3693e13bf13042e5559 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:54 +0200 Subject: [PATCH 0828/1012] mm/sparse: refactor sparse_sections_init() memory_present() really identifies+prepares all early sections so the initialization in sparse_init() can properly iterating them to initialize metadata. Let's just inline memory_present() into sparse_sections_init() and cleaning up the code a bit while at it: make it clear that we are operating on pfns. Note that we call set_section_nid() now only if the section was not already created earlier. Now, there is no more inconsistency between what we (temporarily) store in ms->section_mem_map and what we store in our section->nid array. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-2-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/sparse.c | 41 +++++++++++++++++------------------------ 1 file changed, 17 insertions(+), 24 deletions(-) diff --git a/mm/sparse.c b/mm/sparse.c index 8cd2b06b231ac0..2b285653abf1e8 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -179,40 +179,33 @@ static inline unsigned long first_present_section_nr(void) return next_present_section_nr(-1); } -/* Record a memory area against a node. */ -static void __init memory_present(int nid, unsigned long start, unsigned long end) +void __init sparse_sections_init(void) { - unsigned long pfn; + unsigned long pfn, start_pfn, end_pfn; + int i, nid; + + sparse_extreme_init(); - start &= PAGE_SECTION_MASK; - mminit_validate_memmodel_limits(&start, &end); - for (pfn = start; pfn < end; pfn += PAGES_PER_SECTION) { - unsigned long section_nr = pfn_to_section_nr(pfn); - struct mem_section *ms; + for_each_mem_pfn_range(i, MAX_NUMNODES, &start_pfn, &end_pfn, &nid) { + start_pfn &= PAGE_SECTION_MASK; + mminit_validate_memmodel_limits(&start_pfn, &end_pfn); - sparse_index_init(section_nr, nid); - set_section_nid(section_nr, nid); + for (pfn = start_pfn; pfn < end_pfn; pfn += PAGES_PER_SECTION) { + unsigned long section_nr = pfn_to_section_nr(pfn); + struct mem_section *ms; - ms = __nr_to_section(section_nr); - if (!ms->section_mem_map) { + sparse_index_init(section_nr, nid); + ms = __nr_to_section(section_nr); + if (ms->section_mem_map) + continue; + + set_section_nid(section_nr, nid); ms->section_mem_map = sparse_encode_early_nid(nid) | SECTION_IS_ONLINE; __section_mark_present(ms, section_nr); } } } - -void __init sparse_sections_init(void) -{ - unsigned long start, end; - int i, nid; - - sparse_extreme_init(); - - for_each_mem_pfn_range(i, MAX_NUMNODES, &start, &end, &nid) - memory_present(nid, start, end); -} - #ifndef CONFIG_SPARSEMEM_VMEMMAP struct page __init *__populate_section_memmap(unsigned long pfn, unsigned long nr_pages, int nid, struct vmem_altmap *altmap, From d1d93bb26ebe6ddf24766095124d552918ec9adb Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:55 +0200 Subject: [PATCH 0829/1012] mm/sparse: move initialization of section metadata to sparse_metadata_init() Let's move the code responsible for initializing sparse metadata (usemap, memmap) into a helper. Cleanup the variable while at it (e.g., "map_count"). Drop the rather obvious code comments. No functional change intended. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-3-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/sparse.c | 43 ++++++++++++++++++++++--------------------- 1 file changed, 22 insertions(+), 21 deletions(-) diff --git a/mm/sparse.c b/mm/sparse.c index 2b285653abf1e8..71288daeb36653 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -255,38 +255,39 @@ static void __init sparse_init_nid(int nid, unsigned long pnum_begin, } } +static void __init sparse_metadata_init(void) +{ + unsigned long start_section_nr = first_present_section_nr(); + int nid_begin = sparse_early_nid(__nr_to_section(start_section_nr)); + unsigned long section_nr, nr_sections = 1; + + for_each_present_section_nr(start_section_nr + 1, section_nr) { + const int nid = sparse_early_nid(__nr_to_section(section_nr)); + + if (nid == nid_begin) { + nr_sections++; + continue; + } + sparse_init_nid(nid_begin, start_section_nr, section_nr, nr_sections); + nid_begin = nid; + start_section_nr = section_nr; + nr_sections = 1; + } + sparse_init_nid(nid_begin, start_section_nr, section_nr, nr_sections); +} + /* * Allocate the accumulated non-linear sections, allocate a mem_map * for each and record the physical to section mapping. */ void __init sparse_init(void) { - unsigned long pnum_end, pnum_begin, map_count = 1; - int nid_begin; - if (compound_info_has_mask()) { VM_WARN_ON_ONCE(!IS_ALIGNED((unsigned long) pfn_to_page(0), MAX_FOLIO_VMEMMAP_ALIGN)); } - pnum_begin = first_present_section_nr(); - nid_begin = sparse_early_nid(__nr_to_section(pnum_begin)); - - for_each_present_section_nr(pnum_begin + 1, pnum_end) { - int nid = sparse_early_nid(__nr_to_section(pnum_end)); - - if (nid == nid_begin) { - map_count++; - continue; - } - /* Init node with sections in range [pnum_begin, pnum_end) */ - sparse_init_nid(nid_begin, pnum_begin, pnum_end, map_count); - nid_begin = nid; - pnum_begin = pnum_end; - map_count = 1; - } - /* cover the last node */ - sparse_init_nid(nid_begin, pnum_begin, pnum_end, map_count); + sparse_metadata_init(); sparse_init_subsection_map(); vmemmap_populate_print_last(); } From 9de28305875a88c52726be1900c7e3f324f48b41 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:56 +0200 Subject: [PATCH 0830/1012] mm/sparse: rename and cleanup sparse_init_nid() Let's rename it to "sparse_metadata_init_nid", avoid the "pnum" terminology and drop the function comment. Further, rename the "map" variable to "mem_map" for consistency with other functions. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-4-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/sparse.c | 41 ++++++++++++++++++++--------------------- 1 file changed, 20 insertions(+), 21 deletions(-) diff --git a/mm/sparse.c b/mm/sparse.c index 71288daeb36653..3c19d475590906 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -221,36 +221,33 @@ void __weak __meminit vmemmap_populate_print_last(void) { } -/* - * Initialize sparse on a specific node. The node spans [pnum_begin, pnum_end) - * And number of present sections in this node is map_count. - */ -static void __init sparse_init_nid(int nid, unsigned long pnum_begin, - unsigned long pnum_end, - unsigned long map_count) +static void __init sparse_metadata_init_nid(int nid, + unsigned long start_section_nr, unsigned long end_section_nr, + unsigned long nr_sections) { - unsigned long pnum; struct mem_section_usage *usage; + unsigned long section_nr; - usage = memblock_alloc_node(map_count * mem_section_usage_size(), + usage = memblock_alloc_node(nr_sections * mem_section_usage_size(), SMP_CACHE_BYTES, nid); if (!usage) panic("Failed to allocate usemap for node %d\n", nid); - for_each_present_section_nr(pnum_begin, pnum) { - unsigned long pfn = section_nr_to_pfn(pnum); - struct page *map; + for_each_present_section_nr(start_section_nr, section_nr) { + const unsigned long pfn = section_nr_to_pfn(section_nr); + struct page *mem_map; - if (pnum >= pnum_end) + if (section_nr >= end_section_nr) break; - map = __populate_section_memmap(pfn, PAGES_PER_SECTION, - nid, NULL, NULL); - if (!map) - panic("Failed to allocate memmap for section %lu\n", pnum); + mem_map = __populate_section_memmap(pfn, PAGES_PER_SECTION, nid, + NULL, NULL); + if (!mem_map) + panic("Failed to allocate memmap for section %lu\n", + section_nr); memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION)); - sparse_init_one_section(__nr_to_section(pnum), pnum, map, usage, - SECTION_IS_EARLY); + sparse_init_one_section(__nr_to_section(section_nr), section_nr, + mem_map, usage, SECTION_IS_EARLY); usage = (void *)usage + mem_section_usage_size(); } } @@ -268,12 +265,14 @@ static void __init sparse_metadata_init(void) nr_sections++; continue; } - sparse_init_nid(nid_begin, start_section_nr, section_nr, nr_sections); + sparse_metadata_init_nid(nid_begin, start_section_nr, + section_nr, nr_sections); nid_begin = nid; start_section_nr = section_nr; nr_sections = 1; } - sparse_init_nid(nid_begin, start_section_nr, section_nr, nr_sections); + sparse_metadata_init_nid(nid_begin, start_section_nr, section_nr, + nr_sections); } /* From e1f3426182719f9f37fba8307f321a854a432b20 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:57 +0200 Subject: [PATCH 0831/1012] mm/sparse: cleanup sparse_init_one_section() Let's avoid the "pnum" terminology. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-5-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/sparse.h | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/mm/sparse.h b/mm/sparse.h index 530692cdd516fe..4ee507c34b98d5 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -20,7 +20,7 @@ void sparse_sections_init(void); int sparse_index_init(unsigned long section_nr, int nid); static inline void sparse_init_one_section(struct mem_section *ms, - unsigned long pnum, struct page *mem_map, + unsigned long section_nr, struct page *mem_map, struct mem_section_usage *usage, unsigned long flags) { unsigned long coded_mem_map; @@ -32,7 +32,7 @@ static inline void sparse_init_one_section(struct mem_section *ms, * page_to_pfn() on !CONFIG_SPARSEMEM_VMEMMAP can simply subtract it * from the page pointer to obtain the PFN. */ - coded_mem_map = (unsigned long)(mem_map - section_nr_to_pfn(pnum)); + coded_mem_map = (unsigned long)(mem_map - section_nr_to_pfn(section_nr)); VM_WARN_ON_ONCE(coded_mem_map & ~SECTION_MAP_MASK); ms->section_mem_map &= ~SECTION_MAP_MASK; From 0f33d61d0203b6dca30b82d4e9b26802d5b67379 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:58 +0200 Subject: [PATCH 0832/1012] mm/sparse: rename __highest_present_section_nr to __highest_used_section_nr In preparation for getting rid of SECTION_MARKED_PRESENT, rename __highest_present_section_nr and clarify the comment. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-6-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- include/linux/mmzone.h | 6 +++--- mm/compaction.c | 2 +- mm/sparse.c | 12 ++++-------- mm/sparse.h | 4 ++-- 4 files changed, 10 insertions(+), 14 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 8e4e0bda3b586b..0555936f654db8 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2158,7 +2158,7 @@ static inline struct mem_section *__pfn_to_section(unsigned long pfn) return __nr_to_section(pfn_to_section_nr(pfn)); } -extern unsigned long __highest_present_section_nr; +extern unsigned long __highest_used_section_nr; static inline int subsection_map_index(unsigned long pfn) { @@ -2257,7 +2257,7 @@ static inline unsigned long first_valid_pfn(unsigned long pfn, unsigned long end rcu_read_lock_sched(); - while (nr <= __highest_present_section_nr && pfn < end_pfn) { + while (nr <= __highest_used_section_nr && pfn < end_pfn) { struct mem_section *ms = __pfn_to_section(pfn); if (valid_section(ms) && @@ -2312,7 +2312,7 @@ static inline int pfn_in_present_section(unsigned long pfn) static inline unsigned long next_present_section_nr(unsigned long section_nr) { - while (++section_nr <= __highest_present_section_nr) { + while (++section_nr <= __highest_used_section_nr) { if (present_section_nr(section_nr)) return section_nr; } diff --git a/mm/compaction.c b/mm/compaction.c index 4994e200bbecd7..f1b2060eb20168 100644 --- a/mm/compaction.c +++ b/mm/compaction.c @@ -216,7 +216,7 @@ static unsigned long skip_offline_sections(unsigned long start_pfn) if (online_section_nr(start_nr)) return 0; - while (++start_nr <= __highest_present_section_nr) { + while (++start_nr <= __highest_used_section_nr) { if (online_section_nr(start_nr)) return section_nr_to_pfn(start_nr); } diff --git a/mm/sparse.c b/mm/sparse.c index 3c19d475590906..bb89017254f4da 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -164,15 +164,11 @@ static void __init mminit_validate_memmodel_limits(unsigned long *start_pfn, } /* - * There are a number of times that we loop over NR_MEM_SECTIONS, - * looking for section_present() on each. But, when we have very - * large physical address spaces, NR_MEM_SECTIONS can also be - * very large which makes the loops quite long. - * - * Keeping track of this gives us an easy way to break out of - * those loops early. + * Looping over all possible memory sections is expensive, especially if + * NR_MEM_SECTIONS is large but only a fraction is actually used. Keep track of + * the highest section number we ever used. */ -unsigned long __highest_present_section_nr; +unsigned long __highest_used_section_nr; static inline unsigned long first_present_section_nr(void) { diff --git a/mm/sparse.h b/mm/sparse.h index 4ee507c34b98d5..242b7bab0013c6 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -44,8 +44,8 @@ static inline void sparse_init_one_section(struct mem_section *ms, static inline void __section_mark_present(struct mem_section *ms, unsigned long section_nr) { - if (section_nr > __highest_present_section_nr) - __highest_present_section_nr = section_nr; + if (section_nr > __highest_used_section_nr) + __highest_used_section_nr = section_nr; ms->section_mem_map |= SECTION_MARKED_PRESENT; } From c8839970a9d5f483996e8c8441a92c8b3e91aa90 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:58:59 +0200 Subject: [PATCH 0833/1012] mm/sparse: remove pfn_in_present_section() Unused, let's remove it. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-7-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- include/linux/mmzone.h | 10 ---------- 1 file changed, 10 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 0555936f654db8..7e7aa57d2a0a6c 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2303,13 +2303,6 @@ static inline unsigned long next_valid_pfn(unsigned long pfn, unsigned long end_ #endif -static inline int pfn_in_present_section(unsigned long pfn) -{ - if (pfn_to_section_nr(pfn) >= NR_MEM_SECTIONS) - return 0; - return present_section(__pfn_to_section(pfn)); -} - static inline unsigned long next_present_section_nr(unsigned long section_nr) { while (++section_nr <= __highest_used_section_nr) { @@ -2339,9 +2332,6 @@ static inline unsigned long next_present_section_nr(unsigned long section_nr) #else #define pfn_to_nid(pfn) (0) #endif - -#else -#define pfn_in_present_section pfn_valid #endif /* CONFIG_SPARSEMEM */ /* From 3ef842bf0f8f83ae5931da8b17574ac26ff4f83a Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:59:00 +0200 Subject: [PATCH 0834/1012] mm/sparse: move __highest_used_section_nr handling In preparation for removing __section_mark_present(), let's move __highest_used_section_nr handling into its callers. Verify in sparse_init_one_section() that it was properly updated. In sparse_sections_init() we can just set it to the last processed section_nr. No functional change intended. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-8-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Oscar Salvador Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/sparse-vmemmap.c | 1 + mm/sparse.c | 5 +++-- mm/sparse.h | 4 +--- 3 files changed, 5 insertions(+), 5 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 66de04f8863bfb..74ec396c503cb2 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -898,6 +898,7 @@ int __meminit sparse_add_section(int nid, unsigned long start_pfn, page_init_poison(memmap, sizeof(struct page) * nr_pages); __section_mark_present(ms, section_nr); + __highest_used_section_nr = max(section_nr, __highest_used_section_nr); /* Align memmap to section boundary in the subsection case */ if (section_nr_to_pfn(section_nr) != start_pfn) diff --git a/mm/sparse.c b/mm/sparse.c index bb89017254f4da..a0f50ca5acf5a5 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -177,7 +177,7 @@ static inline unsigned long first_present_section_nr(void) void __init sparse_sections_init(void) { - unsigned long pfn, start_pfn, end_pfn; + unsigned long pfn, start_pfn, end_pfn, section_nr; int i, nid; sparse_extreme_init(); @@ -187,9 +187,9 @@ void __init sparse_sections_init(void) mminit_validate_memmodel_limits(&start_pfn, &end_pfn); for (pfn = start_pfn; pfn < end_pfn; pfn += PAGES_PER_SECTION) { - unsigned long section_nr = pfn_to_section_nr(pfn); struct mem_section *ms; + section_nr = pfn_to_section_nr(pfn); sparse_index_init(section_nr, nid); ms = __nr_to_section(section_nr); if (ms->section_mem_map) @@ -201,6 +201,7 @@ void __init sparse_sections_init(void) __section_mark_present(ms, section_nr); } } + __highest_used_section_nr = section_nr; } #ifndef CONFIG_SPARSEMEM_VMEMMAP struct page __init *__populate_section_memmap(unsigned long pfn, diff --git a/mm/sparse.h b/mm/sparse.h index 242b7bab0013c6..c396d5c05cfc8f 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -26,6 +26,7 @@ static inline void sparse_init_one_section(struct mem_section *ms, unsigned long coded_mem_map; BUILD_BUG_ON(SECTION_MAP_LAST_BIT > PFN_SECTION_SHIFT); + VM_WARN_ON_ONCE(section_nr > __highest_used_section_nr); /* * We encode the start PFN of the section into the mem_map such that @@ -44,9 +45,6 @@ static inline void sparse_init_one_section(struct mem_section *ms, static inline void __section_mark_present(struct mem_section *ms, unsigned long section_nr) { - if (section_nr > __highest_used_section_nr) - __highest_used_section_nr = section_nr; - ms->section_mem_map |= SECTION_MARKED_PRESENT; } From ef5258c1cfbd03edd953517e85ed98322d3d668d Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:59:01 +0200 Subject: [PATCH 0835/1012] scripts/gdb: mm.py: remove fallbacks for SECTION_HAS_MEM_MAP and SECTION_IS_EARLY There is no reason to handle the absence of the corresponding BIT values. So let's remove the fallbacks. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-9-54d81d65e125@kernel.org Link: https://lore.kernel.org/r/vpz5zqg5nwbpolxbkip45vxxdt3lni7j2haal7q42skc3b57tu@uh4cec4x32dq Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Cc: Seongjun Hong Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Oscar Salvador Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- scripts/gdb/linux/mm.py | 8 ++------ 1 file changed, 2 insertions(+), 6 deletions(-) diff --git a/scripts/gdb/linux/mm.py b/scripts/gdb/linux/mm.py index 193a88d763abf7..7398105ff3f677 100644 --- a/scripts/gdb/linux/mm.py +++ b/scripts/gdb/linux/mm.py @@ -71,12 +71,8 @@ def __init__(self): self.NR_SECTION_ROOTS = DIV_ROUND_UP(self.NR_MEM_SECTIONS, self.SECTIONS_PER_ROOT) - try: - self.SECTION_HAS_MEM_MAP = 1 << int(gdb.parse_and_eval('SECTION_HAS_MEM_MAP_BIT')) - self.SECTION_IS_EARLY = 1 << int(gdb.parse_and_eval('SECTION_IS_EARLY_BIT')) - except: - self.SECTION_HAS_MEM_MAP = 1 << 0 - self.SECTION_IS_EARLY = 1 << 3 + self.SECTION_HAS_MEM_MAP = 1 << int(gdb.parse_and_eval('SECTION_HAS_MEM_MAP_BIT')) + self.SECTION_IS_EARLY = 1 << int(gdb.parse_and_eval('SECTION_IS_EARLY_BIT')) self.SUBSECTION_SHIFT = 21 self.PAGES_PER_SUBSECTION = 1 << (self.SUBSECTION_SHIFT - self.PAGE_SHIFT) From 51276f26acc1b904c835d995a5c5406a342ea2ec Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:59:02 +0200 Subject: [PATCH 0836/1012] mm/sparse: remove SECTION_MARKED_PRESENT All present section iterators run before memory hotplug added any further memory sections, Therefore, we can simply use the SECTION_IS_EARLY flag by setting that flag earlier in sparse_sections_init(). Get rid of SECTION_MARKED_PRESENT entirely and rename for_each_present_section_nr() to for_each_early_section_nr(). Take care of the .clang-format for_each_present_section_nr() handling. No functional change intended. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-10-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- .clang-format | 2 +- drivers/base/memory.c | 2 +- include/linux/mmzone.h | 27 ++++++++++----------------- mm/sparse-vmemmap.c | 1 - mm/sparse.c | 17 ++++++++--------- mm/sparse.h | 6 ------ 6 files changed, 20 insertions(+), 35 deletions(-) diff --git a/.clang-format b/.clang-format index 5ef5743b77c95f..395292ab2ac634 100644 --- a/.clang-format +++ b/.clang-format @@ -282,6 +282,7 @@ ForEachMacros: - 'for_each_dpcm_fe' - 'for_each_drhd_unit' - 'for_each_dss_dev' + - 'for_each_early_section_nr' - 'for_each_efi_memory_desc' - 'for_each_efi_memory_desc_in_map' - 'for_each_element' @@ -403,7 +404,6 @@ ForEachMacros: - 'for_each_possible_cpu_wrap' - 'for_each_present_blessed_reg' - 'for_each_present_cpu' - - 'for_each_present_section_nr' - 'for_each_prime_number' - 'for_each_prime_number_from' - 'for_each_probe_cache_entry' diff --git a/drivers/base/memory.c b/drivers/base/memory.c index 5eead3346f1e32..b0338de2f1d82c 100644 --- a/drivers/base/memory.c +++ b/drivers/base/memory.c @@ -972,7 +972,7 @@ void __init memory_dev_init(void) * block so that it can be covered. */ block_id = ULONG_MAX; - for_each_present_section_nr(0, nr) { + for_each_early_section_nr(0, nr) { if (block_id != ULONG_MAX && memory_block_id(nr) == block_id) continue; diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 7e7aa57d2a0a6c..851b14c91cb1a5 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -2062,7 +2062,6 @@ static inline struct mem_section *__nr_to_section(unsigned long nr) * accommodate SECTION_MAP_LAST_BIT. We use BUILD_BUG_ON() to ensure this. */ enum { - SECTION_MARKED_PRESENT_BIT, SECTION_HAS_MEM_MAP_BIT, SECTION_IS_ONLINE_BIT, SECTION_IS_EARLY_BIT, @@ -2072,7 +2071,6 @@ enum { SECTION_MAP_LAST_BIT, }; -#define SECTION_MARKED_PRESENT BIT(SECTION_MARKED_PRESENT_BIT) #define SECTION_HAS_MEM_MAP BIT(SECTION_HAS_MEM_MAP_BIT) #define SECTION_IS_ONLINE BIT(SECTION_IS_ONLINE_BIT) #define SECTION_IS_EARLY BIT(SECTION_IS_EARLY_BIT) @@ -2089,16 +2087,6 @@ static inline struct page *__section_mem_map_addr(struct mem_section *section) return (struct page *)map; } -static inline int present_section(const struct mem_section *section) -{ - return (section && (section->section_mem_map & SECTION_MARKED_PRESENT)); -} - -static inline int present_section_nr(unsigned long nr) -{ - return present_section(__nr_to_section(nr)); -} - static inline int valid_section(const struct mem_section *section) { return (section && (section->section_mem_map & SECTION_HAS_MEM_MAP)); @@ -2114,6 +2102,11 @@ static inline int valid_section_nr(unsigned long nr) return valid_section(__nr_to_section(nr)); } +static inline int early_section_nr(unsigned long nr) +{ + return early_section(__nr_to_section(nr)); +} + static inline int online_section(const struct mem_section *section) { return (section && (section->section_mem_map & SECTION_IS_ONLINE)); @@ -2303,20 +2296,20 @@ static inline unsigned long next_valid_pfn(unsigned long pfn, unsigned long end_ #endif -static inline unsigned long next_present_section_nr(unsigned long section_nr) +static inline unsigned long next_early_section_nr(unsigned long section_nr) { while (++section_nr <= __highest_used_section_nr) { - if (present_section_nr(section_nr)) + if (early_section_nr(section_nr)) return section_nr; } return -1; } -#define for_each_present_section_nr(start, section_nr) \ - for (section_nr = next_present_section_nr(start - 1); \ +#define for_each_early_section_nr(start, section_nr) \ + for (section_nr = next_early_section_nr(start - 1); \ section_nr != -1; \ - section_nr = next_present_section_nr(section_nr)) + section_nr = next_early_section_nr(section_nr)) /* * These are _only_ used during initialisation, therefore they diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index 74ec396c503cb2..e174a46f299abb 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -897,7 +897,6 @@ int __meminit sparse_add_section(int nid, unsigned long start_pfn, if (!section_vmemmap_optimizable(ms)) page_init_poison(memmap, sizeof(struct page) * nr_pages); - __section_mark_present(ms, section_nr); __highest_used_section_nr = max(section_nr, __highest_used_section_nr); /* Align memmap to section boundary in the subsection case */ diff --git a/mm/sparse.c b/mm/sparse.c index a0f50ca5acf5a5..5d2d9f95d4024a 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -170,13 +170,14 @@ static void __init mminit_validate_memmodel_limits(unsigned long *start_pfn, */ unsigned long __highest_used_section_nr; -static inline unsigned long first_present_section_nr(void) +static inline unsigned long first_early_section_nr(void) { - return next_present_section_nr(-1); + return next_early_section_nr(-1); } void __init sparse_sections_init(void) { + const unsigned long flags = SECTION_IS_EARLY | SECTION_IS_ONLINE; unsigned long pfn, start_pfn, end_pfn, section_nr; int i, nid; @@ -196,9 +197,7 @@ void __init sparse_sections_init(void) continue; set_section_nid(section_nr, nid); - ms->section_mem_map = sparse_encode_early_nid(nid) | - SECTION_IS_ONLINE; - __section_mark_present(ms, section_nr); + ms->section_mem_map = sparse_encode_early_nid(nid) | flags; } } __highest_used_section_nr = section_nr; @@ -230,7 +229,7 @@ static void __init sparse_metadata_init_nid(int nid, if (!usage) panic("Failed to allocate usemap for node %d\n", nid); - for_each_present_section_nr(start_section_nr, section_nr) { + for_each_early_section_nr(start_section_nr, section_nr) { const unsigned long pfn = section_nr_to_pfn(section_nr); struct page *mem_map; @@ -244,18 +243,18 @@ static void __init sparse_metadata_init_nid(int nid, section_nr); memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION)); sparse_init_one_section(__nr_to_section(section_nr), section_nr, - mem_map, usage, SECTION_IS_EARLY); + mem_map, usage, 0); usage = (void *)usage + mem_section_usage_size(); } } static void __init sparse_metadata_init(void) { - unsigned long start_section_nr = first_present_section_nr(); + unsigned long start_section_nr = first_early_section_nr(); int nid_begin = sparse_early_nid(__nr_to_section(start_section_nr)); unsigned long section_nr, nr_sections = 1; - for_each_present_section_nr(start_section_nr + 1, section_nr) { + for_each_early_section_nr(start_section_nr + 1, section_nr) { const int nid = sparse_early_nid(__nr_to_section(section_nr)); if (nid == nid_begin) { diff --git a/mm/sparse.h b/mm/sparse.h index c396d5c05cfc8f..19664d702af844 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -42,12 +42,6 @@ static inline void sparse_init_one_section(struct mem_section *ms, ms->usage = usage; } -static inline void __section_mark_present(struct mem_section *ms, - unsigned long section_nr) -{ - ms->section_mem_map |= SECTION_MARKED_PRESENT; -} - static inline size_t mem_section_usage_size(void) { return struct_size_t(struct mem_section_usage, pageblock_flags, From 1678914bd66df47cc07e4b836e8fd3bb04ee9138 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:59:03 +0200 Subject: [PATCH 0837/1012] mm/sparse: remove flags parameter from sparse_init_one_section() The last user of the flags parameter that used to pass SECTION_IS_EARLY was removed, so let's remove the parameter. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-11-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/sparse-vmemmap.c | 2 +- mm/sparse.c | 2 +- mm/sparse.h | 5 ++--- 3 files changed, 4 insertions(+), 5 deletions(-) diff --git a/mm/sparse-vmemmap.c b/mm/sparse-vmemmap.c index e174a46f299abb..111efc80246579 100644 --- a/mm/sparse-vmemmap.c +++ b/mm/sparse-vmemmap.c @@ -902,7 +902,7 @@ int __meminit sparse_add_section(int nid, unsigned long start_pfn, /* Align memmap to section boundary in the subsection case */ if (section_nr_to_pfn(section_nr) != start_pfn) memmap = pfn_to_page(section_nr_to_pfn(section_nr)); - sparse_init_one_section(ms, section_nr, memmap, ms->usage, 0); + sparse_init_one_section(ms, section_nr, memmap, ms->usage); return 0; } diff --git a/mm/sparse.c b/mm/sparse.c index 5d2d9f95d4024a..702904b4170063 100644 --- a/mm/sparse.c +++ b/mm/sparse.c @@ -243,7 +243,7 @@ static void __init sparse_metadata_init_nid(int nid, section_nr); memmap_boot_pages_add(section_nr_vmemmap_pages(pfn, PAGES_PER_SECTION)); sparse_init_one_section(__nr_to_section(section_nr), section_nr, - mem_map, usage, 0); + mem_map, usage); usage = (void *)usage + mem_section_usage_size(); } } diff --git a/mm/sparse.h b/mm/sparse.h index 19664d702af844..fdbc4cacf91d17 100644 --- a/mm/sparse.h +++ b/mm/sparse.h @@ -21,7 +21,7 @@ int sparse_index_init(unsigned long section_nr, int nid); static inline void sparse_init_one_section(struct mem_section *ms, unsigned long section_nr, struct page *mem_map, - struct mem_section_usage *usage, unsigned long flags) + struct mem_section_usage *usage) { unsigned long coded_mem_map; @@ -37,8 +37,7 @@ static inline void sparse_init_one_section(struct mem_section *ms, VM_WARN_ON_ONCE(coded_mem_map & ~SECTION_MAP_MASK); ms->section_mem_map &= ~SECTION_MAP_MASK; - ms->section_mem_map |= coded_mem_map; - ms->section_mem_map |= flags | SECTION_HAS_MEM_MAP; + ms->section_mem_map |= coded_mem_map | SECTION_HAS_MEM_MAP; ms->usage = usage; } From c95a58a52f8f6640a2c7df3fb74fee2ea7e1f7b9 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:59:04 +0200 Subject: [PATCH 0838/1012] fs/proc/page: clarify comment in get_max_dump_pfn() pfn_to_online_page() will only succeed on some PFNs within the same section, not necessarily all. Let's make that clearer. While at it, rephrase it to "Allow inspection of". Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-12-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Acked-by: Oscar Salvador Acked-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- fs/proc/page.c | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/fs/proc/page.c b/fs/proc/page.c index f90e1030825e94..6aa99e42ec2221 100644 --- a/fs/proc/page.c +++ b/fs/proc/page.c @@ -31,10 +31,9 @@ static inline unsigned long get_max_dump_pfn(void) { #ifdef CONFIG_SPARSEMEM /* - * The memmap of early sections is completely populated and marked - * online even if max_pfn does not fall on a section boundary - - * pfn_to_online_page() will succeed on all pages. Allow inspecting - * these memmaps. + * If max_pfn does not fall on a section boundary, pfn_to_online_page() + * can succeed on PFNs beyond max_pfn within the same section. Allow + * inspection of these memmaps. */ return round_up(max_pfn, PAGES_PER_SECTION); #else From f5bca71043f04b1bec8a4542801b3276feab8e4b Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 21 Sep 2026 21:59:05 +0200 Subject: [PATCH 0839/1012] mm/memory_hotplug: drop CONFIG_HAVE_ARCH_PFN_VALID handling from pfn_to_online_page() Drop CONFIG_HAVE_ARCH_PFN_VALID handling, as CONFIG_HAVE_ARCH_PFN_VALID is never used with CONFIG_MEMORY_HOTPLUG, as the latter depends on CONFIG_SPARSEMEM_VMEMMAP. Make sure it stays that way. Link: https://lore.kernel.org/20260921-b4-sparsemem_cleanups-v2-13-54d81d65e125@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Mike Rapoport (Microsoft) Acked-by: Oscar Salvador Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Kairui Song Cc: Qi Zheng Cc: Shakeel Butt Cc: Barry Song Cc: Axel Rasmussen Cc: Yuanchu Xie Cc: Wei Xu Cc: Baoquan He Cc: Baolin Wang Cc: Brendan Jackman Cc: Johannes Weiner Cc: Zi Yan Cc: Jan Kiszka Cc: Kieran Bingham Cc: Greg Kroah-Hartman Cc: Rafael J. Wysocki Cc: Danilo Krummrich --- mm/memory_hotplug.c | 9 ++------- 1 file changed, 2 insertions(+), 7 deletions(-) diff --git a/mm/memory_hotplug.c b/mm/memory_hotplug.c index d7a59167bec43c..796af1028ee239 100644 --- a/mm/memory_hotplug.c +++ b/mm/memory_hotplug.c @@ -345,6 +345,8 @@ struct page *pfn_to_online_page(unsigned long pfn) struct dev_pagemap *pgmap; struct mem_section *ms; + BUILD_BUG_ON(IS_ENABLED(CONFIG_HAVE_ARCH_PFN_VALID)); + if (nr >= NR_MEM_SECTIONS) return NULL; @@ -352,13 +354,6 @@ struct page *pfn_to_online_page(unsigned long pfn) if (!online_section(ms)) return NULL; - /* - * Save some code text when online_section() + - * pfn_section_valid() are sufficient. - */ - if (IS_ENABLED(CONFIG_HAVE_ARCH_PFN_VALID) && !pfn_valid(pfn)) - return NULL; - if (!pfn_section_valid(ms, pfn)) return NULL; From bdb137caeae0454ae4dda2b081659201ddd8fe7a Mon Sep 17 00:00:00 2001 From: Zhijian Han Date: Tue, 22 Sep 2026 11:18:43 +0800 Subject: [PATCH 0840/1012] mm: fix typos in various comments Fix spelling errors found by codespell in several mm source files: - "aray" -> "array" in include/linux/mm.h - "indiciate" -> "indicate" in include/linux/mm_types.h - "aggresive" -> "aggressive" in include/linux/mm_types.h - "progated" -> "propagated" in mm/huge_memory.c - "Pressumably" -> "Presumably" in mm/hugetlb.c - "fime" -> "time" in mm/hugetlb.c - "farest" -> "farthest" in mm/list_lru.c - "trivally" -> "trivially" in mm/madvise.c - "atmost" -> "at most" in mm/memcontrol.c - "splited" -> "split" in mm/memory-failure.c - "appliable" -> "applicable" in mm/memory.c - "incase" -> "in case" in mm/page_io.c - "contigous" -> "contiguous" in mm/vmalloc.c - "probablity" -> "probability" in mm/kfence/core.c - "possesss" -> "possesses" in mm/util.c No functional changes. Link: https://lore.kernel.org/20260922031843.2857104-1-hanzhijian1991@gmail.com Signed-off-by: Zhijian Han Signed-off-by: Andrew Morton --- include/linux/mm.h | 2 +- include/linux/mm_types.h | 4 ++-- mm/huge_memory.c | 2 +- mm/hugetlb.c | 4 ++-- mm/kfence/core.c | 2 +- mm/list_lru.c | 2 +- mm/madvise.c | 2 +- mm/memcontrol.c | 2 +- mm/memory-failure.c | 2 +- mm/memory.c | 2 +- mm/page_io.c | 2 +- mm/util.c | 2 +- mm/vmalloc.c | 2 +- 13 files changed, 15 insertions(+), 15 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 4fd47cc796a6c1..5e35864eb731c5 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4795,7 +4795,7 @@ static inline void mmap_action_simple_ioremap(struct vm_area_desc *desc, * @desc: The VMA descriptor for the VMA requiring kernel pags to be mapped. * @start: The virtual address from which to map them. * @pages: An array of struct page pointers describing the memory to map. - * @nr_pages: The number of entries in the @pages aray. + * @nr_pages: The number of entries in the @pages array. */ static inline void mmap_action_map_kernel_pages(struct vm_area_desc *desc, unsigned long start, struct page **pages, diff --git a/include/linux/mm_types.h b/include/linux/mm_types.h index 95a768fac987a5..6141160ec6526b 100644 --- a/include/linux/mm_types.h +++ b/include/linux/mm_types.h @@ -759,7 +759,7 @@ static inline struct anon_vma_name *anon_vma_name_alloc(const char *name) /* * While __vma_enter_locked() is working to ensure are no read-locks held on a * VMA (either while acquiring a VMA write lock or marking a VMA detached) we - * set the VM_REFCNT_EXCLUDE_READERS_FLAG in vma->vm_refcnt to indiciate to + * set the VM_REFCNT_EXCLUDE_READERS_FLAG in vma->vm_refcnt to indicate to * vma_start_read() that the reference count should be left alone. * * See the comment describing vm_refcnt in vm_area_struct for details as to @@ -2011,7 +2011,7 @@ enum { /* * MMF_HAS_PINNED: Whether this mm has pinned any pages. This can be either * replaced in the future by mm.pinned_vm when it becomes stable, or grow into - * a counter on its own. We're aggresive on this bit for now: even if the + * a counter on its own. We're aggressive on this bit for now: even if the * pinned pages were unpinned later on, we'll still keep this bit set for the * lifecycle of this mm, just for simplicity. */ diff --git a/mm/huge_memory.c b/mm/huge_memory.c index c5c210b3ebc646..ddc631a388b93e 100644 --- a/mm/huge_memory.c +++ b/mm/huge_memory.c @@ -3415,7 +3415,7 @@ static void __split_huge_pmd_locked(struct vm_area_struct *vma, pmd_t *pmd, swp_entry = make_readable_device_private_entry( page_to_pfn(page + i)); /* - * Young and dirty bits are not progated via swp_entry + * Young and dirty bits are not propagated via swp_entry */ entry = swp_entry_to_pte(swp_entry); if (soft_dirty) diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 76d019594b39c3..37f8272f0f2a73 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -5408,7 +5408,7 @@ void __unmap_hugepage_range(struct mmu_gather *tlb, struct vm_area_struct *vma, int rc = vma_needs_reservation(h, vma, address); if (rc < 0) - /* Pressumably allocate_file_region_entries failed + /* Presumably allocate_file_region_entries failed * to allocate a file_region struct. Clear * hugetlb_restore_reserve so that global reserve * count will not be incremented by free_huge_folio. @@ -5621,7 +5621,7 @@ static vm_fault_t hugetlb_wp(struct vm_fault *vmf) * In order to determine where this is a COW on a MAP_PRIVATE mapping it * is enough to check whether the old_folio is anonymous. This means that * the reserve for this address was consumed. If reserves were used, a - * partial faulted mapping at the fime of fork() could consume its reserves + * partial faulted mapping at the time of fork() could consume its reserves * on COW instead of the full address range. */ if (is_vma_resv_set(vma, HPAGE_RESV_OWNER) && diff --git a/mm/kfence/core.c b/mm/kfence/core.c index 90925c646c4c24..55d5c114eadf6c 100644 --- a/mm/kfence/core.c +++ b/mm/kfence/core.c @@ -156,7 +156,7 @@ atomic_t kfence_allocation_gate = ATOMIC_INIT(1); * allocations of the same source filling up the pool. * * Assuming a range of 15%-85% unique allocations in the pool at any point in - * time, the below parameters provide a probablity of 0.02-0.33 for false + * time, the below parameters provide a probability of 0.02-0.33 for false * positive hits respectively: * * P(alloc_traces) = (1 - e^(-HNUM * (alloc_traces / SIZE)) ^ HNUM diff --git a/mm/list_lru.c b/mm/list_lru.c index 7edd79113cc56d..d1d256822ccc73 100644 --- a/mm/list_lru.c +++ b/mm/list_lru.c @@ -587,7 +587,7 @@ static int __memcg_list_lru_alloc(struct mem_cgroup *memcg, */ do { /* - * Keep finding the farest parent that wasn't populated + * Keep finding the farthest parent that wasn't populated * until found memcg itself. */ pos = memcg; diff --git a/mm/madvise.c b/mm/madvise.c index 20135275cb5550..d83ce6abf8c323 100644 --- a/mm/madvise.c +++ b/mm/madvise.c @@ -1471,7 +1471,7 @@ static bool is_discard(int behavior) * We are restricted from madvise()'ing mseal()'d VMAs only in very particular * circumstances - discarding of data from read-only anonymous SEALED mappings. * - * This is because users cannot trivally discard data from these VMAs, and may + * This is because users cannot trivially discard data from these VMAs, and may * only do so via an appropriate madvise() call. */ static bool can_madvise_modify(struct madvise_behavior *madv_behavior) diff --git a/mm/memcontrol.c b/mm/memcontrol.c index a5335da5d4257a..aad0498a7bd658 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -741,7 +741,7 @@ static unsigned long *memcg_events_local_array(struct mem_cgroup *memcg) * * 2) Flush the stats synchronously on reader side only when there are more than * (MEMCG_CHARGE_BATCH * nr_cpus) update events. Though this optimization - * will let stats be out of sync by atmost (MEMCG_CHARGE_BATCH * nr_cpus) but + * will let stats be out of sync by at most (MEMCG_CHARGE_BATCH * nr_cpus) but * only for 2 seconds due to (1). */ static void flush_memcg_stats_dwork(struct work_struct *w); diff --git a/mm/memory-failure.c b/mm/memory-failure.c index d237f556b3a09e..0c96eb5119976e 100644 --- a/mm/memory-failure.c +++ b/mm/memory-failure.c @@ -2560,7 +2560,7 @@ int memory_failure(unsigned long pfn, int flags) /* * We're only intended to deal with the non-Compound page here. * The page cannot become compound pages again as folio has been - * splited and extra refcnt is held. + * split and extra refcnt is held. */ WARN_ON(folio_test_large(folio)); diff --git a/mm/memory.c b/mm/memory.c index 79fa57a381ce00..330cde31bf8b40 100644 --- a/mm/memory.c +++ b/mm/memory.c @@ -5165,7 +5165,7 @@ vm_fault_t do_swap_page(struct vm_fault *vmf) * have changed, so sub pages might got charged to the wrong cgroup, * or even should be shmem. So we have to free it and fallback. * Nothing should have touched it, both anon and shmem checks if a - * large folio is fully appliable before use. + * large folio is fully applicable before use. * * This will be removed once we unify folio allocation in the swap cache * layer, where allocation of a folio stabilizes the swap entries. diff --git a/mm/page_io.c b/mm/page_io.c index 0808808f0309ce..c6824fcd483e0a 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -135,7 +135,7 @@ static bool is_folio_zero_filled(struct folio *folio) for (i = 0; i < folio_nr_pages(folio); i++) { data = kmap_local_folio(folio, i * PAGE_SIZE); /* - * Check last word first, incase the page is zero-filled at + * Check last word first, in case the page is zero-filled at * the start and has non-zero data at the end, which is common * in real-world workloads. */ diff --git a/mm/util.c b/mm/util.c index c5ee52aede1e41..ab67cfc7357165 100644 --- a/mm/util.c +++ b/mm/util.c @@ -1252,7 +1252,7 @@ EXPORT_SYMBOL(__compat_vma_mmap); /** * compat_vma_mmap() - Apply the file's .mmap_prepare() hook to an * existing VMA and execute any requested actions. - * @file: The file which possesss an f_op->mmap_prepare() hook. + * @file: The file which possesses an f_op->mmap_prepare() hook. * @vma: The VMA to apply the .mmap_prepare() hook to. * * Ordinarily, .mmap_prepare() is invoked directly upon mmap(). However, certain diff --git a/mm/vmalloc.c b/mm/vmalloc.c index fad918765f054e..b24896aedac212 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -3698,7 +3698,7 @@ vm_area_alloc_pages(gfp_t gfp, int nid, /* * Initially, attempt to have the page allocator give us large order * pages. Do not attempt allocating smaller than order chunks since - * __vmap_pages_range() expects physically contigous pages of exactly + * __vmap_pages_range() expects physically contiguous pages of exactly * order long chunks. */ while (large_order > order && nr_remaining) { From 0c3103eafbd55a47f8c96c783681e80b42919ced Mon Sep 17 00:00:00 2001 From: Li Zhe Date: Mon, 28 Sep 2026 10:47:23 +0800 Subject: [PATCH 0841/1012] mm/hugetlb: fix overbroad MMU notifiers for unshared PMDs Hugetlb currently expands MMU notifier ranges to PUD boundaries whenever PMD sharing is possible. That is only needed when huge_pmd_unshare() actually detaches a shared PMD page table, because clearing the PUD invalidates the whole PUD-sized virtual address range. For hugetlbfs hole punch, and similarly for other hugetlb unmap paths, a shared mapping can pass the "PMD sharing is possible" range test in adjust_range_if_pmd_sharing_possible() even when the hugetlbfs file does not currently have any shared PMD page tables. KVM then receives a 1G invalidation for a 2M operation and zaps unrelated secondary mappings, so the guest has to fault them back in. Avoid this by tracking active PMD-sharing attachments per hugetlbfs inode. The count is incremented only after huge_pmd_share() successfully installs a shared PMD table, and decremented when __huge_pmd_unshare() actually detaches one. Since huge_pmd_share() can run concurrently under i_mmap_lock_read(), use a 64-bit atomic counter. A zero count is used to skip the conservative notifier range expansion only after excluding concurrent PMD sharing with the mapping write lock. On a Redis-in-VM workload that punches cold 2M hugetlb pages, this patch improves P99 QPS stability while punching pages, reducing the QPS degradation ratio from 7.09% to 1.45%. Link: https://lore.kernel.org/20260928024723.87708-1-lizhe.67@bytedance.com Signed-off-by: Li Zhe Signed-off-by: Andrew Morton Reported-by: aiqi.i7 Acked-by: Muchun Song Cc: David Hildenbrand Cc: Oscar Salvador --- fs/hugetlbfs/inode.c | 1 + include/linux/hugetlb.h | 37 +++++++++++++++++++++++++++++++++++++ mm/hugetlb.c | 13 ++++++++++--- 3 files changed, 48 insertions(+), 3 deletions(-) diff --git a/fs/hugetlbfs/inode.c b/fs/hugetlbfs/inode.c index f643855d2acc70..ab1e4dec3f77b8 100644 --- a/fs/hugetlbfs/inode.c +++ b/fs/hugetlbfs/inode.c @@ -920,6 +920,7 @@ static struct inode *hugetlbfs_get_inode(struct super_block *sb, simple_inode_init_ts(inode); info->resv_map = resv_map; info->seals = F_SEAL_SEAL; + hugetlbfs_pmd_sharing_init(inode); switch (mode & S_IFMT) { default: init_special_inode(inode, mode, dev); diff --git a/include/linux/hugetlb.h b/include/linux/hugetlb.h index 24727ece20fe52..5029c72418631b 100644 --- a/include/linux/hugetlb.h +++ b/include/linux/hugetlb.h @@ -11,6 +11,7 @@ #include #include #include +#include #include #include #include @@ -507,6 +508,9 @@ struct hugetlbfs_inode_info { struct inode vfs_inode; struct resv_map *resv_map; unsigned int seals; +#ifdef CONFIG_HUGETLB_PMD_PAGE_TABLE_SHARING + atomic64_t pmd_sharing_count; +#endif }; static inline struct hugetlbfs_inode_info *HUGETLBFS_I(struct inode *inode) @@ -514,6 +518,39 @@ static inline struct hugetlbfs_inode_info *HUGETLBFS_I(struct inode *inode) return container_of(inode, struct hugetlbfs_inode_info, vfs_inode); } +#ifdef CONFIG_HUGETLB_PMD_PAGE_TABLE_SHARING +static inline void hugetlbfs_pmd_sharing_init(struct inode *inode) +{ + atomic64_set(&HUGETLBFS_I(inode)->pmd_sharing_count, 0); +} + +static inline void hugetlbfs_pmd_sharing_inc(struct inode *inode) +{ + atomic64_inc(&HUGETLBFS_I(inode)->pmd_sharing_count); +} + +static inline void hugetlbfs_pmd_sharing_dec(struct inode *inode) +{ + atomic64_dec(&HUGETLBFS_I(inode)->pmd_sharing_count); +} + +static inline bool hugetlbfs_pmd_sharing_active(struct inode *inode) +{ + return atomic64_read(&HUGETLBFS_I(inode)->pmd_sharing_count) != 0; +} +#else +static inline void hugetlbfs_pmd_sharing_init(struct inode *inode) {} + +static inline void hugetlbfs_pmd_sharing_inc(struct inode *inode) {} + +static inline void hugetlbfs_pmd_sharing_dec(struct inode *inode) {} + +static inline bool hugetlbfs_pmd_sharing_active(struct inode *inode) +{ + return false; +} +#endif + extern const struct vm_operations_struct hugetlb_vm_ops; struct file *hugetlb_file_setup(const char *name, size_t size, vma_flags_t acct, int creat_flags, int page_size_log); diff --git a/mm/hugetlb.c b/mm/hugetlb.c index 37f8272f0f2a73..a1b51251103b36 100644 --- a/mm/hugetlb.c +++ b/mm/hugetlb.c @@ -5438,10 +5438,12 @@ void __hugetlb_zap_begin(struct vm_area_struct *vma, if (!vma->vm_file) /* hugetlbfs_file_mmap error */ return; - adjust_range_if_pmd_sharing_possible(vma, start, end); hugetlb_vma_lock_write(vma); - if (vma->vm_file) + if (vma->vm_file) { i_mmap_lock_write(vma->vm_file->f_mapping); + if (hugetlbfs_pmd_sharing_active(file_inode(vma->vm_file))) + adjust_range_if_pmd_sharing_possible(vma, start, end); + } } void __hugetlb_zap_end(struct vm_area_struct *vma, @@ -5480,7 +5482,10 @@ void unmap_hugepage_range(struct vm_area_struct *vma, unsigned long start, mmu_notifier_range_init(&range, MMU_NOTIFY_CLEAR, 0, vma->vm_mm, start, end); - adjust_range_if_pmd_sharing_possible(vma, &range.start, &range.end); + i_mmap_assert_write_locked(vma->vm_file->f_mapping); + if (hugetlbfs_pmd_sharing_active(file_inode(vma->vm_file))) + adjust_range_if_pmd_sharing_possible(vma, &range.start, + &range.end); mmu_notifier_invalidate_range_start(&range); tlb_gather_mmu(&tlb, vma->vm_mm); @@ -7083,6 +7088,7 @@ pte_t *huge_pmd_share(struct mm_struct *mm, struct vm_area_struct *vma, if (pud_none(*pud)) { pud_populate(mm, pud, (pmd_t *)((unsigned long)spte & PAGE_MASK)); + hugetlbfs_pmd_sharing_inc(file_inode(vma->vm_file)); mm_inc_nr_pmds(mm); } else { ptdesc_pmd_pts_dec(virt_to_ptdesc(spte)); @@ -7114,6 +7120,7 @@ static int __huge_pmd_unshare(struct mmu_gather *tlb, pud_clear(pud); tlb_unshare_pmd_ptdesc(tlb, virt_to_ptdesc(ptep), addr); + hugetlbfs_pmd_sharing_dec(file_inode(vma->vm_file)); mm_dec_nr_pmds(mm); return 1; From d3f988d8f78f7ae9e08ddb8e15bda3e3a50ab0b5 Mon Sep 17 00:00:00 2001 From: Park Tae-sun Date: Wed, 23 Sep 2026 18:26:48 +0900 Subject: [PATCH 0842/1012] selftests/mm: fix mlock2 errno handling and false PASS on ENOSYS While inspecting selftests/mm syscall wrappers, I noticed that mlock2_() in mlock2.h handles the syscall return value differently from other wrappers: int ret = syscall(__NR_mlock2, start, len, flags); if (ret) { errno = ret; return -1; } Commit 1ddae9d67ee1 ("selftests/mm/mlock: print error on failure") introduced this intending to make mlock2_() behave like libc by setting errno and returning -1. However, glibc syscall(2) already returns -1 on failure and sets positive errno. Assigning "errno = ret;" overwrites errno with -1. To verify this, mlock2 was disabled in the kernel (via sys_ni_syscall) to return -ENOSYS. Testing revealed two interrelated defects: 1. In the unmodified test, mlock2_() clobbered errno to -1. The check "if (ret && errno == ENOSYS)" in main() was bypassed, resulting in an immediate crash in the first test: ~ # ./mlock2-tests TAP version 13 1..15 Bail out! mlock2(0): Unknown error -1 # Planned tests != run tests (15 != 0) # Totals: pass:0 fail:0 xfail:0 xpass:0 skip:0 error:0 (exit code: 1 - FAIL) 2. After restoring mlock2_() to directly return syscall(), errno correctly retained ENOSYS (38), entering the ENOSYS check in main(). However, it then called ksft_finished(): ~ # ./mlock2-tests TAP version 13 # Totals: pass:0 fail:0 xfail:0 xpass:0 skip:0 error:0 ~ # echo $? 0 Because ksft_set_plan() had not been called yet (ksft_plan == 0) and zero tests ran (ksft_pass == 0), ksft_finished() evaluated 0 == 0 as success and exited with KSFT_PASS (code 0) without any TAP skip header. Fix both issues by: 1. Returning the syscall() result directly in mlock2_() so that errno is preserved. 2. Calling ksft_exit_skip() on ENOSYS so unsupported kernels report a TAP skip ("1..0 # SKIP ...") and exit with KSFT_SKIP (code 4). Verification on the mlock2-disabled kernel: ~ # ./mlock2-tests TAP version 13 1..0 # SKIP mlock2() syscall is not supported ~ # echo $? 4 Re-enabling mlock2 in the kernel confirmed all 15 tests pass cleanly: ~ # ./mlock2-tests TAP version 13 1..15 ok 1 test_mlock_lock: Locked ... ok 15 test_mlockall_future_droppable: droppable memory not locked # Totals: pass:15 fail:0 xfail:0 xpass:0 skip:0 error:0 ~ # echo $? 0 Link: https://lore.kernel.org/20260923-selftests-mm-mlock2-fix-v1-1-750b627854c6@dgu.ac.kr Fixes: 1ddae9d67ee1 ("selftests/mm/mlock: print error on failure") Fixes: 65c89684896d ("selftests/mm: mlock2-tests: conform test to TAP format output") Signed-off-by: Park Tae-sun Signed-off-by: Andrew Morton Reviewed-by: Gregory Price Acked-by: David Hildenbrand (Arm) Reviewed-by: Muhammad Usama Anjum Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan Cc: Brendan Jackman --- tools/testing/selftests/mm/mlock2-tests.c | 2 +- tools/testing/selftests/mm/mlock2.h | 8 +------- 2 files changed, 2 insertions(+), 8 deletions(-) diff --git a/tools/testing/selftests/mm/mlock2-tests.c b/tools/testing/selftests/mm/mlock2-tests.c index e16e288cc7c1f2..144b550813a6e1 100644 --- a/tools/testing/selftests/mm/mlock2-tests.c +++ b/tools/testing/selftests/mm/mlock2-tests.c @@ -502,7 +502,7 @@ int main(int argc, char **argv) ret = mlock2_(map, size, MLOCK_ONFAULT); if (ret && errno == ENOSYS) - ksft_finished(); + ksft_exit_skip("mlock2() syscall is not supported\n"); munmap(map, size); diff --git a/tools/testing/selftests/mm/mlock2.h b/tools/testing/selftests/mm/mlock2.h index 81e77fa41901a0..4417eaa5cfb78b 100644 --- a/tools/testing/selftests/mm/mlock2.h +++ b/tools/testing/selftests/mm/mlock2.h @@ -6,13 +6,7 @@ static int mlock2_(void *start, size_t len, int flags) { - int ret = syscall(__NR_mlock2, start, len, flags); - - if (ret) { - errno = ret; - return -1; - } - return 0; + return syscall(__NR_mlock2, start, len, flags); } static FILE *seek_to_smaps_entry(unsigned long addr) From 104a105a1c0eff5a8da5c739c6cac17a33290f10 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Thu, 24 Sep 2026 20:11:16 +0100 Subject: [PATCH 0843/1012] kselftest: mm: prevent random failure of huge page split for khugepaged MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "kselftest: mm: fix some failure of split_huge_page_test", v8. split_huge_page_test can fail for the following reasons: 1. During the test, khugepaged may collapse previously split pages again, causing intermittent failures. 2. Since glibc commit 321e1fc73f (“malloc: Enable 2MB THP by default on AArch64”), glibc may call madvise(MADV_HUGEPAGE) for sufficiently large allocations made by memalign(). The underlying VMA may start at a different address from the aligned address returned by memalign(). Moreover, a subsequent madvise(MADV_HUGEPAGE) call does not split the VMA because it already has the same advice. This causes the test to fail because the check_huge_xxx() helpers incorrectly require the address returned by memalign() to match the VMA start address reported in /proc/self/smaps. Address these issues by applying MADV_NOHUGEPAGE after faulting in the huge page, preventing khugepaged from collapsing it again, and instead of relying on /proc/self/smaps, use /proc/self/pagemap and /proc/kpageflags to detect huge-page mappings and large folios: 1. If hpage_size == pmd_pagesize, check PAGE_IS_HUGE instead of using check_large_folios(), since only the mapping type matters. This identifies PMD-mapped huge pages. 2. Otherwise, use check_large_folios() to detect large folios. This covers mTHP cases. 3. Check the folio flags according to the type of huge page. Since check_huge_shmem() was required to distinguish shmem huge pages because /proc/self/smaps reports them using a dedicated “ShmemPmdMapped” entry, as opposed to “FilePmdMapped” for file-backed huge pages. Now that /proc/self/smaps is no longer used to detect huge pages and /proc/kpageflags is used instead, it is sufficient to distinguish between file-backed and anonymous pages since the ShmemPmdMapped is also kind of file-backed. Therefore, remove check_huge_shmem() and use check_huge_file() instead. This patch (of 4): There're some random failure for split_huge_page_test when khugepaged collapses pages into pmd again which had split by the test. Prevent the khugepaged's collapses for split page by setting the mapped pmd-huge-page with MADV_NOHUGEPAGE before split. Link: https://lore.kernel.org/20260924-fix_split-v8-0-cba7359d882a@arm.com Link: https://lore.kernel.org/20260924-fix_split-v8-1-cba7359d882a@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Suggested-by: Kevin Brodsky Suggested-by: Lorenzo Stoakes (ARM) Reviewed-by: Lorenzo Stoakes (ARM) Reviewed-by: Zi Yan Acked-by: David Hildenbrand (Arm) Reviewed-by: Sarthak Sharma Cc: Baolin Wang Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan --- .../selftests/mm/split_huge_page_test.c | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/tools/testing/selftests/mm/split_huge_page_test.c b/tools/testing/selftests/mm/split_huge_page_test.c index c5d96a4b1db355..ef4058662b9144 100644 --- a/tools/testing/selftests/mm/split_huge_page_test.c +++ b/tools/testing/selftests/mm/split_huge_page_test.c @@ -108,6 +108,18 @@ static char *allocate_zero_filled_hugepage(size_t len) return result; } +static void disable_khugepaged(void *addr, size_t len) +{ + /* + * Disables khugepaged from collapsing THPs in range, existing THP + * pages remain. + */ + if (!madvise(addr, len, MADV_NOHUGEPAGE)) + return; + + ksft_exit_fail_msg("MADV_NOHUGEPAGE failed, err=%d\n", errno); +} + static void verify_rss_anon_split_huge_page_all_zeroes(char *one_page, int nr_hpages, size_t len) { unsigned long rss_anon_before, rss_anon_after; @@ -120,6 +132,8 @@ static void verify_rss_anon_split_huge_page_all_zeroes(char *one_page, int nr_hp if (!rss_anon_before) ksft_exit_fail_msg("No RssAnon is allocated before split\n"); + disable_khugepaged(one_page, len); + /* split all THPs */ write_debugfs(PID_FMT, getpid(), (uint64_t)one_page, (uint64_t)one_page + len, 0); @@ -167,6 +181,8 @@ static void split_pmd_thp_to_order(int order) if (!check_huge_anon(one_page, 4 * pmd_pagesize, 4, pmd_pagesize)) ksft_exit_fail_msg("No THP is allocated\n"); + disable_khugepaged(one_page, len); + /* split all THPs */ write_debugfs(PID_FMT, getpid(), (uint64_t)one_page, (uint64_t)one_page + len, order); @@ -215,6 +231,8 @@ static void split_pte_mapped_thp(void) goto out; } + disable_khugepaged(thp_area, thp_area_size); + /* * To challenge spitting code, we will mremap a single page of each * THP (page[i] of thp[i]) in the thp_area into page_area. This will @@ -482,6 +500,7 @@ static int create_pagecache_thp_and_fd(const char *testfile, size_t fd_size, ksft_test_result_skip("Pagecache folio split skipped\n"); return -2; } + disable_khugepaged(*addr, fd_size); return 0; err_out_close: close(*fd); From 2ee3b95e07493ff550a3265fc0e9cafe06d15b02 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Thu, 24 Sep 2026 20:11:17 +0100 Subject: [PATCH 0844/1012] kselftest: mm: replace usage of /proc/self/smaps for __check_pmd_huge() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Since glibc commit 321e1fc73f (“malloc: Enable 2MB THP by default on AArch64”), glibc may call madvise(MADV_HUGEPAGE) for sufficiently large allocations made by memalign(). The underlying VMA may start at a different address from the aligned address returned by memalign(). Furthermore, a subsequent madvise(MADV_HUGEPAGE) call does not split the VMA because the flag is already set. This causes split_huge_page_test to fail because the check_pmd_huge() helpers incorrectly require the address returned by memalign() to match the VMA start address reported in /proc/self/smaps. Instead of relying on /proc/self/smaps, use /proc/self/pagemap and /proc/kpageflags to detect huge-page mappings checking PAGE_IS_HUGE and PAGE_IS_FILE according to type of huge page. Since shmem pages are also file-backed, simply check whether the page is file-backed. Link: https://lore.kernel.org/20260924-fix_split-v8-2-cba7359d882a@arm.com Fixes: 642bc52aed9c ("selftests: vm: bring common functions to a new file") Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Acked-by: David Hildenbrand (Arm) Reviewed-by: Baolin Wang Tested-by: Baolin Wang Acked-by: Zi Yan Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan Cc: Kevin Brodsky --- tools/testing/selftests/mm/vm_util.c | 78 +++++++++++++++++----------- 1 file changed, 49 insertions(+), 29 deletions(-) diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index a0ab78ceedb130..969d7202500cfd 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -351,24 +351,6 @@ char *__get_smap_entry(void *addr, const char *pattern, char *buf, size_t len) return entry; } -static bool __check_pmd_huge(void *addr, char *pattern, int nr_hpages, - uint64_t hpage_size) -{ - char buffer[MAX_LINE_LENGTH]; - uint64_t thp = -1; - char *entry; - - entry = __get_smap_entry(addr, pattern, buffer, sizeof(buffer)); - if (!entry) - goto err_out; - - if (sscanf(entry, "%9" SCNu64 " kB", &thp) != 1) - ksft_exit_fail_msg("Reading smap error\n"); - -err_out: - return thp == (nr_hpages * (hpage_size >> 10)); -} - static bool check_large_folios(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) { @@ -410,20 +392,51 @@ static bool check_large_folios(void *addr, size_t len, int nr_hpages, return ret; } -bool check_huge_anon(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) +enum check_huge_type { + CHECK_HUGE_ANON, + CHECK_HUGE_FILE, +}; + +static bool check_huge_type(uint64_t categories, enum check_huge_type type) { - uint64_t pmd_pagesize = read_pmd_pagesize(); + const bool file = categories & PAGE_IS_FILE; - if (!pmd_pagesize) - ksft_exit_fail_msg("reading PMD pagesize failed\n"); + switch (type) { + case CHECK_HUGE_ANON: + return !file; + case CHECK_HUGE_FILE: + return file; + } - if (hpage_size == pmd_pagesize) - return __check_pmd_huge(addr, "AnonHugePages: ", nr_hpages, hpage_size); + return false; +} - return check_large_folios(addr, len, nr_hpages, hpage_size); +static bool __check_pmd_huge(void *addr, size_t len, int nr_hpages, + uint64_t hpage_size, enum check_huge_type type) +{ + int pagemap_fd; + int nr_pmd_mappings = 0; + uint64_t categories; + char *start = addr; + char *end = start + len; + + pagemap_fd = open(PAGEMAP_PATH, O_RDONLY); + if (pagemap_fd < 0) + ksft_exit_fail_msg("open pagemap fail\n"); + + for (; start < end; start += hpage_size) { + categories = pagemap_scan_get_categories(pagemap_fd, start); + if (!(categories & PAGE_IS_HUGE)) + continue; + if (check_huge_type(categories, type)) + nr_pmd_mappings++; + } + close(pagemap_fd); + + return nr_hpages == nr_pmd_mappings; } -bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) +bool check_huge_anon(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) { uint64_t pmd_pagesize = read_pmd_pagesize(); @@ -431,12 +444,13 @@ bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) ksft_exit_fail_msg("reading PMD pagesize failed\n"); if (hpage_size == pmd_pagesize) - return __check_pmd_huge(addr, "FilePmdMapped:", nr_hpages, hpage_size); + return __check_pmd_huge(addr, len, nr_hpages, hpage_size, + CHECK_HUGE_ANON); return check_large_folios(addr, len, nr_hpages, hpage_size); } -bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) +bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) { uint64_t pmd_pagesize = read_pmd_pagesize(); @@ -444,11 +458,17 @@ bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size ksft_exit_fail_msg("reading PMD pagesize failed\n"); if (hpage_size == pmd_pagesize) - return __check_pmd_huge(addr, "ShmemPmdMapped:", nr_hpages, hpage_size); + return __check_pmd_huge(addr, len, nr_hpages, hpage_size, + CHECK_HUGE_FILE); return check_large_folios(addr, len, nr_hpages, hpage_size); } +bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) +{ + return check_huge_file(addr, len, nr_hpages, hpage_size); +} + int64_t allocate_transhuge(void *ptr, int pagemap_fd) { uint64_t ent[2]; From 8cdda64831881f30305ca76456fc18fd5d314550 Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Thu, 24 Sep 2026 20:11:18 +0100 Subject: [PATCH 0845/1012] kselftest: mm: integrate huge page checks check_large_folios() only checks for large folios without distinguishing between anonymous and file-backed huge pages. To add huge page type checking, integrate the huge page checks into __check_huge(): 1. If hpage_size == pmd_pagesize, check PAGE_IS_HUGE instead of using check_large_folios(), since only the mapping type matters. This identifies PMD-mapped huge pages. 2. Otherwise, use check_large_folios() to detect large folios. This covers mTHP cases. 3. Check the folio flags according to the huge page type. Link: https://lore.kernel.org/20260924-fix_split-v8-3-cba7359d882a@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Suggested-by: Zi Yan Reviewed-by: Sarthak Sharma Reviewed-by: Baolin Wang Tested-by: Baolin Wang Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan Cc: Kevin Brodsky --- tools/testing/selftests/mm/vm_util.c | 90 +++++++++++++++------------- 1 file changed, 48 insertions(+), 42 deletions(-) diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index 969d7202500cfd..7d23038af1df51 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -351,13 +351,13 @@ char *__get_smap_entry(void *addr, const char *pattern, char *buf, size_t len) return entry; } -static bool check_large_folios(void *addr, size_t len, int nr_hpages, +static bool check_large_folios(int pagemap_fd, int kpageflags_fd, + void *addr, size_t len, int nr_hpages, uint64_t hpage_size) { int order = 0, pagesize = getpagesize(); unsigned int nr_pages = hpage_size / pagesize; int orders[MAX_NR_ORDERS], status; - int pagemap_fd, kpageflags_fd; bool ret = false; if (!nr_pages) @@ -368,15 +368,6 @@ static bool check_large_folios(void *addr, size_t len, int nr_hpages, ksft_exit_fail_msg("invalid order\n"); memset(orders, 0, sizeof(int) * MAX_NR_ORDERS); - pagemap_fd = open(PAGEMAP_PATH, O_RDONLY); - if (pagemap_fd == -1) - ksft_exit_fail_msg("read pagemap fail\n"); - - kpageflags_fd = open(KPAGEFLAGS_PATH, O_RDONLY); - if (kpageflags_fd == -1) { - close(pagemap_fd); - ksft_exit_fail_msg("read kpageflags fail\n"); - } status = gather_folio_orders(addr, len, pagemap_fd, kpageflags_fd, orders, MAX_NR_ORDERS); @@ -387,8 +378,6 @@ static bool check_large_folios(void *addr, size_t len, int nr_hpages, ret = true; out: - close(pagemap_fd); - close(kpageflags_fd); return ret; } @@ -411,57 +400,74 @@ static bool check_huge_type(uint64_t categories, enum check_huge_type type) return false; } -static bool __check_pmd_huge(void *addr, size_t len, int nr_hpages, - uint64_t hpage_size, enum check_huge_type type) +static bool __check_huge(void *addr, size_t len, int nr_hpages, + uint64_t hpage_size, enum check_huge_type type) { - int pagemap_fd; + bool ret = false; + int pagemap_fd, kpageflags_fd; int nr_pmd_mappings = 0; + uint64_t pmd_pagesize, scan_mapping_size; uint64_t categories; + unsigned long pfn; + bool check_pmd_mapping, allow_nonpresent; char *start = addr; char *end = start + len; + pmd_pagesize = read_pmd_pagesize(); + if (!pmd_pagesize) + ksft_exit_fail_msg("reading PMD pagesize failed\n"); + + check_pmd_mapping = hpage_size == pmd_pagesize; + scan_mapping_size = (nr_hpages > 0) ? hpage_size : psize(); + /* Some mTHP tests check a partially populated PMD-sized range. */ + allow_nonpresent = (uint64_t)nr_hpages * hpage_size < len; + pagemap_fd = open(PAGEMAP_PATH, O_RDONLY); if (pagemap_fd < 0) ksft_exit_fail_msg("open pagemap fail\n"); - for (; start < end; start += hpage_size) { + kpageflags_fd = open(KPAGEFLAGS_PATH, O_RDONLY); + if (kpageflags_fd < 0) + ksft_exit_fail_msg("open kpageflags fail\n"); + + if (!check_pmd_mapping && + !check_large_folios(pagemap_fd, kpageflags_fd, + addr, len, nr_hpages, hpage_size)) + goto out; + + for (; start < end; start += scan_mapping_size) { categories = pagemap_scan_get_categories(pagemap_fd, start); - if (!(categories & PAGE_IS_HUGE)) - continue; - if (check_huge_type(categories, type)) + pfn = pagemap_get_pfn(pagemap_fd, start); + if (pfn == -1UL) { + if (!allow_nonpresent) + goto out; + else + continue; + } + if (check_pmd_mapping && (categories & PAGE_IS_HUGE)) nr_pmd_mappings++; + if (!check_huge_type(categories, type)) + goto out; } - close(pagemap_fd); - return nr_hpages == nr_pmd_mappings; + if (check_pmd_mapping && (nr_pmd_mappings != nr_hpages)) + goto out; + ret = true; + +out: + close(pagemap_fd); + close(kpageflags_fd); + return ret; } bool check_huge_anon(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) { - uint64_t pmd_pagesize = read_pmd_pagesize(); - - if (!pmd_pagesize) - ksft_exit_fail_msg("reading PMD pagesize failed\n"); - - if (hpage_size == pmd_pagesize) - return __check_pmd_huge(addr, len, nr_hpages, hpage_size, - CHECK_HUGE_ANON); - - return check_large_folios(addr, len, nr_hpages, hpage_size); + return __check_huge(addr, len, nr_hpages, hpage_size, CHECK_HUGE_ANON); } bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) { - uint64_t pmd_pagesize = read_pmd_pagesize(); - - if (!pmd_pagesize) - ksft_exit_fail_msg("reading PMD pagesize failed\n"); - - if (hpage_size == pmd_pagesize) - return __check_pmd_huge(addr, len, nr_hpages, hpage_size, - CHECK_HUGE_FILE); - - return check_large_folios(addr, len, nr_hpages, hpage_size); + return __check_huge(addr, len, nr_hpages, hpage_size, CHECK_HUGE_FILE); } bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) From a9e8b9722a5a64522ad5c9aefafd4b31bee7681d Mon Sep 17 00:00:00 2001 From: Yeoreum Yun Date: Thu, 24 Sep 2026 20:11:19 +0100 Subject: [PATCH 0846/1012] kselftest: mm: remove check_huge_shmem() MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit check_huge_shmem() was required to distinguish shmem huge pages because /proc/self/smaps reports them using a dedicated “ShmemPmdMapped” entry, as opposed to “FilePmdMapped” for file-backed huge pages. Now that /proc/self/smaps is no longer used to detect huge pages and /proc/kpageflags is used instead, it is sufficient to distinguish between file-backed and anonymous pages since the ShmemPmdMapped is also kind of file-backed. Therefore, remove check_huge_shmem() and use check_huge_file() instead and cleanup khugepaged's check_huge operation in mem_ops. Link: https://lore.kernel.org/20260924-fix_split-v8-4-cba7359d882a@arm.com Signed-off-by: Yeoreum Yun Signed-off-by: Andrew Morton Suggested-by: David Hildenbrand (Arm) Reviewed-by: Baolin Wang Acked-by: Zi Yan Acked-by: Lorenzo Stoakes (ARM) Acked-by: David Hildenbrand (Arm) Reviewed-by: Sarthak Sharma Cc: Liam R. Howlett Cc: Nico Pache Cc: Ryan Roberts Cc: Dev Jain Cc: Barry Song Cc: Lance Yang Cc: Usama Arif Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Shuah Khan Cc: Kevin Brodsky --- .../selftests/mm/folio_split_race_test.c | 2 +- tools/testing/selftests/mm/khugepaged.c | 38 +++---------------- tools/testing/selftests/mm/uffd-common.c | 4 +- tools/testing/selftests/mm/vm_util.c | 5 --- tools/testing/selftests/mm/vm_util.h | 1 - 5 files changed, 9 insertions(+), 41 deletions(-) diff --git a/tools/testing/selftests/mm/folio_split_race_test.c b/tools/testing/selftests/mm/folio_split_race_test.c index e4660bf89b624a..1a9840b73e6189 100644 --- a/tools/testing/selftests/mm/folio_split_race_test.c +++ b/tools/testing/selftests/mm/folio_split_race_test.c @@ -181,7 +181,7 @@ static uint64_t run_iteration(void) for (i = 0; i < TOTAL_PAGES; i++) fill_page(mmap_base, i); - if (!check_huge_shmem(mmap_base, FILE_SIZE, NR_PMD_PAGE, pmd_pagesize)) + if (!check_huge_file(mmap_base, FILE_SIZE, NR_PMD_PAGE, pmd_pagesize)) ksft_exit_fail_msg("No shmem THP is allocated\n"); if (pthread_barrier_init(&ctl.barrier, NULL, NUM_READER_THREADS + 1) != 0) diff --git a/tools/testing/selftests/mm/khugepaged.c b/tools/testing/selftests/mm/khugepaged.c index 4e888b7bf31007..1bc7acc66a4697 100644 --- a/tools/testing/selftests/mm/khugepaged.c +++ b/tools/testing/selftests/mm/khugepaged.c @@ -57,7 +57,7 @@ struct mem_ops { void *(*setup_area)(int nr_hpages); void (*cleanup_area)(void *p, unsigned long size); void (*fault)(void *p, unsigned long start, unsigned long end); - bool (*check_huge)(void *addr, size_t len, int nr_hpages, unsigned long hpage_size); + bool (*check_huge)(void *addr, size_t len, int nr_hpages, uint64_t hpage_size); const char *name; }; @@ -343,12 +343,6 @@ static void anon_fault(void *p, unsigned long start, unsigned long end) fill_memory(p, start, end); } -static bool anon_check_huge(void *addr, size_t len, int nr_hpages, - unsigned long hpage_size) -{ - return check_huge_anon(addr, len, nr_hpages, hpage_size); -} - static void *file_setup_area_common(int nr_hpages, enum file_setup_ops setup) { const int open_opt = setup == FILE_SETUP_READ_ONLY_FS ? O_RDONLY : O_RDWR; @@ -444,20 +438,6 @@ static void file_fault_write(void *p, unsigned long start, unsigned long end) ksft_exit_fail_perror("madvise(MADV_POPULATE_WRITE)"); } -static bool file_check_huge(void *addr, size_t len, int nr_hpages, - unsigned long hpage_size) -{ - switch (finfo.type) { - case VMA_FILE: - return check_huge_file(addr, len, nr_hpages, hpage_size); - case VMA_SHMEM: - return check_huge_shmem(addr, len, nr_hpages, hpage_size); - default: - ksft_exit_fail_msg("Unknown VMA type\n"); - return false; - } -} - static void *shmem_setup_area(int nr_hpages) { void *p; @@ -481,17 +461,11 @@ static void shmem_cleanup_area(void *p, unsigned long size) close(finfo.fd); } -static bool shmem_check_huge(void *addr, size_t len, int nr_hpages, - unsigned long hpage_size) -{ - return check_huge_shmem(addr, len, nr_hpages, hpage_size); -} - static struct mem_ops __anon_ops = { .setup_area = &anon_setup_area, .cleanup_area = &anon_cleanup_area, .fault = &anon_fault, - .check_huge = &anon_check_huge, + .check_huge = &check_huge_anon, .name = "anon", }; @@ -499,7 +473,7 @@ static struct mem_ops __read_only_file_ops = { .setup_area = &file_setup_read_only_area, .cleanup_area = &file_cleanup_area, .fault = &file_fault_read, - .check_huge = &file_check_huge, + .check_huge = &check_huge_file, .name = "file", }; @@ -507,7 +481,7 @@ static struct mem_ops __read_write_file_read_ops = { .setup_area = &file_setup_read_write_fs_read_area, .cleanup_area = &file_cleanup_area, .fault = &file_fault_read_and_flush, - .check_huge = &file_check_huge, + .check_huge = &check_huge_file, .name = "file", }; @@ -515,7 +489,7 @@ static struct mem_ops __read_write_file_write_ops = { .setup_area = &file_setup_read_write_fs_write_area, .cleanup_area = &file_cleanup_area, .fault = &file_fault_write, - .check_huge = &file_check_huge, + .check_huge = &check_huge_file, .name = "file", }; @@ -523,7 +497,7 @@ static struct mem_ops __shmem_ops = { .setup_area = &shmem_setup_area, .cleanup_area = &shmem_cleanup_area, .fault = &anon_fault, - .check_huge = &shmem_check_huge, + .check_huge = &check_huge_file, .name = "shmem", }; diff --git a/tools/testing/selftests/mm/uffd-common.c b/tools/testing/selftests/mm/uffd-common.c index 1fb967ef498540..a76851aced4d0b 100644 --- a/tools/testing/selftests/mm/uffd-common.c +++ b/tools/testing/selftests/mm/uffd-common.c @@ -196,8 +196,8 @@ static void shmem_check_pmd_mapping(uffd_global_test_opts_t *gopts, void *p, int { size_t len = expect_nr_hpages * read_pmd_pagesize(); - if (!check_huge_shmem(gopts->area_dst_alias, len, expect_nr_hpages, - read_pmd_pagesize())) + if (!check_huge_file(gopts->area_dst_alias, len, expect_nr_hpages, + read_pmd_pagesize())) err("Did not find expected %d number of hugepages", expect_nr_hpages); } diff --git a/tools/testing/selftests/mm/vm_util.c b/tools/testing/selftests/mm/vm_util.c index 7d23038af1df51..65bc4761d1c9c0 100644 --- a/tools/testing/selftests/mm/vm_util.c +++ b/tools/testing/selftests/mm/vm_util.c @@ -470,11 +470,6 @@ bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) return __check_huge(addr, len, nr_hpages, hpage_size, CHECK_HUGE_FILE); } -bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size) -{ - return check_huge_file(addr, len, nr_hpages, hpage_size); -} - int64_t allocate_transhuge(void *ptr, int pagemap_fd) { uint64_t ent[2]; diff --git a/tools/testing/selftests/mm/vm_util.h b/tools/testing/selftests/mm/vm_util.h index 072a6c756c5170..b5d59729d43275 100644 --- a/tools/testing/selftests/mm/vm_util.h +++ b/tools/testing/selftests/mm/vm_util.h @@ -96,7 +96,6 @@ uint64_t read_pmd_pagesize(void); unsigned long rss_anon(void); bool check_huge_anon(void *addr, size_t len, int nr_hpages, uint64_t hpage_size); bool check_huge_file(void *addr, size_t len, int nr_hpages, uint64_t hpage_size); -bool check_huge_shmem(void *addr, size_t len, int nr_hpages, uint64_t hpage_size); int64_t allocate_transhuge(void *ptr, int pagemap_fd); int pageflags_get(unsigned long pfn, int kpageflags_fd, uint64_t *flags); int gather_folio_orders(char *vaddr_start, size_t len, From 49fbc64a98779e96754ed209d302fffe98193236 Mon Sep 17 00:00:00 2001 From: Usama Arif Date: Thu, 24 Sep 2026 12:11:03 -0700 Subject: [PATCH 0847/1012] mm: remove the unused zone->unaccepted_cleanup Commit fefc07518227 ("mm/page_alloc: fix race condition in unaccepted memory handling") removed the zones_with_unaccepted_pages static key and the work that decremented it, but left the work_struct behind in struct zone. Nothing uses it anymore, so let's remove it. Link: https://lore.kernel.org/20260924191103.3475117-1-usama.arif@linux.dev Signed-off-by: Usama Arif Signed-off-by: Andrew Morton Acked-by: David Hildenbrand (Arm) Reviewed-by: Shakeel Butt Reviewed-by: Johannes Weiner Reviewed-by: Barry Song Reviewed-by: Kiryl Shutsemau (Meta) Cc: Axel Rasmussen Cc: Baolin Wang Cc: Baoquan He Cc: Kairui Song Cc: Liam R. Howlett Cc: Lorenzo Stoakes Cc: Michal Hocko Cc: Qi Zheng Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Vlastimil Babka Cc: Wei Xu Cc: Yuanchu Xie --- include/linux/mmzone.h | 3 --- 1 file changed, 3 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 851b14c91cb1a5..3b96d6c7123b45 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -1077,9 +1077,6 @@ struct zone { #ifdef CONFIG_UNACCEPTED_MEMORY /* Pages to be accepted. All pages on the list are MAX_PAGE_ORDER */ struct list_head unaccepted_pages; - - /* To be called once the last page in the zone is accepted */ - struct work_struct unaccepted_cleanup; #endif /* zone flags, see below */ From 47d476dcbd683dde35efd04de00d0188b964637b Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Wed, 16 Sep 2026 10:19:32 +0530 Subject: [PATCH 0848/1012] arm64/mm: move __check_safe_pte_update() Patch series "arm64/mm: Standardize printing for pgtable entries", v3. Standardize printing for pgtable entries using recently introduced generic helper ptval_bytes_to_hex_str() in core MM which automatically enables 128 bits entries when added later. But first move __check_safe_pte_update() outside to avoid a cyclic dependency while accessing these afore mentioned core MM helpers defined in . This replaces the original page table entry print standardisation proposal which was part of the D128 series [1]. This patch (of 2): The page table entry print helpers and related macros which are defined in will not be accessible in platform which is basically caused by cycling dependency. Move __check_safe_pte_update() inside arch/arm64/mm/mmu.c as a preparation for subsequent usage of the afore mentioned generic MM helpers. While here drop IS_ENABLED(CONFIG_DEBUG_VM), although wrap __check_safe_pte_update() inside #ifdef CONFIG_DEBUG_VM that preserves the current code optimization which is achieved via the static inline functions. This does not cause any functional change. Link: https://lore.kernel.org/20260916044933.2689426-2-anshuman.khandual@arm.com Link: https://lore.kernel.org/linux-mm/20260729122452.3797443-11-anshuman.khandual@arm.com/ [1] Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Catalin Marinas Cc: Will Deacon Cc: Mike Rapoport Cc: Lorenzo Stoakes --- arch/arm64/include/asm/pgtable.h | 49 ++++---------------------------- arch/arm64/mm/mmu.c | 44 ++++++++++++++++++++++++++++ 2 files changed, 49 insertions(+), 44 deletions(-) diff --git a/arch/arm64/include/asm/pgtable.h b/arch/arm64/include/asm/pgtable.h index e89ec5f4787b49..763c5a411d64e3 100644 --- a/arch/arm64/include/asm/pgtable.h +++ b/arch/arm64/include/asm/pgtable.h @@ -387,52 +387,13 @@ static inline pte_t __ptep_get(pte_t *ptep) extern void __sync_icache_dcache(pte_t pteval); bool pgattr_change_is_safe(pteval_t old, pteval_t new); -/* - * PTE bits configuration in the presence of hardware Dirty Bit Management - * (PTE_WRITE == PTE_DBM): - * - * Dirty Writable | PTE_RDONLY PTE_WRITE PTE_DIRTY (sw) - * 0 0 | 1 0 0 - * 0 1 | 1 1 0 - * 1 0 | 1 0 1 - * 1 1 | 0 1 x - * - * When hardware DBM is not present, the software PTE_DIRTY bit is updated via - * the page fault mechanism. Checking the dirty status of a pte becomes: - * - * PTE_DIRTY || (PTE_WRITE && !PTE_RDONLY) - */ - -static inline void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, - pte_t pte) +#ifdef CONFIG_DEBUG_VM +void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, pte_t pte); +#else +static inline void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, pte_t pte) { - pte_t old_pte; - - if (!IS_ENABLED(CONFIG_DEBUG_VM)) - return; - - old_pte = __ptep_get(ptep); - - if (!pte_valid(old_pte) || !pte_valid(pte)) - return; - if (mm != current->active_mm && atomic_read(&mm->mm_users) <= 1) - return; - - /* - * Check for potential race with hardware updates of the pte - * (__ptep_set_access_flags safely changes valid ptes without going - * through an invalid entry). - */ - VM_WARN_ONCE(!pte_young(pte), - "%s: racy access flag clearing: 0x%016llx -> 0x%016llx", - __func__, pte_val(old_pte), pte_val(pte)); - VM_WARN_ONCE(pte_write(old_pte) && !pte_dirty(pte), - "%s: racy dirty state clearing: 0x%016llx -> 0x%016llx", - __func__, pte_val(old_pte), pte_val(pte)); - VM_WARN_ONCE(!pgattr_change_is_safe(pte_val(old_pte), pte_val(pte)), - "%s: unsafe attribute change: 0x%016llx -> 0x%016llx", - __func__, pte_val(old_pte), pte_val(pte)); } +#endif /* CONFIG_DEBUG_VM */ static inline void __sync_cache_and_tags(pte_t pte, unsigned int nr_pages) { diff --git a/arch/arm64/mm/mmu.c b/arch/arm64/mm/mmu.c index 79d90226fd5dc9..cb49469707a877 100644 --- a/arch/arm64/mm/mmu.c +++ b/arch/arm64/mm/mmu.c @@ -2392,4 +2392,48 @@ int arch_set_user_pkey_access(int pkey, unsigned long init_val) return 0; } + +/* + * PTE bits configuration in the presence of hardware Dirty Bit Management + * (PTE_WRITE == PTE_DBM): + * + * Dirty Writable | PTE_RDONLY PTE_WRITE PTE_DIRTY (sw) + * 0 0 | 1 0 0 + * 0 1 | 1 1 0 + * 1 0 | 1 0 1 + * 1 1 | 0 1 x + * + * When hardware DBM is not present, the software PTE_DIRTY bit is updated via + * the page fault mechanism. Checking the dirty status of a pte becomes: + * + * PTE_DIRTY || (PTE_WRITE && !PTE_RDONLY) + */ +#ifdef CONFIG_DEBUG_VM +void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, pte_t pte) +{ + pte_t old_pte; + + old_pte = __ptep_get(ptep); + + if (!pte_valid(old_pte) || !pte_valid(pte)) + return; + if (mm != current->active_mm && atomic_read(&mm->mm_users) <= 1) + return; + + /* + * Check for potential race with hardware updates of the pte + * (__ptep_set_access_flags safely changes valid ptes without going + * through an invalid entry). + */ + VM_WARN_ONCE(!pte_young(pte), + "%s: racy access flag clearing: 0x%016llx -> 0x%016llx", + __func__, pte_val(old_pte), pte_val(pte)); + VM_WARN_ONCE(pte_write(old_pte) && !pte_dirty(pte), + "%s: racy dirty state clearing: 0x%016llx -> 0x%016llx", + __func__, pte_val(old_pte), pte_val(pte)); + VM_WARN_ONCE(!pgattr_change_is_safe(pte_val(old_pte), pte_val(pte)), + "%s: unsafe attribute change: 0x%016llx -> 0x%016llx", + __func__, pte_val(old_pte), pte_val(pte)); +} +#endif /* CONFIG_DEBUG_VM */ #endif From 5d2d5f0c4c9d12c2461e9ff351a91bd023abd5d8 Mon Sep 17 00:00:00 2001 From: "David Hildenbrand (Arm)" Date: Mon, 28 Sep 2026 09:42:06 +0200 Subject: [PATCH 0849/1012] arm64-mm-move-__check_safe_pte_update-fix fix this: >> aarch64-linux-ld: Unexpected GOT/PLT entries detected! >> aarch64-linux-ld: Unexpected run-time procedure linkages detected! aarch64-linux-ld: arch/arm64/mm/mmu.o: in function `__set_ptes_anysz.isra.40.constprop.47': >> mmu.c:(.text+0x9a8): undefined reference to `__check_safe_pte_update' aarch64-linux-ld: arch/arm64/mm/hugetlbpage.o: in function `__set_ptes_anysz.isra.32': >> hugetlbpage.c:(.text+0x1a8): undefined reference to `__check_safe_pte_update' aarch64-linux-ld: kernel/bpf/arena.o: in function `apply_range_set_cb': >> arena.c:(.text+0x11fc): undefined reference to `__check_safe_pte_update' aarch64-linux-ld: mm/gup.o: in function `follow_page_pte': >> gup.c:(.text+0x2b48): undefined reference to `__check_safe_pte_update' aarch64-linux-ld: mm/memory.o: in function `__set_ptes.isra.207': >> memory.c:(.text+0x3418): undefined reference to `__check_safe_pte_update' aarch64-linux-ld: mm/mprotect.o:mprotect.c:(.text+0x1080): more undefined references to `__check_safe_pte_update' follow Link: https://lore.kernel.org/99023678-fd8a-4eef-8e43-a07b504e3e96@kernel.org Signed-off-by: David Hildenbrand (Arm) Signed-off-by: Andrew Morton Reported-by: kernel test robot Closes: https://lore.kernel.org/oe-kbuild-all/202609270030.Y0QVBPmK-lkp@intel.com/ Cc: Anshuman Khandual Cc: Catalin Marinas --- arch/arm64/mm/mmu.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/arch/arm64/mm/mmu.c b/arch/arm64/mm/mmu.c index cb49469707a877..05b7b52d761cdf 100644 --- a/arch/arm64/mm/mmu.c +++ b/arch/arm64/mm/mmu.c @@ -2392,6 +2392,7 @@ int arch_set_user_pkey_access(int pkey, unsigned long init_val) return 0; } +#endif /* * PTE bits configuration in the presence of hardware Dirty Bit Management @@ -2436,4 +2437,3 @@ void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, pte_t pte) __func__, pte_val(old_pte), pte_val(pte)); } #endif /* CONFIG_DEBUG_VM */ -#endif From fc00776ee78ab6d6ab9bc9e5ca2179b91a6de299 Mon Sep 17 00:00:00 2001 From: Anshuman Khandual Date: Wed, 16 Sep 2026 10:19:33 +0530 Subject: [PATCH 0850/1012] arm64/mm: standardize printing for pgtable entries Standardize printing for pgtable entries using recently introduced generic helper ptval_bytes_to_hex_str() in core MM which automatically enables 128 bits entries when added later. Link: https://lore.kernel.org/20260916044933.2689426-3-anshuman.khandual@arm.com Signed-off-by: Anshuman Khandual Signed-off-by: Andrew Morton Reviewed-by: David Hildenbrand (Arm) Reviewed-by: Catalin Marinas Cc: Will Deacon Cc: Mike Rapoport Cc: Lorenzo Stoakes --- arch/arm64/mm/fault.c | 16 +++++++++++----- arch/arm64/mm/mmu.c | 16 ++++++++++------ 2 files changed, 21 insertions(+), 11 deletions(-) diff --git a/arch/arm64/mm/fault.c b/arch/arm64/mm/fault.c index 75c3e463df2ef2..2cecf6ba6df7cf 100644 --- a/arch/arm64/mm/fault.c +++ b/arch/arm64/mm/fault.c @@ -131,6 +131,7 @@ static inline unsigned long mm_to_pgd_phys(struct mm_struct *mm) */ static void show_pte(unsigned long addr) { + char pxd_str[PTVAL_STR_MAX]; struct mm_struct *mm; pgd_t *pgdp; pgd_t pgd; @@ -160,7 +161,8 @@ static void show_pte(unsigned long addr) pgdp = pgd_offset(mm, addr); pgd = READ_ONCE(*pgdp); - pr_alert("[%016lx] pgd=%016llx", addr, pgd_val(pgd)); + ptval_to_str(pxd_str, pgd_val(pgd)); + pr_alert("[%016lx] pgd=%s", addr, pxd_str); do { p4d_t *p4dp, p4d; @@ -173,19 +175,22 @@ static void show_pte(unsigned long addr) p4dp = p4d_offset_lockless(pgdp, pgd, addr); p4d = READ_ONCE(*p4dp); - pr_cont(", p4d=%016llx", p4d_val(p4d)); + ptval_to_str(pxd_str, p4d_val(p4d)); + pr_cont(", p4d=%s", pxd_str); if (p4d_none(p4d) || p4d_bad(p4d)) break; pudp = pud_offset_lockless(p4dp, p4d, addr); pud = READ_ONCE(*pudp); - pr_cont(", pud=%016llx", pud_val(pud)); + ptval_to_str(pxd_str, pud_val(pud)); + pr_cont(", pud=%s", pxd_str); if (pud_none(pud) || pud_bad(pud)) break; pmdp = pmd_offset_lockless(pudp, pud, addr); pmd = READ_ONCE(*pmdp); - pr_cont(", pmd=%016llx", pmd_val(pmd)); + ptval_to_str(pxd_str, pmd_val(pmd)); + pr_cont(", pmd=%s", pxd_str); if (pmd_none(pmd) || pmd_bad(pmd)) break; @@ -194,7 +199,8 @@ static void show_pte(unsigned long addr) break; pte = __ptep_get(ptep); - pr_cont(", pte=%016llx", pte_val(pte)); + ptval_to_str(pxd_str, pte_val(pte)); + pr_cont(", pte=%s", pxd_str); pte_unmap(ptep); } while(0); diff --git a/arch/arm64/mm/mmu.c b/arch/arm64/mm/mmu.c index 05b7b52d761cdf..7343ac9294f8d6 100644 --- a/arch/arm64/mm/mmu.c +++ b/arch/arm64/mm/mmu.c @@ -2412,6 +2412,8 @@ int arch_set_user_pkey_access(int pkey, unsigned long init_val) #ifdef CONFIG_DEBUG_VM void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, pte_t pte) { + char pte_str_old[PTVAL_STR_MAX]; + char pte_str[PTVAL_STR_MAX]; pte_t old_pte; old_pte = __ptep_get(ptep); @@ -2426,14 +2428,16 @@ void __check_safe_pte_update(struct mm_struct *mm, pte_t *ptep, pte_t pte) * (__ptep_set_access_flags safely changes valid ptes without going * through an invalid entry). */ + ptval_to_str(pte_str, pte_val(pte)); + ptval_to_str(pte_str_old, pte_val(old_pte)); VM_WARN_ONCE(!pte_young(pte), - "%s: racy access flag clearing: 0x%016llx -> 0x%016llx", - __func__, pte_val(old_pte), pte_val(pte)); + "%s: racy access flag clearing: %s -> %s", + __func__, pte_str_old, pte_str); VM_WARN_ONCE(pte_write(old_pte) && !pte_dirty(pte), - "%s: racy dirty state clearing: 0x%016llx -> 0x%016llx", - __func__, pte_val(old_pte), pte_val(pte)); + "%s: racy dirty state clearing: %s -> %s", + __func__, pte_str_old, pte_str); VM_WARN_ONCE(!pgattr_change_is_safe(pte_val(old_pte), pte_val(pte)), - "%s: unsafe attribute change: 0x%016llx -> 0x%016llx", - __func__, pte_val(old_pte), pte_val(pte)); + "%s: unsafe attribute change: %s -> %s", + __func__, pte_str_old, pte_str); } #endif /* CONFIG_DEBUG_VM */ From d4cabc260ed84d74542e1d45f91ee6ef8faad01c Mon Sep 17 00:00:00 2001 From: "Mike Rapoport (Microsoft)" Date: Sat, 26 Sep 2026 12:29:29 +0300 Subject: [PATCH 0851/1012] arch, mm: promote DEBUG_WX to CHECK_WX Verification that the kernel does not have writable + executable mappings is about detecting security risks rather than a pure debug feature. Major distribution configurations enable it in their kernels as well as defconfigs of most architectures that have ARCH_HAS_DEBUG_WX. Rename relevant generic configuration options to use CHECK_WX and move their definitions from mm/Kconfig.debug to mm/Kconfig. Rename *debug_checkwx() funcitons and macros to *pgtable_checkwx(). For arm that does not widely enable it, only rename its variants of the config options. Enabling CHECK_WX adds a few kilobytes to the kernel binary and while the added size can be slightly reduced with churny updates of architecture implementations of ptdump, the core functionality takes most of the added size. It cannot be moved to .init.text because the verification has to happen after init sections are freed. With this, make generic CHECK_WX default to STRICT_KERNEL_RWX while still leaving users targeting small kernels the possibility to opt-out. Link: https://lore.kernel.org/20260926-direct-map-verify-wx-v2-1-efcd64a6b74a@kernel.org Signed-off-by: Mike Rapoport (Microsoft) Signed-off-by: Andrew Morton Suggested-by: Dave Hansen Acked-by: Lorenzo Stoakes (ARM) Acked-by: Dave Hansen Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Borislav Petkov Cc: Catalin Marinas Cc: Christophe Leroy Cc: Christian Borntraeger Cc: David Hildenbrand Cc: Gerald Schaefer Cc: Heiko Carstens Cc: Ingo Molnar Cc: Liam R. Howlett Cc: Madhavan Srinivasan Cc: Mark Rutland Cc: Michael Ellerman Cc: Michal Hocko Cc: Nicholas Piggin Cc: Palmer Dabbelt Cc: Paul Walmsley Cc: H. Peter Anvin Cc: Ritesh Harjani (IBM) Cc: Russell King Cc: Shrikanth Hegde Cc: Suren Baghdasaryan Cc: Sven Schnelle Cc: Thomas Gleixner Cc: Vasily Gorbik Cc: Vlastimil Babka Cc: Will Deacon --- arch/arm/Kconfig.debug | 2 +- arch/arm/configs/aspeed_g4_defconfig | 2 +- arch/arm/configs/aspeed_g5_defconfig | 2 +- arch/arm/configs/shmobile_defconfig | 2 +- arch/arm/include/asm/ptdump.h | 6 ++-- arch/arm/mm/init.c | 2 +- arch/arm64/Kconfig | 2 +- arch/powerpc/Kconfig | 2 +- arch/powerpc/configs/ppc64_defconfig | 2 +- arch/powerpc/mm/ptdump/ptdump.c | 2 +- arch/riscv/Kconfig | 2 +- arch/s390/Kconfig | 2 +- arch/s390/configs/debug_defconfig | 2 +- arch/s390/configs/defconfig | 2 +- arch/s390/mm/dump_pagetables.c | 2 +- arch/x86/Kconfig | 2 +- arch/x86/configs/x86_64_defconfig | 2 +- arch/x86/include/asm/pgtable.h | 6 ++-- arch/x86/mm/pti.c | 2 +- include/linux/ptdump.h | 4 +-- init/main.c | 2 +- kernel/configs/debug.config | 2 +- mm/Kconfig | 41 ++++++++++++++++++++++++++++ mm/Kconfig.debug | 39 -------------------------- 24 files changed, 68 insertions(+), 66 deletions(-) diff --git a/arch/arm/Kconfig.debug b/arch/arm/Kconfig.debug index 366f162e147d11..abcf14f10276be 100644 --- a/arch/arm/Kconfig.debug +++ b/arch/arm/Kconfig.debug @@ -17,7 +17,7 @@ config ARM_PTDUMP_DEBUGFS kernel. If in doubt, say "N" -config ARM_DEBUG_WX +config ARM_CHECK_WX bool "Warn on W+X mappings at boot" depends on MMU select ARM_PTDUMP_CORE diff --git a/arch/arm/configs/aspeed_g4_defconfig b/arch/arm/configs/aspeed_g4_defconfig index f86dd4ce7d0def..2c9d5a644ae965 100644 --- a/arch/arm/configs/aspeed_g4_defconfig +++ b/arch/arm/configs/aspeed_g4_defconfig @@ -249,7 +249,7 @@ CONFIG_DEBUG_INFO_REDUCED=y CONFIG_GDB_SCRIPTS=y CONFIG_STRIP_ASM_SYMS=y CONFIG_DEBUG_FS=y -CONFIG_ARM_DEBUG_WX=y +CONFIG_ARM_CHECK_WX=y CONFIG_SCHED_STACK_END_CHECK=y CONFIG_PANIC_ON_OOPS=y CONFIG_PANIC_TIMEOUT=-1 diff --git a/arch/arm/configs/aspeed_g5_defconfig b/arch/arm/configs/aspeed_g5_defconfig index 45b937419dbdaa..1327a09e163ab8 100644 --- a/arch/arm/configs/aspeed_g5_defconfig +++ b/arch/arm/configs/aspeed_g5_defconfig @@ -300,7 +300,7 @@ CONFIG_DEBUG_INFO_REDUCED=y CONFIG_GDB_SCRIPTS=y CONFIG_STRIP_ASM_SYMS=y CONFIG_DEBUG_FS=y -CONFIG_ARM_DEBUG_WX=y +CONFIG_ARM_CHECK_WX=y CONFIG_SCHED_STACK_END_CHECK=y CONFIG_PANIC_ON_OOPS=y CONFIG_PANIC_TIMEOUT=-1 diff --git a/arch/arm/configs/shmobile_defconfig b/arch/arm/configs/shmobile_defconfig index 6f9696e9fe17dd..cc22e22b989e97 100644 --- a/arch/arm/configs/shmobile_defconfig +++ b/arch/arm/configs/shmobile_defconfig @@ -225,4 +225,4 @@ CONFIG_CMA_SIZE_MBYTES=64 CONFIG_PRINTK_TIME=y CONFIG_DEBUG_KERNEL=y CONFIG_DEBUG_FS=y -CONFIG_ARM_DEBUG_WX=y +CONFIG_ARM_CHECK_WX=y diff --git a/arch/arm/include/asm/ptdump.h b/arch/arm/include/asm/ptdump.h index 46a4575146ee85..3c5245220ecfee 100644 --- a/arch/arm/include/asm/ptdump.h +++ b/arch/arm/include/asm/ptdump.h @@ -32,10 +32,10 @@ void ptdump_check_wx(void); #endif /* CONFIG_ARM_PTDUMP_CORE */ -#ifdef CONFIG_ARM_DEBUG_WX -#define arm_debug_checkwx() ptdump_check_wx() +#ifdef CONFIG_ARM_CHECK_WX +#define arm_pgtable_checkwx() ptdump_check_wx() #else -#define arm_debug_checkwx() do { } while (0) +#define arm_pgtable_checkwx() do { } while (0) #endif #endif /* __ASM_PTDUMP_H */ diff --git a/arch/arm/mm/init.c b/arch/arm/mm/init.c index 0cc1bf04686d83..c515faf22eebe9 100644 --- a/arch/arm/mm/init.c +++ b/arch/arm/mm/init.c @@ -403,7 +403,7 @@ static int __mark_rodata_ro(void *unused) void mark_rodata_ro(void) { stop_machine(__mark_rodata_ro, NULL, NULL); - arm_debug_checkwx(); + arm_pgtable_checkwx(); } #else diff --git a/arch/arm64/Kconfig b/arch/arm64/Kconfig index b6c2dd8b26124d..b51d23a62c9224 100644 --- a/arch/arm64/Kconfig +++ b/arch/arm64/Kconfig @@ -11,7 +11,7 @@ config ARM64 select ACPI_MCFG if (ACPI && PCI) select ACPI_SPCR_TABLE if ACPI select ACPI_PPTT if ACPI - select ARCH_HAS_DEBUG_WX + select ARCH_HAS_CHECK_WX select ARCH_BINFMT_ELF_EXTRA_PHDRS select ARCH_BINFMT_ELF_STATE select ARCH_ENABLE_HUGEPAGE_MIGRATION if HUGETLB_PAGE && MIGRATION diff --git a/arch/powerpc/Kconfig b/arch/powerpc/Kconfig index 0767cfcbaa422b..877393207e3b54 100644 --- a/arch/powerpc/Kconfig +++ b/arch/powerpc/Kconfig @@ -130,7 +130,7 @@ config PPC select ARCH_HAS_CURRENT_STACK_POINTER select ARCH_HAS_DEBUG_VIRTUAL select ARCH_HAS_DEBUG_VM_PGTABLE - select ARCH_HAS_DEBUG_WX if STRICT_KERNEL_RWX + select ARCH_HAS_CHECK_WX if STRICT_KERNEL_RWX select ARCH_HAS_DEVMEM_IS_ALLOWED select ARCH_HAS_DMA_MAP_DIRECT if PPC_PSERIES select ARCH_HAS_DMA_OPS if PPC64 diff --git a/arch/powerpc/configs/ppc64_defconfig b/arch/powerpc/configs/ppc64_defconfig index 1eb8e3457e8bff..5c33f0bba0e361 100644 --- a/arch/powerpc/configs/ppc64_defconfig +++ b/arch/powerpc/configs/ppc64_defconfig @@ -393,7 +393,7 @@ CONFIG_MAGIC_SYSRQ=y CONFIG_PAGE_OWNER=y CONFIG_PAGE_POISONING=y CONFIG_DEBUG_RODATA_TEST=y -CONFIG_DEBUG_WX=y +CONFIG_CHECK_WX=y CONFIG_DEBUG_STACK_USAGE=y CONFIG_DEBUG_VM=y # CONFIG_DEBUG_VM_PGTABLE is not set diff --git a/arch/powerpc/mm/ptdump/ptdump.c b/arch/powerpc/mm/ptdump/ptdump.c index 0d499aebee72ff..3451351b756b4d 100644 --- a/arch/powerpc/mm/ptdump/ptdump.c +++ b/arch/powerpc/mm/ptdump/ptdump.c @@ -191,7 +191,7 @@ static void note_prot_wx(struct pg_state *st, unsigned long addr) if (!pte_write(pte) || !pte_exec(pte)) return; - WARN_ONCE(IS_ENABLED(CONFIG_DEBUG_WX), + WARN_ONCE(IS_ENABLED(CONFIG_CHECK_WX), "powerpc/mm: Found insecure W+X mapping at address %p/%pS\n", (void *)st->start_address, (void *)st->start_address); diff --git a/arch/riscv/Kconfig b/arch/riscv/Kconfig index 0db108ea146626..f409f264d8c413 100644 --- a/arch/riscv/Kconfig +++ b/arch/riscv/Kconfig @@ -29,7 +29,7 @@ config RISCV select ARCH_HAS_CURRENT_STACK_POINTER select ARCH_HAS_DEBUG_VIRTUAL if MMU select ARCH_HAS_DEBUG_VM_PGTABLE - select ARCH_HAS_DEBUG_WX + select ARCH_HAS_CHECK_WX select ARCH_HAS_DELAY_TIMER select ARCH_HAS_ELF_CORE_EFLAGS if BINFMT_ELF && ELF_CORE select ARCH_HAS_FAST_MULTIPLIER diff --git a/arch/s390/Kconfig b/arch/s390/Kconfig index a34376c05f6e3c..4bbb2c89bce057 100644 --- a/arch/s390/Kconfig +++ b/arch/s390/Kconfig @@ -92,7 +92,7 @@ config S390 select ARCH_HAS_CURRENT_STACK_POINTER select ARCH_HAS_DEBUG_VIRTUAL select ARCH_HAS_DEBUG_VM_PGTABLE - select ARCH_HAS_DEBUG_WX + select ARCH_HAS_CHECK_WX select ARCH_HAS_DEVMEM_IS_ALLOWED select ARCH_HAS_DMA_OPS if PCI select ARCH_HAS_ELF_RANDOMIZE diff --git a/arch/s390/configs/debug_defconfig b/arch/s390/configs/debug_defconfig index 3dae7147433305..68d53c0bc8dbe5 100644 --- a/arch/s390/configs/debug_defconfig +++ b/arch/s390/configs/debug_defconfig @@ -841,7 +841,7 @@ CONFIG_DEBUG_PAGEALLOC=y CONFIG_SLUB_DEBUG_ON=y CONFIG_PAGE_OWNER=y CONFIG_DEBUG_RODATA_TEST=y -CONFIG_DEBUG_WX=y +CONFIG_CHECK_WX=y CONFIG_PTDUMP_DEBUGFS=y CONFIG_DEBUG_OBJECTS=y CONFIG_DEBUG_OBJECTS_SELFTEST=y diff --git a/arch/s390/configs/defconfig b/arch/s390/configs/defconfig index 6f5722634b4d39..8e5cfc69512119 100644 --- a/arch/s390/configs/defconfig +++ b/arch/s390/configs/defconfig @@ -820,7 +820,7 @@ CONFIG_DEBUG_INFO_DWARF4=y CONFIG_GDB_SCRIPTS=y CONFIG_DEBUG_SECTION_MISMATCH=y CONFIG_MAGIC_SYSRQ=y -CONFIG_DEBUG_WX=y +CONFIG_CHECK_WX=y CONFIG_PTDUMP_DEBUGFS=y CONFIG_DEBUG_MEMORY_INIT=y CONFIG_PANIC_ON_OOPS=y diff --git a/arch/s390/mm/dump_pagetables.c b/arch/s390/mm/dump_pagetables.c index 89badbe72ae706..a23a0bd4d8a885 100644 --- a/arch/s390/mm/dump_pagetables.c +++ b/arch/s390/mm/dump_pagetables.c @@ -86,7 +86,7 @@ static void note_prot_wx(struct pg_state *st, unsigned long addr) */ if (addr == PAGE_SIZE && (nospec_uses_trampoline() || !cpu_has_bear())) return; - WARN_ONCE(IS_ENABLED(CONFIG_DEBUG_WX), + WARN_ONCE(IS_ENABLED(CONFIG_CHECK_WX), "s390/mm: Found insecure W+X mapping at address %pS\n", (void *)st->start_address); st->wx_pages += (addr - st->start_address) / PAGE_SIZE; diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig index 6e5e462ec059a1..6cb70e47d0bda2 100644 --- a/arch/x86/Kconfig +++ b/arch/x86/Kconfig @@ -109,7 +109,7 @@ config X86 select ARCH_HAS_SYNC_CORE_BEFORE_USERMODE select ARCH_HAS_SYSCALL_WRAPPER select ARCH_HAS_UBSAN - select ARCH_HAS_DEBUG_WX + select ARCH_HAS_CHECK_WX select ARCH_HAS_ZONE_DMA_SET if EXPERT select ARCH_HAVE_NMI_SAFE_CMPXCHG select ARCH_HAVE_EXTRA_ELF_NOTES diff --git a/arch/x86/configs/x86_64_defconfig b/arch/x86/configs/x86_64_defconfig index 269f7d808be4ec..e6896aeb77d81d 100644 --- a/arch/x86/configs/x86_64_defconfig +++ b/arch/x86/configs/x86_64_defconfig @@ -263,7 +263,7 @@ CONFIG_SECURITY_SELINUX_BOOTPARAM=y CONFIG_PRINTK_TIME=y CONFIG_DEBUG_KERNEL=y CONFIG_MAGIC_SYSRQ=y -CONFIG_DEBUG_WX=y +CONFIG_CHECK_WX=y CONFIG_DEBUG_STACK_USAGE=y CONFIG_SCHEDSTATS=y CONFIG_BLK_DEV_IO_TRACE=y diff --git a/arch/x86/include/asm/pgtable.h b/arch/x86/include/asm/pgtable.h index d551120a7c889b..ef0252a09c2804 100644 --- a/arch/x86/include/asm/pgtable.h +++ b/arch/x86/include/asm/pgtable.h @@ -41,10 +41,10 @@ void ptdump_walk_user_pgd_level_checkwx(void); #define pgprot_encrypted(prot) __pgprot(cc_mkenc(pgprot_val(prot))) #define pgprot_decrypted(prot) __pgprot(cc_mkdec(pgprot_val(prot))) -#ifdef CONFIG_DEBUG_WX -#define debug_checkwx_user() ptdump_walk_user_pgd_level_checkwx() +#ifdef CONFIG_CHECK_WX +#define pgtable_checkwx_user() ptdump_walk_user_pgd_level_checkwx() #else -#define debug_checkwx_user() do { } while (0) +#define pgtable_checkwx_user() do { } while (0) #endif extern spinlock_t pgd_lock; diff --git a/arch/x86/mm/pti.c b/arch/x86/mm/pti.c index 598f553cc8713c..31055ee6f1de8e 100644 --- a/arch/x86/mm/pti.c +++ b/arch/x86/mm/pti.c @@ -688,5 +688,5 @@ void pti_finalize(void) pti_clone_entry_text(true); pti_clone_kernel_text(); - debug_checkwx_user(); + pgtable_checkwx_user(); } diff --git a/include/linux/ptdump.h b/include/linux/ptdump.h index 240bd3bff18dd2..af18d1459b2f49 100644 --- a/include/linux/ptdump.h +++ b/include/linux/ptdump.h @@ -31,9 +31,9 @@ bool ptdump_walk_pgd_level_core(struct seq_file *m, void ptdump_walk_pgd(struct ptdump_state *st, struct mm_struct *mm, pgd_t *pgd); bool ptdump_check_wx(void); -static inline void debug_checkwx(void) +static inline void pgtable_checkwx(void) { - if (IS_ENABLED(CONFIG_DEBUG_WX)) + if (IS_ENABLED(CONFIG_CHECK_WX)) ptdump_check_wx(); } diff --git a/init/main.c b/init/main.c index 31f2bf54976ab9..a87a3d52f3e200 100644 --- a/init/main.c +++ b/init/main.c @@ -1535,7 +1535,7 @@ static void mark_readonly(void) flush_module_init_free_work(); jump_label_init_ro(); mark_rodata_ro(); - debug_checkwx(); + pgtable_checkwx(); rodata_test(); } else if (IS_ENABLED(CONFIG_STRICT_KERNEL_RWX)) { pr_info("Kernel memory protection disabled.\n"); diff --git a/kernel/configs/debug.config b/kernel/configs/debug.config index 307c97ac5fa9c3..ac878669c19365 100644 --- a/kernel/configs/debug.config +++ b/kernel/configs/debug.config @@ -50,7 +50,7 @@ CONFIG_DEBUG_NET=y # CONFIG_DEBUG_PAGEALLOC is not set # CONFIG_DEBUG_KMEMLEAK_DEFAULT_OFF is not set # CONFIG_DEBUG_RODATA_TEST is not set -# CONFIG_DEBUG_WX is not set +# CONFIG_CHECK_WX is not set # CONFIG_KFENCE is not set # CONFIG_PAGE_POISONING is not set # CONFIG_SLUB_STATS is not set diff --git a/mm/Kconfig b/mm/Kconfig index edb4a6c0a87021..acefc994d9a8b9 100644 --- a/mm/Kconfig +++ b/mm/Kconfig @@ -1494,6 +1494,47 @@ config LAZY_MMU_MODE_KUNIT_TEST If unsure, say N. +config ARCH_HAS_CHECK_WX + bool + +config CHECK_WX + bool "Warn on W+X mappings at boot" + default STRICT_KERNEL_RWX + depends on ARCH_HAS_CHECK_WX + depends on ARCH_HAS_PTDUMP + depends on MMU + select PTDUMP + help + Generate a warning if any W+X mappings are found at boot. + + This is useful for discovering cases where the kernel is leaving W+X + mappings after applying NX, as such mappings are a security risk. + + Look for a message in dmesg output like this: + + /mm: Checked W+X mappings: passed, no W+X pages found. + + or like this, if the check failed: + + /mm: Checked W+X mappings: failed, W+X pages found. + + Note that even if the check fails, your kernel is possibly + still fine, as W+X mappings are not a security hole in + themselves, what they do is that they make the exploitation + of other unfixed kernel bugs easier. + + There is no runtime or memory usage effect of this option + once the kernel has booted up - it's a one time check. + + If in doubt, say "Y". + +config ARCH_HAS_PTDUMP + bool + +config PTDUMP + bool + + source "mm/damon/Kconfig" endmenu diff --git a/mm/Kconfig.debug b/mm/Kconfig.debug index 9eaa25d1cf2340..9ae75229c76da1 100644 --- a/mm/Kconfig.debug +++ b/mm/Kconfig.debug @@ -180,45 +180,6 @@ config DEBUG_RODATA_TEST help This option enables a testcase for the setting rodata read-only. -config ARCH_HAS_DEBUG_WX - bool - -config DEBUG_WX - bool "Warn on W+X mappings at boot" - depends on ARCH_HAS_DEBUG_WX - depends on ARCH_HAS_PTDUMP - depends on MMU - select PTDUMP - help - Generate a warning if any W+X mappings are found at boot. - - This is useful for discovering cases where the kernel is leaving W+X - mappings after applying NX, as such mappings are a security risk. - - Look for a message in dmesg output like this: - - /mm: Checked W+X mappings: passed, no W+X pages found. - - or like this, if the check failed: - - /mm: Checked W+X mappings: failed, W+X pages found. - - Note that even if the check fails, your kernel is possibly - still fine, as W+X mappings are not a security hole in - themselves, what they do is that they make the exploitation - of other unfixed kernel bugs easier. - - There is no runtime or memory usage effect of this option - once the kernel has booted up - it's a one time check. - - If in doubt, say "Y". - -config ARCH_HAS_PTDUMP - bool - -config PTDUMP - bool - config PTDUMP_DEBUGFS bool "Export kernel pagetable layout to userspace via debugfs" depends on DEBUG_KERNEL From 60d660b1404f89464172cd0fbcc2f30d3a015911 Mon Sep 17 00:00:00 2001 From: Palla Raghunath Date: Fri, 25 Sep 2026 21:54:49 +0100 Subject: [PATCH 0852/1012] mm/vmalloc: do not warn on -ENOMEM from va_clip() in pcpu_get_vm_areas() When pcpu_get_vm_areas() has to split a free vmap_area in the middle (NE_FIT_TYPE), va_clip() needs an extra vmap_area object. It takes the per-cpu ne_fit_preload_node if one is there, and otherwise falls back to kmem_cache_alloc(GFP_NOWAIT), which may fail and return -ENOMEM. pcpu_get_vm_areas() never preloads, and a single call can do more than one such split: on a NUMA system it places one area per node group, so the first split consumes the preloaded object and the next one depends on the GFP_NOWAIT allocation. That failure is expected and already handled: the recovery path returns the areas clipped so far to the free tree, purges lazily freed areas and retries. But the error is checked with WARN_ON_ONCE(), so a transient allocation failure under memory pressure or fault injection triggers a kernel warning, and a panic with panic_on_warn. syzbot hit this on a two-node VM while creating a per-cpu BPF array map. Keep the WARN_ON_ONCE() for errors other than -ENOMEM, which do indicate a bug, and take the recovery path either way. This matches what commit b9183788a2de ("mm/vmalloc: do not warn on -ENOMEM from va_alloc()") did for the other va_clip() caller. Link: https://lore.kernel.org/20260925205450.21262-1-raghunathpalla.0209@gmail.com Fixes: 1b23ff80b399 ("mm/vmalloc: invoke classify_va_fit_type() in adjust_va_to_fit_type()") Signed-off-by: Palla Raghunath Signed-off-by: Andrew Morton Reported-by: Closes: https://syzkaller.appspot.com/bug?extid=442828bb356b10813a47 Reviewed-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Reviewed-by: Dev Jain Cc: Shuah Khan Cc: Brigham Campbell Cc: Baoquan He --- mm/vmalloc.c | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index b24896aedac212..4e4cb8d785bdb4 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -5135,9 +5135,14 @@ struct vm_struct **pcpu_get_vm_areas(const unsigned long *offsets, ret = va_clip(&free_vmap_area_root, &free_vmap_area_list, va, start, size); - if (WARN_ON_ONCE(unlikely(ret))) - /* It is a BUG(), but trigger recovery instead. */ + if (unlikely(ret)) { + /* + * -ENOMEM from the GFP_NOWAIT fallback is expected. + * Anything else is a BUG(), but trigger recovery instead. + */ + WARN_ON_ONCE(ret != -ENOMEM); goto recovery; + } /* Allocated area. */ va = vas[area]; From 94f2a977e5aaa8783ce97fa9842112b50558bff3 Mon Sep 17 00:00:00 2001 From: "Lorenzo Stoakes (ARM)" Date: Fri, 25 Sep 2026 19:32:20 +0100 Subject: [PATCH 0853/1012] mm/vma: don't remove VMA from rmap if pgoff unchanged When updating a VMA, vma_prepare() unconditionally removes it from its rmap interval trees under the rmap lock, and vma_complete() reinserts it before releasing the lock. This is wholly unnecessary if its page offset (file rmap) or anonymous page offset (anon rmap) is unchanged. So, track whether they will change in the newly introduced vp->anon_pgoff_unchanged and vp->pgoff_unchanged fields, and use them to determine whether to remove the VMA or not. The rmap lock keeps things safe as no rmap walks can concurrently occur during the operation. Additionally, some architectures (arm, parisc, nios2, csky) have dcache flush rmap walkers which take only flush_dcache_mmap_lock(), which is likewise held across the operation. It's also necessary to keep the rb_subtree_last field updated in the interval tree so implement anon_rmap_tree_update_inplace() and mapping_rmap_tree_update_inplace() to do that. This is done in vma_complete(), after the VMA's range has been updated, so in the interim the field may be invalid. However, given the locks described above, this cannot be observed until after the state is corrected. The anonymous rmap is keyed on anon_vma_chains not VMAs, so in those instances anon_rmap_tree_update_vma_inplace() iterates over vma->anon_vma_chain, invoking anon_rmap_tree_update_inplace() on each one. For the anon rmap case, with CONFIG_DEBUG_VM_RB set, avc->cached_vma_last is also updated in anon_rmap_tree_update_inplace(). When performing a VMA shrink or a split where the VMA is the lower one, the page offset cannot change, so set the flags unconditionally in these cases. When merging VMAs the page offset is unchanged only in some cases, so update init_multi_vma_prep() to set the flags only if the page offsets remain the same. These changes ultimately result in less rmap lock contention. Link: https://lore.kernel.org/20260925-speed-up-inplace-rmap-v1-1-babc48ce7c83@kernel.org Signed-off-by: Lorenzo Stoakes Signed-off-by: Andrew Morton Suggested-by: Pan Deng Link: https://lore.kernel.org/linux-mm/20260924054301.2330822-1-pan.deng@intel.com/ Reviewed-by: Rik van Riel Cc: David Hildenbrand Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Harry Yoo Cc: Jann Horn Cc: Lance Yang Cc: Pedro Falcato --- include/linux/mm.h | 3 +++ mm/interval_tree.c | 33 +++++++++++++++++++++++++ mm/vma.c | 40 +++++++++++++++++++++++++++---- mm/vma.h | 2 ++ tools/testing/vma/include/stubs.h | 8 +++++++ 5 files changed, 81 insertions(+), 5 deletions(-) diff --git a/include/linux/mm.h b/include/linux/mm.h index 5e35864eb731c5..c038d06825c39c 100644 --- a/include/linux/mm.h +++ b/include/linux/mm.h @@ -4358,6 +4358,8 @@ void mapping_rmap_tree_insert_after(struct vm_area_struct *vma, struct address_space *mapping); void mapping_rmap_tree_remove(struct vm_area_struct *vma, struct address_space *mapping); +void mapping_rmap_tree_update_inplace(struct vm_area_struct *vma); + struct vm_area_struct * mapping_rmap_tree_iter_first(struct address_space *mapping, pgoff_t pgoff_start, pgoff_t pgoff_last); @@ -4375,6 +4377,7 @@ void anon_rmap_tree_insert(struct anon_vma_chain *avc, struct anon_vma *anon_vma); void anon_rmap_tree_remove(struct anon_vma_chain *avc, struct anon_vma *anon_vma); +void anon_rmap_tree_update_inplace(struct anon_vma_chain *avc); struct anon_vma_chain * anon_rmap_tree_iter_first(struct anon_vma *anon_vma, pgoff_t pgoff_start, pgoff_t pgoff_last); diff --git a/mm/interval_tree.c b/mm/interval_tree.c index 7bbbf15cfbf0c7..eafde5d12ef5f0 100644 --- a/mm/interval_tree.c +++ b/mm/interval_tree.c @@ -64,6 +64,21 @@ void mapping_rmap_tree_remove(struct vm_area_struct *vma, __mapping_rmap_tree_remove(vma, &mapping->i_mmap); } +/** + * mapping_rmap_tree_update_inplace() - Update file rmap tree to reflect an + * in-place change in a VMA's size. + * @vma: The VMA whose size has changed. + * + * The file rmap lock must be held. + * + * Invalid to do so if @vma->vm_pgoff has changed. + */ +void mapping_rmap_tree_update_inplace(struct vm_area_struct *vma) +{ + /* Propagate all the way up the tree. */ + __mapping_rmap_tree_augment.propagate(&vma->shared.rb, NULL); +} + struct vm_area_struct * mapping_rmap_tree_iter_first(struct address_space *mapping, pgoff_t pgoff_start, pgoff_t pgoff_last) @@ -111,6 +126,24 @@ void anon_rmap_tree_remove(struct anon_vma_chain *avc, __anon_rmap_tree_remove(avc, &anon_vma->rb_root); } +/** + * anon_rmap_tree_update_inplace() - Update anon rmap tree to reflect an + * in-place change in the size of @avc's VMA. + * @avc: The anon_vma_chain whose VMA's size has changed. + * + * The anon rmap root lock must be held. + * + * Invalid to do so if the VMA's anonymous pgoff has changed. + */ +void anon_rmap_tree_update_inplace(struct anon_vma_chain *avc) +{ +#ifdef CONFIG_DEBUG_VM_RB + avc->cached_vma_last = avc_last_pgoff(avc); +#endif + /* Propagate all the way up the tree. */ + __anon_rmap_tree_augment.propagate(&avc->rb, NULL); +} + struct anon_vma_chain * anon_rmap_tree_iter_first(struct anon_vma *anon_vma, pgoff_t pgoff_start, pgoff_t pgoff_last) diff --git a/mm/vma.c b/mm/vma.c index ea4dc3032657a1..cfe31dbf0e03c6 100644 --- a/mm/vma.c +++ b/mm/vma.c @@ -201,8 +201,15 @@ static void init_multi_vma_prep(struct vma_prepare *vp, if (vp->file) vp->mapping = vma->vm_file->f_mapping; - if (vmg && vmg->skip_vma_uprobe) + if (!vmg) + return; + + if (vmg->skip_vma_uprobe) vp->skip_vma_uprobe = true; + if (vma_start_pgoff(vma) == vmg_start_pgoff(vmg)) + vp->pgoff_unchanged = true; + if (vma_start_anon_pgoff(vma) == vmg_start_anon_pgoff(vmg)) + vp->anon_pgoff_unchanged = true; } /* @@ -331,6 +338,15 @@ anon_rmap_tree_post_update_vma(struct vm_area_struct *vma) anon_rmap_tree_insert(avc, avc->anon_vma); } +static void +anon_rmap_tree_update_vma_inplace(struct vm_area_struct *vma) +{ + struct anon_vma_chain *avc; + + list_for_each_entry(avc, &vma->anon_vma_chain, same_vma) + anon_rmap_tree_update_inplace(avc); +} + /* * vma_prepare() - Helper function for handling locking VMAs prior to altering * @vp: The initialized vma_prepare struct @@ -359,14 +375,16 @@ static void vma_prepare(struct vma_prepare *vp) if (vp->anon_vma) { anon_vma_lock_write(vp->anon_vma); - anon_rmap_tree_pre_update_vma(vp->vma); + if (!vp->anon_pgoff_unchanged) + anon_rmap_tree_pre_update_vma(vp->vma); if (vp->adj_next) anon_rmap_tree_pre_update_vma(vp->adj_next); } if (vp->file) { flush_dcache_mmap_lock(vp->mapping); - mapping_rmap_tree_remove(vp->vma, vp->mapping); + if (!vp->pgoff_unchanged) + mapping_rmap_tree_remove(vp->vma, vp->mapping); if (vp->adj_next) mapping_rmap_tree_remove(vp->adj_next, vp->mapping); } @@ -387,7 +405,11 @@ static void vma_complete(struct vma_prepare *vp, struct vma_iterator *vmi, if (vp->file) { if (vp->adj_next) mapping_rmap_tree_insert(vp->adj_next, vp->mapping); - mapping_rmap_tree_insert(vp->vma, vp->mapping); + /* Need only propagate the change inplace. */ + if (vp->pgoff_unchanged) + mapping_rmap_tree_update_inplace(vp->vma); + else + mapping_rmap_tree_insert(vp->vma, vp->mapping); flush_dcache_mmap_unlock(vp->mapping); } @@ -406,7 +428,11 @@ static void vma_complete(struct vma_prepare *vp, struct vma_iterator *vmi, } if (vp->anon_vma) { - anon_rmap_tree_post_update_vma(vp->vma); + /* Need only propagate the change inplace. */ + if (vp->anon_pgoff_unchanged) + anon_rmap_tree_update_vma_inplace(vp->vma); + else + anon_rmap_tree_post_update_vma(vp->vma); if (vp->adj_next) anon_rmap_tree_post_update_vma(vp->adj_next); anon_vma_unlock_write(vp->anon_vma); @@ -593,6 +619,8 @@ __split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma, init_vma_prep(&vp, vma); vp.insert = new; + vp.pgoff_unchanged = !new_below; + vp.anon_pgoff_unchanged = !new_below; vma_prepare(&vp); /* @@ -1346,6 +1374,8 @@ int vma_shrink(struct vma_iterator *vmi, struct vm_area_struct *vma, vma_start_write(vma); init_vma_prep(&vp, vma); + vp.pgoff_unchanged = true; + vp.anon_pgoff_unchanged = true; vma_prepare(&vp); vma_adjust_trans_huge(vma, vma->vm_start, end, NULL); diff --git a/mm/vma.h b/mm/vma.h index 7a683272c0a82a..336b4ced82c918 100644 --- a/mm/vma.h +++ b/mm/vma.h @@ -28,6 +28,8 @@ struct vma_prepare { struct vm_area_struct *remove2; bool skip_vma_uprobe :1; + bool pgoff_unchanged :1; + bool anon_pgoff_unchanged :1; }; struct unlink_vma_file_batch { diff --git a/tools/testing/vma/include/stubs.h b/tools/testing/vma/include/stubs.h index e4acc6f1fe7bab..f0a69393c02a84 100644 --- a/tools/testing/vma/include/stubs.h +++ b/tools/testing/vma/include/stubs.h @@ -267,6 +267,10 @@ static inline void mapping_rmap_tree_remove(struct vm_area_struct *vma, { } +static inline void mapping_rmap_tree_update_inplace(struct vm_area_struct *vma) +{ +} + static inline void flush_dcache_mmap_unlock(struct address_space *mapping) { } @@ -281,6 +285,10 @@ static inline void anon_rmap_tree_remove(struct anon_vma_chain *avc, { } +static inline void anon_rmap_tree_update_inplace(struct anon_vma_chain *avc) +{ +} + static inline void uprobe_mmap(struct vm_area_struct *vma) { } From d516caab6183e7baa6f538bda4f6e84909d592a5 Mon Sep 17 00:00:00 2001 From: Zhang Yi Date: Mon, 28 Sep 2026 20:08:30 +0800 Subject: [PATCH 0854/1012] mm/truncate: align truncation boundaries to mapping minimum folio order Patch series "mm/truncate: fix data loss when truncating straddling large folios", v5. This is the fifth version fixing data loss when truncating straddling large folios caught on the upcomming ext4 + iomap buffered I/O conversion. When truncate_inode_pages_range() punches a hole or truncates a file, truncate_inode_partial_folio() splits a large folio so that the caller can drop the in-range sub-folios while keeping the out-of-range tail intact. This series fixes three distinct problems in that path that can each lose the valid out-of-range tail of a straddling folio, plus a follow-up that clarifies the return value semantics. Patch 01 aligns the truncation boundaries inwards to the mapping minimum folio order in truncate_inode_pages_range(). With a non-zero min_order, folio_split() stops at min_order instead of order 0, so a boundary computed at page granularity can land inside a min-order-aligned sub-folio and the truncate loop drops that whole chunk, valid tail included, causing data loss. Patch 02 looks the end-edge straddler up by its page index through __filemap_get_folio() in truncate_inode_partial_folio(). After the first split the straddler is unlocked and only transiently ref'd in the page cache, so the page pointer derived from the original folio can be freed and reallocated as a different folio in the same mapping, and the mapping check cannot catch it, which may cause incorrect splitting and potential data loss. Patch 03 reworks the contract between truncate_inode_partial_folio() and its callers. If the second split of the straddler fails, the function reported success unconditionally, and the leftover incorrect end position could cause the truncate loop to drop that valid tail. After rework, it tells the caller the exact page range safe to discard via new pstart/pend out-parameters, so the truncate loop never touches a straddling folio that still holds valid out-of-range data. Patch 04 clarifies the return value semantics to "at least one split succeeded", which is all the shmem caller needs to decide whether to reset its scan loop. The second patch fixes a pre-existing race issue that is reachable today, so it is Cc'd to stable. Patches 01 and 03 require a dirty large folio that carries no filesystem private data, so they are not reachable on current filesystems. They were found while developing the upcoming ext4 iomap buffered I/O path. [1] In addition, another two pre-existing issues were found during the development of this series: - On 32-bit systems with a non-zero min_order filesystem, truncating or any operation operation involving folio_next_index() on the folio containing MAX_LFS_FILESIZE can cause the index calculation to overflow, posing an unpredictable risk. [2] - In shmem, at the last call site of truncate_inode_partial_folio() in shmem_undo_range(), the restart-loop check is wrong, which leaves split sub-folios behind. [3] These two issues need to be fixed separately. This patch (of 4): When the mapping has a non-zero minimum folio order (min_order), folio_split() in truncate_inode_partial_folio() stops at min_order instead of order 0, so the sub-folio containing a split point stays aligned to 1 << min_order rather than to a single page. The original boundaries in truncate_inode_pages_range() were based on page granularity, so either boundary could land inside the min_order chunk at its edge, and the truncation loop would drop that whole chunk, valid out-of-range tail included. For example, a 64K (order-4) folio with min_order = 2 (16K) punched from offset 0 to 36K: split @p0 -> [p0-p3, p4-p7, p8-p15] # non-uniform, min_order folio2 = p8-p15 # straddles: p8 in range, p9-p15 tail valid 2nd split of folio2 -> [p8-p11, p12-p15] # success end(old) = p9 # BUG: p9 inside [p8-p11] loop truncates ... p8-p11 # p9-p11's valid tail is lost It has gone unnoticed so far for two reasons. A non-zero min_order is only used by filesystems with a block or sector size larger than the page size, and those either always write back the affected range before punching a hole or truncating, or they carry filesystem private data on dirty folios (e.g. buffer_head), which makes filemap_release_folio() fail and folio_split() abort with -EBUSY, so the folio is never split and the old start/end boundaries remain valid. The bug only becomes reachable on paths that truncate dirty large folios without prior writeback and without filesystem private data, such as the upcoming ext4 iomap buffered I/O path. Align both start (rounded up) and end (rounded down) to the mapping minimum folio order so they always fall on a folio boundary. Link: https://lore.kernel.org/20260928120833.3440834-2-yi.zhang@huaweicloud.com Link: https://lore.kernel.org/linux-fsdevel/a638a8fb-c184-4069-ae33-379ec12cd514@huaweicloud.com/ [1] Link: https://lore.kernel.org/linux-mm/5a454f2a-8ae2-491d-b903-750c945cfb9d@huaweicloud.com/ [2] Link: https://lore.kernel.org/linux-mm/5pthbyxtn7q6xi4fmkofvksmcjzfnujcw2g4fxmxjzfin5pbgf@zui3vcimb4cv/ [3] Fixes: e220917fa5077 ("mm: split a folio in minimum folio order chunks") Signed-off-by: Zhang Yi Signed-off-by: Andrew Morton Reported-by: Joanne Koong Closes: https://lore.kernel.org/linux-mm/CAJnrk1bQYUe6+1ryyJur5EEnZYrC+_5AYsy=OWzVRgD4202y1g@mail.gmail.com/ Suggested-by: Zi Yan Reviewed-by: Jan Kara Reviewed-by: Brian Foster Reviewed-by: Zi Yan Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Darrick J. Wong Cc: Cc: Yang Erkun Cc: Zhihao Cheng Cc: Kefeng Wang Cc: --- mm/truncate.c | 18 ++++++++++++------ 1 file changed, 12 insertions(+), 6 deletions(-) diff --git a/mm/truncate.c b/mm/truncate.c index b58ba940be4740..8a28f4a2126776 100644 --- a/mm/truncate.c +++ b/mm/truncate.c @@ -345,9 +345,11 @@ long mapping_evict_folio(struct address_space *mapping, struct folio *folio) * @lstart: offset from which to truncate * @lend: offset to which to truncate (inclusive) * - * Truncate the page cache, removing the pages that are between - * specified offsets (and zeroing out partial pages - * if lstart or lend + 1 is not page aligned). + * Truncate the page cache, removing the folios that are between specified + * offsets (and zeroing out partial folios if lstart or lend + 1 is not + * folio aligned). For mappings with a non-zero minimum folio order, the + * boundaries are aligned inwards to 1 << min_order so the edge sub-folio + * straddling the range is kept. * * Truncate takes two passes - the first pass is nonblocking. It will not * block on page locks and it will not block on writeback. The second pass @@ -366,6 +368,7 @@ long mapping_evict_folio(struct address_space *mapping, struct folio *folio) void truncate_inode_pages_range(struct address_space *mapping, loff_t lstart, uoff_t lend) { + pgoff_t min_nrpages = mapping_min_folio_nrpages(mapping); pgoff_t start; /* inclusive */ pgoff_t end; /* exclusive */ struct folio_batch fbatch; @@ -379,9 +382,8 @@ void truncate_inode_pages_range(struct address_space *mapping, return; /* - * 'start' and 'end' always covers the range of pages to be fully - * truncated. Partial pages are covered with 'partial_start' at the - * start of the range and 'partial_end' at the end of the range. + * 'start' and 'end' always covers the range of folios to be fully + * truncated, with both boundaries aligned inwards to 1 << min_order. * Note that 'end' is exclusive while 'lend' is inclusive. */ start = (lstart + PAGE_SIZE - 1) >> PAGE_SHIFT; @@ -395,6 +397,10 @@ void truncate_inode_pages_range(struct address_space *mapping, else end = (lend + 1) >> PAGE_SHIFT; + start = round_up(start, min_nrpages); + if (end != (pgoff_t)-1) + end = round_down(end, min_nrpages); + folio_batch_init(&fbatch); index = start; while (index < end && find_lock_entries(mapping, &index, end - 1, From fbdecb775aefd47a8898fb07961621b291cd01de Mon Sep 17 00:00:00 2001 From: Zhang Yi Date: Mon, 28 Sep 2026 20:08:31 +0800 Subject: [PATCH 0855/1012] mm/truncate: look up the end-edge straddler by index In truncate_inode_partial_folio(), after the first split at the start edge, folio_split() unlocks and drops the refcount of the after-split sub-folios. The sub-folio straddling the end of the truncation range is therefore unlocked and only transiently ref'd in the page cache while the code still derives it from a page pointer inside the original folio. Between the first split finishing and page_folio() resolving split_at2, that tail page can be reclaimed, freed and reallocated as a new large folio in the same mapping at a different file offset. folio2 then points at a folio that does not cover the end boundary, yet folio2->mapping == folio->mapping still holds, so the stale pointer passes the mapping check and folio_split_or_unmap() splits a folio at a wrong position (or, with a transient refcount, a use-after-free window opens between try_get and the split). __folio_split()'s own folio != page_folio(split_at) check cannot catch this either since split_at2 has been reallocated as part of the new folio, so page_folio(split_at2) resolves back to folio2. Look the straddler up by its page index instead. __filemap_get_folio() returns the folio currently covering the boundary, ref'd and locked, with the mapping validated under the lock, so the split target is always the real folio at the end edge. Link: https://lore.kernel.org/20260928120833.3440834-3-yi.zhang@huaweicloud.com Link: https://lore.kernel.org/linux-mm/DLGXT0ERY79Z.3C5DYVJVX6S9Z@nvidia.com/ Fixes: 7460b470a131 ("mm/truncate: use folio_split() in truncate operation") Signed-off-by: Zhang Yi Signed-off-by: Andrew Morton Reported-by: Sashiko Closes: https://sashiko.dev/#/message/20260916094500.C30061F00893%40smtp.kernel.org Suggested-by: Jan Kara Suggested-by: Zi Yan Reviewed-by: Jan Kara Reviewed-by: Brian Foster Acked-by: Zi Yan Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Joanne Koong Cc: Darrick J. Wong Cc: Cc: Yang Erkun Cc: Zhihao Cheng Cc: Kefeng Wang Cc: Cc: --- mm/truncate.c | 30 ++++++++++++++++-------------- 1 file changed, 16 insertions(+), 14 deletions(-) diff --git a/mm/truncate.c b/mm/truncate.c index 8a28f4a2126776..f252619ae6c39b 100644 --- a/mm/truncate.c +++ b/mm/truncate.c @@ -259,30 +259,32 @@ bool truncate_inode_partial_folio(struct folio *folio, loff_t start, loff_t end) * for shmem truncate */ struct folio *folio2; + pgoff_t end_idx; if (offset + length == size) goto no_split; - split_at2 = folio_page(folio, - PAGE_ALIGN_DOWN(offset + length) / PAGE_SIZE); - folio2 = page_folio(split_at2); - - if (!folio_try_get(folio2)) + /* + * After the first split at the start edge, the folio at the + * end edge may be freed and reused concurrently. + * __filemap_get_folio() looks up the straddler at end_idx + * and returns it locked and ref'd with the mapping + * validated. + */ + end_idx = (pos + offset + length) >> PAGE_SHIFT; + folio2 = __filemap_get_folio(folio->mapping, end_idx, + FGP_LOCK | FGP_NOWAIT, 0); + if (IS_ERR(folio2)) goto no_split; + /* make sure folio2 is large */ if (!folio_test_large(folio2)) goto out; - if (!folio_trylock(folio2)) - goto out; - - /* make sure folio2 is large and does not change its mapping */ - if (folio_test_large(folio2) && - folio2->mapping == folio->mapping) - folio_split_or_unmap(folio2, split_at2, min_order); - - folio_unlock(folio2); + split_at2 = folio_page(folio2, (end_idx - folio2->index)); + folio_split_or_unmap(folio2, split_at2, min_order); out: + folio_unlock(folio2); folio_put(folio2); no_split: return true; From de6c2601218766b522666c938f45496212f8601d Mon Sep 17 00:00:00 2001 From: Zhang Yi Date: Mon, 28 Sep 2026 20:08:32 +0800 Subject: [PATCH 0856/1012] mm/truncate: fix data loss when splitting straddling large folios fails truncate_inode_partial_folio() splits a large folio so that the caller's truncate loop can drop the in-range sub-folios while keeping the out-of-range tail. The first split at the punch start edge is non-uniform, which leaves the sub-folio at the truncation end edge as large as possible, this means it may still straddle the range, holding both zeroed in-range and valid out-of-range data. The function then attempts a second split at offset + length to isolate that tail. If the second split fails the straddling sub-folio stays merged. The function returned true unconditionally on all exit paths of the success block, telling the caller it was fully handled. The caller kept its default end and the truncate loop truncated every sub-folio below it, including the merged straddler, discarding the valid out-of-range tail. For example, a 4-page order-2 folio punched from offset 0 to the middle of the last page: truncate_inode_pages_range() truncate_inode_partial_folio() # same_folio == true 1st split at page0 -> [p0, p1, p2-3] # non-uniform, success folio2 = p2-3 # straddles: p2 zeroed, p3 tail valid 2nd split of folio2 fails / cannot lock return true # BUG: caller keeps default end end = 3 loop truncates p0, p1, p2-3 # p3's valid tail is lost This became reachable after commit 7460b470a131 ("mm/truncate: use folio_split() in truncate operation") replaced the atomic split_folio() with folio_split(), whose non-uniform split can partially split a folio and leave the end edge merged. It has gone unnoticed because a dirty large folio normally carries the filesystem's private data, for example buffer_head, so filemap_release_folio() fails on a dirty folio and folio_split() aborts with -EBUSY before any split, leaving the straddler safely unsplit. The bug is only reachable on paths that produce dirty large folios without filesystem private data, and it was caught on the upcoming ext4 iomap buffered I/O path when no ifs is attached. Rework the contract so the caller is told the folio range to discard: - Add pgoff_t *pstart and *pend out-parameters that receive the folio range fully covered by [lstart, lend] after any split (or none), aligned inwards to min_order, i.e. the folios wholly within the range and safe to discard. - Report a reliable end position to the caller. The straddler is looked up at an index aligned inwards to the mapping minimum folio order, and *pend is set to that boundary on success. If nothing covers the boundary, discarding up to it stays safe. If the straddler is locked by someone else, fall back to folio->index. This best-effort fallback may leave the in-range sub-folios to a later pass but never discards the out-of-range tail. If the straddler cannot be split, fall back to folio2->index so the caller keeps the out-of-range tail. - Rename the byte-range parameters start/end to lstart/lend to better express their semantics. Callers in truncate_inode_pages_range() and shmem_undo_range() pass &pstart for the folio at the start edge and &pend for the folio at the end edge, so the truncate loop drops exactly the fully covered pages and never touches a straddling folio that still holds valid out-of-range data. Link: https://lore.kernel.org/20260928120833.3440834-4-yi.zhang@huaweicloud.com Fixes: 7460b470a131 ("mm/truncate: use folio_split() in truncate operation") Signed-off-by: Zhang Yi Signed-off-by: Andrew Morton Suggested-by: Brian Foster Link: https://lore.kernel.org/linux-fsdevel/anH-WKA1coW6wtfG@bfoster/ Reviewed-by: Jan Kara Reviewed-by: Brian Foster Acked-by: Zi Yan Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Joanne Koong Cc: Darrick J. Wong Cc: Cc: Yang Erkun Cc: Zhihao Cheng Cc: Kefeng Wang Cc: --- mm/internal.h | 4 +-- mm/shmem.c | 13 +++---- mm/truncate.c | 93 +++++++++++++++++++++++++++++++++++---------------- 3 files changed, 71 insertions(+), 39 deletions(-) diff --git a/mm/internal.h b/mm/internal.h index 0434dfcfc36f14..ff28b940e0cd2c 100644 --- a/mm/internal.h +++ b/mm/internal.h @@ -625,8 +625,8 @@ unsigned find_lock_entries(struct address_space *mapping, pgoff_t *start, unsigned find_get_entries(struct address_space *mapping, pgoff_t *start, pgoff_t end, struct folio_batch *fbatch, pgoff_t *indices); int truncate_inode_folio(struct address_space *mapping, struct folio *folio); -bool truncate_inode_partial_folio(struct folio *folio, loff_t start, - loff_t end); +bool truncate_inode_partial_folio(struct folio *folio, loff_t lstart, + loff_t lend, pgoff_t *pstart, pgoff_t *pend); long mapping_evict_folio(struct address_space *mapping, struct folio *folio); unsigned long mapping_try_invalidate(struct address_space *mapping, pgoff_t start, pgoff_t end, unsigned long *nr_failed); diff --git a/mm/shmem.c b/mm/shmem.c index 07b2855dfb7bd9..ae08cff4500c3b 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -1386,11 +1386,8 @@ static void shmem_undo_range(struct inode *inode, loff_t lstart, uoff_t lend, if (folio) { same_folio = lend < folio_next_pos(folio); folio_mark_dirty(folio); - if (!truncate_inode_partial_folio(folio, lstart, lend)) { - start = folio_next_index(folio); - if (same_folio) - end = folio->index; - } + truncate_inode_partial_folio(folio, lstart, lend, &start, + same_folio ? &end : NULL); folio_unlock(folio); folio_put(folio); folio = NULL; @@ -1400,8 +1397,7 @@ static void shmem_undo_range(struct inode *inode, loff_t lstart, uoff_t lend, folio = shmem_get_partial_folio(inode, lend >> PAGE_SHIFT); if (folio) { folio_mark_dirty(folio); - if (!truncate_inode_partial_folio(folio, lstart, lend)) - end = folio->index; + truncate_inode_partial_folio(folio, lstart, lend, NULL, &end); folio_unlock(folio); folio_put(folio); } @@ -1469,7 +1465,8 @@ static void shmem_undo_range(struct inode *inode, loff_t lstart, uoff_t lend, if (!folio_test_large(folio)) { truncate_inode_folio(mapping, folio); - } else if (truncate_inode_partial_folio(folio, lstart, lend)) { + } else if (truncate_inode_partial_folio(folio, + lstart, lend, NULL, NULL)) { /* * If we split a page, reset the loop so * that we pick up the new sub pages. diff --git a/mm/truncate.c b/mm/truncate.c index f252619ae6c39b..9eb1087691490e 100644 --- a/mm/truncate.c +++ b/mm/truncate.c @@ -206,30 +206,43 @@ static int folio_split_or_unmap(struct folio *folio, struct page *split_at, /* * Handle partial folios. The folio may be entirely within the * range if a split has raced with us. If not, we zero the part of the - * folio that's within the [start, end] range, and then split the folio if + * folio that's within the [lstart, lend] range, and then split the folio if * it's large. split_page_range() will discard pages which now lie beyond * i_size, and we rely on the caller to discard pages which lie within a * newly created hole. * + * When @pstart and/or @pend are non-NULL they receive the indexes of the + * folio range fully covered by [lstart, lend] after any split (or none), + * aligned inwards to min_order, i.e. the range of folios wholly within + * [lstart, lend] and so safe to discard. + * * Returns false if splitting failed so the caller can avoid * discarding the entire folio which is stubbornly unsplit. */ -bool truncate_inode_partial_folio(struct folio *folio, loff_t start, loff_t end) +bool truncate_inode_partial_folio(struct folio *folio, loff_t lstart, + loff_t lend, pgoff_t *pstart, pgoff_t *pend) { loff_t pos = folio_pos(folio); size_t size = folio_size(folio); unsigned int offset, length; struct page *split_at, *split_at2; + unsigned long min_nrbytes; unsigned int min_order; - if (pos < start) - offset = start - pos; + if (pos < lstart) + offset = lstart - pos; else offset = 0; - if (pos + size <= (u64)end) + if (pos + size <= (u64)lend) length = size - offset; else - length = end + 1 - pos - offset; + length = lend + 1 - pos - offset; + + if (pstart) + *pstart = offset ? folio_next_index(folio) : folio->index; + if (pend) + *pend = (pos + size > (u64)lend) ? folio->index : + folio_next_index(folio); folio_wait_writeback(folio); if (length == size) { @@ -247,10 +260,12 @@ bool truncate_inode_partial_folio(struct folio *folio, loff_t start, loff_t end) if (folio_needs_release(folio)) folio_invalidate(folio, offset, length); - if (!folio_test_large(folio)) - return true; min_order = mapping_min_folio_order(folio->mapping); + min_nrbytes = mapping_min_folio_nrbytes(folio->mapping); + if (folio_order(folio) == min_order) + return true; + split_at = folio_page(folio, PAGE_ALIGN_DOWN(offset) / PAGE_SIZE); if (!folio_split_or_unmap(folio, split_at, min_order)) { /* @@ -259,34 +274,57 @@ bool truncate_inode_partial_folio(struct folio *folio, loff_t start, loff_t end) * for shmem truncate */ struct folio *folio2; - pgoff_t end_idx; + pgoff_t end, aligned_end = round_down(pos + offset + length, + min_nrbytes) >> PAGE_SHIFT; + if (pstart) + *pstart = round_up(pos + offset, + min_nrbytes) >> PAGE_SHIFT; + + end = aligned_end; if (offset + length == size) - goto no_split; + goto out; /* * After the first split at the start edge, the folio at the * end edge may be freed and reused concurrently. - * __filemap_get_folio() looks up the straddler at end_idx + * __filemap_get_folio() looks up the straddler at aligned_end * and returns it locked and ref'd with the mapping * validated. */ - end_idx = (pos + offset + length) >> PAGE_SHIFT; - folio2 = __filemap_get_folio(folio->mapping, end_idx, + folio2 = __filemap_get_folio(folio->mapping, aligned_end, FGP_LOCK | FGP_NOWAIT, 0); - if (IS_ERR(folio2)) - goto no_split; - - /* make sure folio2 is large */ - if (!folio_test_large(folio2)) + if (IS_ERR(folio2)) { + /* + * No sub-folio straddles the boundary when + * aligned_end is empty, so discarding up to it is + * safe. Otherwise the straddler is locked by + * someone else and we cannot obtain a reliable end + * position, so we fall back to folio->index, which + * is safe but leaves the sub-folios split off at + * the offset edge in the page cache. + */ + if (PTR_ERR(folio2) != -ENOENT) + end = folio->index; goto out; + } - split_at2 = folio_page(folio2, (end_idx - folio2->index)); - folio_split_or_unmap(folio2, split_at2, min_order); -out: + /* Already at the minimum order, nothing to split */ + if (folio_order(folio2) == min_order) + goto out_put; + + split_at2 = folio_page(folio2, (aligned_end - folio2->index)); + + /* Split failed, keep the straddler intact */ + if (folio_split_or_unmap(folio2, split_at2, min_order)) + end = folio2->index; + +out_put: folio_unlock(folio2); folio_put(folio2); -no_split: +out: + if (pend) + *pend = end; return true; } if (folio_test_dirty(folio)) @@ -421,11 +459,8 @@ void truncate_inode_pages_range(struct address_space *mapping, folio = __filemap_get_folio(mapping, lstart >> PAGE_SHIFT, FGP_LOCK, 0); if (!IS_ERR(folio)) { same_folio = lend < folio_next_pos(folio); - if (!truncate_inode_partial_folio(folio, lstart, lend)) { - start = folio_next_index(folio); - if (same_folio) - end = folio->index; - } + truncate_inode_partial_folio(folio, lstart, lend, &start, + same_folio ? &end : NULL); folio_unlock(folio); folio_put(folio); folio = NULL; @@ -435,8 +470,8 @@ void truncate_inode_pages_range(struct address_space *mapping, folio = __filemap_get_folio(mapping, lend >> PAGE_SHIFT, FGP_LOCK, 0); if (!IS_ERR(folio)) { - if (!truncate_inode_partial_folio(folio, lstart, lend)) - end = folio->index; + truncate_inode_partial_folio(folio, lstart, lend, + NULL, &end); folio_unlock(folio); folio_put(folio); } From babcaeecf042bec615e17c8f8b52824b7105a53f Mon Sep 17 00:00:00 2001 From: Zhang Yi Date: Mon, 28 Sep 2026 20:08:33 +0800 Subject: [PATCH 0857/1012] mm/truncate: clarify return value of truncate_inode_partial_folio() With the earlier rework the callers no longer rely on the return value of truncate_inode_partial_folio() to decide whether to adjust the truncation range. The pstart/pend out-parameters carry that information instead. The callers now only use the return value as a flag indicating whether the loop should be reset to pick up newly split sub-folios on the shmem path. Return true if at least one split succeeded, and false otherwise. This clarifies the existing confusing return value semantics. Link: https://lore.kernel.org/20260928120833.3440834-5-yi.zhang@huaweicloud.com Signed-off-by: Zhang Yi Signed-off-by: Andrew Morton Reviewed-by: Jan Kara Reviewed-by: Brian Foster Acked-by: Zi Yan Cc: David Hildenbrand Cc: Lorenzo Stoakes Cc: Liam R. Howlett Cc: Vlastimil Babka Cc: Mike Rapoport Cc: Suren Baghdasaryan Cc: Michal Hocko Cc: Hugh Dickins Cc: Baolin Wang Cc: Matthew Wilcox (Oracle) Cc: Joanne Koong Cc: Darrick J. Wong Cc: Cc: Yang Erkun Cc: Zhihao Cheng Cc: Kefeng Wang Cc: --- mm/truncate.c | 14 ++++++-------- 1 file changed, 6 insertions(+), 8 deletions(-) diff --git a/mm/truncate.c b/mm/truncate.c index 9eb1087691490e..5b1b13cf8ff0a3 100644 --- a/mm/truncate.c +++ b/mm/truncate.c @@ -216,8 +216,7 @@ static int folio_split_or_unmap(struct folio *folio, struct page *split_at, * aligned inwards to min_order, i.e. the range of folios wholly within * [lstart, lend] and so safe to discard. * - * Returns false if splitting failed so the caller can avoid - * discarding the entire folio which is stubbornly unsplit. + * Return %true if at least one split succeeded, %false otherwise. */ bool truncate_inode_partial_folio(struct folio *folio, loff_t lstart, loff_t lend, pgoff_t *pstart, pgoff_t *pend) @@ -247,7 +246,7 @@ bool truncate_inode_partial_folio(struct folio *folio, loff_t lstart, folio_wait_writeback(folio); if (length == size) { truncate_inode_folio(folio->mapping, folio); - return true; + return false; } /* @@ -264,7 +263,7 @@ bool truncate_inode_partial_folio(struct folio *folio, loff_t lstart, min_order = mapping_min_folio_order(folio->mapping); min_nrbytes = mapping_min_folio_nrbytes(folio->mapping); if (folio_order(folio) == min_order) - return true; + return false; split_at = folio_page(folio, PAGE_ALIGN_DOWN(offset) / PAGE_SIZE); if (!folio_split_or_unmap(folio, split_at, min_order)) { @@ -327,10 +326,9 @@ bool truncate_inode_partial_folio(struct folio *folio, loff_t lstart, *pend = end; return true; } - if (folio_test_dirty(folio)) - return false; - truncate_inode_folio(folio->mapping, folio); - return true; + if (!folio_test_dirty(folio)) + truncate_inode_folio(folio->mapping, folio); + return false; } /* From 11ac78805d39a839115af6b32661981d62199723 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Mon, 28 Sep 2026 01:58:33 -0700 Subject: [PATCH 0858/1012] mm/damon/core: preserve the quota passed to damon_new_scheme() Patch series "mm/damon/core: preserve quota state when constructing schemes", v3. damon_commit_ctx() first commits the running context's parameters to a temporary context for validating proposed updates. Constructing the temporary schemes clears the running schemes' quota state because damon_new_scheme() initializes the quota passed as a parameter before copying it to the new scheme. Even an update later rejected with -EINVAL loses the running quota state. Initialize the new scheme's copy instead, and add KUnit tests for the constructor and for accepted and rejected context updates. Note: below are test results and changelogs that could be removed from the final commit log. In the v1 live test with damo, a scheme with a plain 64 KiB size quota and a 60-second reset interval uses its quota, and a full "damo tune" with unchanged parameters then lets it try another 64 KiB within the same window. With the fix, sz_tried stays at 64 KiB. KUnit was rerun after the rebase on x86-64 and i386. With only patch 2 applied, the two new tests fail. With the fix, all 46 DAMON KUnit tests pass on both architectures. The v1 DAMON selftests showed no new failures (QEMU TCG guest; the wss_estimation test missed its accuracy bounds with and without the fix). This patch (of 2): damon_commit_ctx() first commits the running context's parameters to a temporary context for validating proposed updates. When damon_commit_schemes() creates the temporary schemes, it passes the running scheme's quota as the quota parameter of damon_new_scheme(). damon_new_scheme() calls damos_quota_init() on that quota before copying it to the new scheme. This clears the running scheme's effective quota, feedback input and charging state. Even an update rejected with -EINVAL loses the running quota state. For a size quota, this discards the bytes already charged and allows the scheme to use a fresh quota before the reset interval has elapsed. For a goal-driven quota, the consist tuner loses its accumulated input and restarts from its minimum input. A time quota loses its throughput estimate and falls back to the initial estimate. The end users will show DAMOS works more or less aggressively than expected for online-commit updates of quotas. DAMON provides best efforts by default. DAMON parameters online commit is supposed to be executed only occasionally. Hence, the issue wouldn't be critical on sane setups. For user_input type quota goals, online commit of the user input score is expected to be frequent. But, for the case commit_schemes_quota_goals command is recommended for optimal execution, and it doesn't have this bug. Commit 60bd24f272d0 ("mm/damon/sysfs: test commit input against realistic destination") introduced this problem in v6.19 when sysfs validation began committing the running context's parameters to a temporary context. Commit b90408ef1163 ("mm/damon/core: safely validate src on damon_commit_ctx()") later moved that validation into the core API, exposing other callers including DAMON_RECLAIM and DAMON_LRU_SORT. Sashiko reported the same side effect [1] on the RFC of the core API change. Copy the quota to the new scheme first, then initialize that copy. Make damos_quota_init() return void, since its return value is no longer needed. Link: https://lore.kernel.org/20260928085835.7675-1-sj@kernel.org Link: https://lore.kernel.org/20260928085835.7675-2-sj@kernel.org Link: https://lore.kernel.org/r/20260702212143.0CB6D1F00A3D@smtp.kernel.org/ [1] Fixes: 60bd24f272d0 ("mm/damon/sysfs: test commit input against realistic destination") Signed-off-by: Karl Mehltretter Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Assisted-by: LLM Cc: Bijan Tabatabai Cc: --- mm/damon/core.c | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 733025b367457b..3a61e5fdfb625d 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -734,7 +734,7 @@ static bool damos_quota_goals_empty(struct damos_quota *q) } /* initialize fields of @quota that normally API users wouldn't set */ -static struct damos_quota *damos_quota_init(struct damos_quota *quota) +static void damos_quota_init(struct damos_quota *quota) { quota->esz = 0; quota->total_charged_sz = 0; @@ -744,7 +744,6 @@ static struct damos_quota *damos_quota_init(struct damos_quota *quota) quota->charge_target_from = NULL; quota->charge_addr_from = 0; quota->esz_bp = 0; - return quota; } struct damos *damon_new_scheme(struct damos_access_pattern *pattern, @@ -776,7 +775,8 @@ struct damos *damon_new_scheme(struct damos_access_pattern *pattern, scheme->last_applied = NULL; INIT_LIST_HEAD(&scheme->list); - scheme->quota = *(damos_quota_init(quota)); + scheme->quota = *quota; + damos_quota_init(&scheme->quota); /* quota.goals should be separately set by caller */ INIT_LIST_HEAD(&scheme->quota.goals); From 921fd80b97ee45801df3c99ebac8cfd2c5eaaffc Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Mon, 28 Sep 2026 01:58:34 -0700 Subject: [PATCH 0859/1012] mm/damon/tests/core-kunit: test preservation of quota state Check that damon_new_scheme() initializes the new scheme's quota without changing the quota passed as a parameter. Cover all eight fields initialized by damos_quota_init(). Also check that damon_commit_ctx() preserves those fields in the destination scheme for both accepted and rejected parameter updates. Use an invalid min_region_sz for the rejected update and confirm that returning -EINVAL leaves the running quota state unchanged. Without the preceding fix, all eight fields are cleared in the constructor test and in both context update cases. Link: https://lore.kernel.org/20260928085835.7675-3-sj@kernel.org Signed-off-by: Karl Mehltretter Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Assisted-by: LLM Cc: Bijan Tabatabai Cc: Brendan Higgins Cc: David Gow --- mm/damon/tests/core-kunit.h | 107 ++++++++++++++++++++++++++++++++++++ 1 file changed, 107 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index df84d9cc7d204a..ce17f44583087e 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -812,6 +812,48 @@ static void damos_test_new_filter(struct kunit *test) damos_destroy_filter(filter); } +static void damos_test_new_scheme_keeps_src_quota(struct kunit *test) +{ + struct damos_access_pattern pattern = {}; + struct damon_target target = {}; + struct damos_quota quota = { + .sz = SZ_64K, + .esz = 123, + .esz_bp = 456, + .total_charged_sz = 789, + .total_charged_ns = 1011, + .charged_sz = 12, + .charged_from = 13, + .charge_target_from = &target, + .charge_addr_from = 14, + }; + struct damos_watermarks wmarks = {}; + struct damos *s; + + s = damon_new_scheme(&pattern, DAMOS_STAT, 0, "a, &wmarks, + NUMA_NO_NODE); + if (!s) + kunit_skip(test, "scheme alloc fail"); + KUNIT_EXPECT_EQ(test, s->quota.sz, (unsigned long)SZ_64K); + KUNIT_EXPECT_EQ(test, s->quota.esz, 0ul); + KUNIT_EXPECT_EQ(test, s->quota.esz_bp, 0ul); + KUNIT_EXPECT_EQ(test, s->quota.total_charged_sz, 0ul); + KUNIT_EXPECT_EQ(test, s->quota.total_charged_ns, 0ul); + KUNIT_EXPECT_EQ(test, s->quota.charged_sz, 0ul); + KUNIT_EXPECT_EQ(test, s->quota.charged_from, 0ul); + KUNIT_EXPECT_PTR_EQ(test, s->quota.charge_target_from, NULL); + KUNIT_EXPECT_EQ(test, s->quota.charge_addr_from, 0ul); + KUNIT_EXPECT_EQ(test, quota.esz, 123ul); + KUNIT_EXPECT_EQ(test, quota.esz_bp, 456ul); + KUNIT_EXPECT_EQ(test, quota.total_charged_sz, 789ul); + KUNIT_EXPECT_EQ(test, quota.total_charged_ns, 1011ul); + KUNIT_EXPECT_EQ(test, quota.charged_sz, 12ul); + KUNIT_EXPECT_EQ(test, quota.charged_from, 13ul); + KUNIT_EXPECT_PTR_EQ(test, quota.charge_target_from, &target); + KUNIT_EXPECT_EQ(test, quota.charge_addr_from, 14ul); + damon_destroy_scheme(s); +} + static void damos_test_commit_quota_goal_for(struct kunit *test, struct damos_quota_goal *dst, struct damos_quota_goal *src) @@ -1572,6 +1614,69 @@ static void damon_test_commit_ctx(struct kunit *test) damon_destroy_ctx(dst); } +static void damon_test_commit_ctx_keeps_quota_for(struct kunit *test, + unsigned long min_region_sz, int expected_err) +{ + struct damos_access_pattern pattern = {}; + struct damos_quota quota = {.sz = SZ_64K}; + struct damos_watermarks wmarks = {}; + struct damon_ctx *src, *dst; + struct damon_target *target; + struct damos *s; + + dst = damon_new_ctx(); + if (!dst) + kunit_skip(test, "dst alloc fail"); + target = damon_new_target(); + if (!target) { + damon_destroy_ctx(dst); + kunit_skip(test, "target alloc fail"); + } + damon_add_target(dst, target); + s = damon_new_scheme(&pattern, DAMOS_STAT, 0, "a, &wmarks, + NUMA_NO_NODE); + if (!s) { + damon_destroy_ctx(dst); + kunit_skip(test, "scheme alloc fail"); + } + damon_add_scheme(dst, s); + + /* Copy the parameters before populating dst's runtime quota state. */ + src = damon_new_test_ctx(dst); + if (!src) { + damon_destroy_ctx(dst); + kunit_skip(test, "src alloc fail"); + } + src->min_region_sz = min_region_sz; + s->quota.esz = 123; + s->quota.esz_bp = 456; + s->quota.total_charged_sz = 789; + s->quota.total_charged_ns = 1011; + s->quota.charged_sz = 12; + s->quota.charged_from = 13; + s->quota.charge_target_from = target; + s->quota.charge_addr_from = 14; + + KUNIT_EXPECT_EQ(test, damon_commit_ctx(dst, src), expected_err); + KUNIT_EXPECT_EQ(test, s->quota.esz, 123ul); + KUNIT_EXPECT_EQ(test, s->quota.esz_bp, 456ul); + KUNIT_EXPECT_EQ(test, s->quota.total_charged_sz, 789ul); + KUNIT_EXPECT_EQ(test, s->quota.total_charged_ns, 1011ul); + KUNIT_EXPECT_EQ(test, s->quota.charged_sz, 12ul); + KUNIT_EXPECT_EQ(test, s->quota.charged_from, 13ul); + KUNIT_EXPECT_PTR_EQ(test, s->quota.charge_target_from, target); + KUNIT_EXPECT_EQ(test, s->quota.charge_addr_from, 14ul); + damon_destroy_ctx(src); + damon_destroy_ctx(dst); +} + +static void damon_test_commit_ctx_keeps_quota(struct kunit *test) +{ + /* Only power of two min_region_sz is allowed. */ + damon_test_commit_ctx_keeps_quota_for(test, 4096, 0); + damon_test_commit_ctx_keeps_quota_for(test, 4095, -EINVAL); +} + static void damon_test_valid_probe_params(struct kunit *test) { struct damon_ctx *ctx; @@ -2259,6 +2364,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damon_test_mvsum), KUNIT_CASE(damon_test_nr_accesses_mvsum), KUNIT_CASE(damos_test_new_filter), + KUNIT_CASE(damos_test_new_scheme_keeps_src_quota), KUNIT_CASE(damos_test_commit_quota_goal), KUNIT_CASE(damos_test_commit_quota_goals), KUNIT_CASE(damos_test_commit_quota), @@ -2270,6 +2376,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damon_test_commit_filter), KUNIT_CASE(damon_test_commit_probes), KUNIT_CASE(damon_test_commit_ctx), + KUNIT_CASE(damon_test_commit_ctx_keeps_quota), KUNIT_CASE(damon_test_valid_probe_params), KUNIT_CASE(damos_test_filter_out), KUNIT_CASE(damos_test_apply_scheme_filtered_sz), From d2aa5e88ee03892c8a8b3d5057c21ef1080850bf Mon Sep 17 00:00:00 2001 From: Donggeun Yoo Date: Mon, 28 Sep 2026 01:48:14 -0700 Subject: [PATCH 0860/1012] mm/damon/core: prevent size quota overflow in the temporal goal tuner Patch series "mm/damon: fix the temporal goal tuner's size quota conversion", v5. damos_goal_tune_esz_bp_temporal() hands the size quota to damos_set_effective_quota() through quota->esz_bp in basis points, and the multiply that gets it there is unchecked. On 32-bit it wraps above 429496 bytes, and a wrapped product below 10000 divides to a zero effective quota. damos_quota_is_full() is then true on the first test of every charge window, so the scheme makes no progress for as long as the goal is unachieved. Patch 1 bounds the conversion. Patch 2 pins the boundary in the core kunit suite, where the new test would fail without patch 1 on any word size. This patch (of 2): damos_goal_tune_esz_bp_temporal() converts the scheme's size quota into basis points with "quota->esz_bp = quota->sz * 10000", both unsigned long, and damos_set_effective_quota() divides the result back by 10000. quotas/bytes is unbounded; bytes_store() hands it to kstrtoul() as is. On 32-bit the product wraps for any size quota above ULONG_MAX / 10000, that is 429496 bytes. A wrapped product below 10000 divides to a zero effective quota: 429497 gives 0. damos_quota_is_full() is then true on the first test of every charge window. Other wrapped values are wrong without being zero: 500000 gives 70503. Triggering this needs a scheme with a quota goal, the temporal goal tuner, and a size quota above ULONG_MAX / 10000 -- 429496 bytes on 32-bit, 1844674407370955 on 64-bit. The scheme then makes no progress for as long as the goal is unachieved, which is easy to notice, and writing a smaller size quota restores it. Nothing is corrupted and nothing leaks. This is unlikely to be hit on a tested setup. addr_unit does not cover this. It only scales the numbers a paddr context writes to quotas/bytes, so a large enough scaled value wraps just the same, and vaddr and fvaddr contexts take raw byte values. Bound the multiply. Link: https://lore.kernel.org/20260928084816.5575-1-sj@kernel.org Link: https://lore.kernel.org/20260928084816.5575-2-sj@kernel.org Fixes: af738a6a00c1 ("mm/damon/core: introduce DAMOS_QUOTA_GOAL_TUNER_TEMPORAL") Signed-off-by: Donggeun Yoo Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: # 7.1.x --- mm/damon/core.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 3a61e5fdfb625d..46ef570a4766fd 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3291,7 +3291,7 @@ static void damos_goal_tune_esz_bp_temporal(struct damon_ctx *c, if (score >= 10000) quota->esz_bp = 0; - else if (quota->sz) + else if (quota->sz && quota->sz <= ULONG_MAX / 10000) quota->esz_bp = quota->sz * 10000; else quota->esz_bp = ULONG_MAX; From 5aec30c1961cc7d835ce2a39a9ae569bf34e550c Mon Sep 17 00:00:00 2001 From: Donggeun Yoo Date: Mon, 28 Sep 2026 01:48:15 -0700 Subject: [PATCH 0861/1012] mm/damon/tests/core-kunit: test the temporal tuner's size quota conversion damos_goal_tune_esz_bp_temporal() encodes the size quota in basis points, so the conversion is exact only up to ULONG_MAX / 10000. Pin the three sizes around that boundary: the largest one that fits, the first one that does not, and ULONG_MAX. Link: https://lore.kernel.org/20260928084816.5575-3-sj@kernel.org Signed-off-by: Donggeun Yoo Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Brendan Higgins Cc: David Gow --- mm/damon/tests/core-kunit.h | 48 +++++++++++++++++++++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index ce17f44583087e..a8e22246facf96 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -2347,6 +2347,53 @@ static void damon_test_rand(struct kunit *test) } } +static void damos_test_esz_goal_temporal(struct kunit *test) +{ + struct damos_access_pattern pattern = {}; + struct damos_watermarks wmarks = {}; + struct damos_quota quota = { + .goal_tuner = DAMOS_QUOTA_GOAL_TUNER_TEMPORAL, + }; + struct damos_quota_goal *goal; + struct damon_ctx *ctx; + struct damos *s; + + ctx = damon_new_ctx(); + KUNIT_ASSERT_NOT_NULL(test, ctx); + + s = damon_new_scheme(&pattern, DAMOS_STAT, 0, "a, &wmarks, + NUMA_NO_NODE); + if (!s) { + damon_destroy_ctx(ctx); + kunit_skip(test, "scheme alloc fail"); + } + damon_add_scheme(ctx, s); + + goal = damos_new_quota_goal(DAMOS_QUOTA_USER_INPUT, 10000); + if (!goal) { + damon_destroy_ctx(ctx); + kunit_skip(test, "quota goal alloc fail"); + } + goal->current_value = 0; + damos_add_quota_goal(&s->quota, goal); + + /* The largest size quota the basis-point conversion can hold. */ + s->quota.sz = ULONG_MAX / 10000; + damos_set_effective_quota(ctx, s); + KUNIT_EXPECT_EQ(test, s->quota.esz, ULONG_MAX / 10000); + + /* Any larger one saturates instead of wrapping. */ + s->quota.sz = ULONG_MAX / 10000 + 1; + damos_set_effective_quota(ctx, s); + KUNIT_EXPECT_EQ(test, s->quota.esz, ULONG_MAX / 10000); + + s->quota.sz = ULONG_MAX; + damos_set_effective_quota(ctx, s); + KUNIT_EXPECT_EQ(test, s->quota.esz, ULONG_MAX / 10000); + + damon_destroy_ctx(ctx); +} + static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damon_test_target), KUNIT_CASE(damon_test_regions), @@ -2388,6 +2435,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damon_test_is_last_region), KUNIT_CASE(damon_test_walk_control_obsolete), KUNIT_CASE(damon_test_rand), + KUNIT_CASE(damos_test_esz_goal_temporal), {}, }; From d13b62f27cb2f78d8e3fe979664fbf2882aea5a6 Mon Sep 17 00:00:00 2001 From: "Zenghui Yu (Huawei)" Date: Mon, 28 Sep 2026 01:39:55 -0700 Subject: [PATCH 0862/1012] mm/damon/api: remove unused NR_DAMOS_* enumerators Patch series "mm/damon/core: cleanup code, reduce stack usage, and add kunit". Three various improvements. Patch 1 from Zenghui Yu (Huawei) cleans up unused enums. Patch 2 from Arnd Bergmann reduces kdamond's stack usage. Patch 3 from Karl Mehltretter adds a kunit test for uninitialized PSI quota goal stage logic. This patch (of 3): NR_DAMOS_ACTIONS, NR_DAMOS_QUOTA_GOAL_METRICS, and NR_DAMOS_WMARK_METRICS are not referenced by any code. Remove them. They can be reintroduced if a real user comes up. Link: https://lore.kernel.org/20260928083959.4030-1-sj@kernel.org Link: https://lore.kernel.org/20260928083959.4030-2-sj@kernel.org Signed-off-by: Zenghui Yu (Huawei) Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park --- include/linux/damon.h | 8 -------- 1 file changed, 8 deletions(-) diff --git a/include/linux/damon.h b/include/linux/damon.h index 844b175120f09b..6a29dc2ac8db79 100644 --- a/include/linux/damon.h +++ b/include/linux/damon.h @@ -117,7 +117,6 @@ struct damon_target { * @DAMOS_MIGRATE_HOT: Migrate the regions prioritizing warmer regions. * @DAMOS_MIGRATE_COLD: Migrate the regions prioritizing colder regions. * @DAMOS_STAT: Do nothing but count the stat. - * @NR_DAMOS_ACTIONS: Total number of DAMOS actions * * The support of each action is up to running &struct damon_operations. * Refer to 'Operation Action' section of Documentation/mm/damon/design.rst for @@ -137,7 +136,6 @@ enum damos_action { DAMOS_MIGRATE_HOT, DAMOS_MIGRATE_COLD, DAMOS_STAT, /* Do nothing but only record the stat */ - NR_DAMOS_ACTIONS, }; /** @@ -154,9 +152,6 @@ enum damos_action { * @DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP: Scheme-eligible memory ratio of a * node in basis points (0-10000). * @DAMOS_QUOTA_HUGEPAGE_MEM_BP: Huge page to total used memory ratio. - * @NR_DAMOS_QUOTA_GOAL_METRICS: Number of DAMOS quota goal metrics. - * - * Metrics equal to larger than @NR_DAMOS_QUOTA_GOAL_METRICS are unsupported. */ enum damos_quota_goal_metric { DAMOS_QUOTA_USER_INPUT, @@ -169,7 +164,6 @@ enum damos_quota_goal_metric { DAMOS_QUOTA_INACTIVE_MEM_BP, DAMOS_QUOTA_NODE_ELIGIBLE_MEM_BP, DAMOS_QUOTA_HUGEPAGE_MEM_BP, - NR_DAMOS_QUOTA_GOAL_METRICS, }; /** @@ -313,12 +307,10 @@ struct damos_quota { * * @DAMOS_WMARK_NONE: Ignore the watermarks of the given scheme. * @DAMOS_WMARK_FREE_MEM_RATE: Free memory rate of the system in [0,1000]. - * @NR_DAMOS_WMARK_METRICS: Total number of DAMOS watermark metrics */ enum damos_wmark_metric { DAMOS_WMARK_NONE, DAMOS_WMARK_FREE_MEM_RATE, - NR_DAMOS_WMARK_METRICS, }; /** From 07ce30dd154562f67bb4fd4ae59ea5576c32e77a Mon Sep 17 00:00:00 2001 From: Arnd Bergmann Date: Mon, 28 Sep 2026 01:39:56 -0700 Subject: [PATCH 0863/1012] mm/damon/core: reduce stack usage further In a previous patch, I had annotated kdamond_tune_intervals() as noinline_for_stack in order to not exceed the stack frame warning limit. Unfortunately, my current linux-next randconfig builds show a similar problem again with clang-21: mm/damon/core.c:3953:12: error: stack frame size (1288) exceeds limit (1280) in 'kdamond_fn' [-Werror,-Wframe-larger-than] 3953 | static int kdamond_fn(void *data) Do the same thing with kdamond_apply_schemes(), kdamond_merge_regions(), and kdamond_split_regions(), which also have individually large stacks. This should work to keep the deepest total stack depth down much more as well as avoid the warning. I also tried to reduce the complexity of kdamond_fn() itself by splitting out the while() loop into a separate function. While this arguably led to slightly more readable code, it had no effect on the total stack usage and just made the new function the largest stack user and had a nonzero risk of me getting the conversion wrong. Link: https://lore.kernel.org/20260928083959.4030-3-sj@kernel.org Fixes: 5a00cae64de1 ("mm/damon/core: reduce kernel stack usage") Signed-off-by: Arnd Bergmann Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Cc: Nick Desaulniers Cc: Bill Wendling Cc: Justin Stitt Cc: Ravi Jonnalagadda Cc: Nathan Chancellor --- mm/damon/core.c | 9 +++++---- 1 file changed, 5 insertions(+), 4 deletions(-) diff --git a/mm/damon/core.c b/mm/damon/core.c index 46ef570a4766fd..5ecbea5d71e1dc 100644 --- a/mm/damon/core.c +++ b/mm/damon/core.c @@ -3433,7 +3433,7 @@ static void damos_trace_stat(struct damon_ctx *c, struct damos *s) trace_call__damos_stat_after_apply_interval(cidx, sidx, &s->stat); } -static void kdamond_apply_schemes(struct damon_ctx *c) +static noinline_for_stack void kdamond_apply_schemes(struct damon_ctx *c) { struct damon_target *t; struct damos *s; @@ -3589,8 +3589,9 @@ static void damon_merge_regions_of(struct damon_target *t, unsigned int thres, * while DAMON is running. For such a case, repeat merging until the limit is * met while increasing @threshold up to possible maximum level. */ -static void kdamond_merge_regions(struct damon_ctx *c, unsigned int threshold, - unsigned long sz_limit) +static noinline_for_stack void kdamond_merge_regions(struct damon_ctx *c, + unsigned int threshold, + unsigned long sz_limit) { struct damon_target *t; unsigned int nr_regions; @@ -3734,7 +3735,7 @@ static void damon_split_some_regions(struct damon_ctx *ctx, * split was unnecessarily made, later 'kdamond_merge_regions()' will revert * it. */ -static void kdamond_split_regions(struct damon_ctx *ctx) +static noinline_for_stack void kdamond_split_regions(struct damon_ctx *ctx) { struct damon_target *t; unsigned long nr_regions = 0; From 1ae326d01939896143e00ad2033836493330f96f Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Mon, 28 Sep 2026 01:39:57 -0700 Subject: [PATCH 0864/1012] mm/damon/tests/core-kunit: test PSI goal values with explicit samples Test the PSI current-value helper with explicit totals so the result does not depend on the test system's memory pressure. Cover an unmeasured consist goal, unmeasured temporal goals with zero and nonzero effective quotas, and measured rounds for both tuners. Check last_psi_total after each call. Link: https://lore.kernel.org/20260928083959.4030-4-sj@kernel.org Signed-off-by: Karl Mehltretter Signed-off-by: SJ Park Signed-off-by: Andrew Morton Reviewed-by: SJ Park Assisted-by: LLM Cc: Lian Wang Cc: Kunwu Chan Cc: Brendan Higgins Cc: David Gow --- mm/damon/tests/core-kunit.h | 43 +++++++++++++++++++++++++++++++++++++ 1 file changed, 43 insertions(+) diff --git a/mm/damon/tests/core-kunit.h b/mm/damon/tests/core-kunit.h index a8e22246facf96..2111faa581532a 100644 --- a/mm/damon/tests/core-kunit.h +++ b/mm/damon/tests/core-kunit.h @@ -952,6 +952,48 @@ static void damos_test_commit_quota_goal(struct kunit *test) }); } +static void damos_test_set_psi_current_val(struct kunit *test) +{ + struct damos s = { + .quota.goal_tuner = DAMOS_QUOTA_GOAL_TUNER_CONSIST, + }; + struct damos_quota_goal goal = { + .metric = DAMOS_QUOTA_SOME_MEM_PSI_US, + .target_value = 100, + .last_psi_total = U64_MAX, + }; + + /* uninitialized last_psi_total keeps the consist tuner quota */ + damos_set_psi_current_val(1000, &goal, &s); + KUNIT_EXPECT_EQ(test, goal.current_value, 100ul); + KUNIT_EXPECT_EQ(test, goal.last_psi_total, 1000ull); + + /* initialized last_psi_total gives the delta */ + damos_set_psi_current_val(1030, &goal, &s); + KUNIT_EXPECT_EQ(test, goal.current_value, 30ul); + KUNIT_EXPECT_EQ(test, goal.last_psi_total, 1030ull); + + /* temporal tuner keeps a zero quota */ + s.quota.goal_tuner = DAMOS_QUOTA_GOAL_TUNER_TEMPORAL; + s.quota.esz = 0; + goal.last_psi_total = U64_MAX; + damos_set_psi_current_val(2000, &goal, &s); + KUNIT_EXPECT_EQ(test, goal.current_value, 100ul); + KUNIT_EXPECT_EQ(test, goal.last_psi_total, 2000ull); + + /* temporal tuner keeps a non-zero quota */ + s.quota.esz = SZ_64K; + goal.last_psi_total = U64_MAX; + damos_set_psi_current_val(3000, &goal, &s); + KUNIT_EXPECT_EQ(test, goal.current_value, 0ul); + KUNIT_EXPECT_EQ(test, goal.last_psi_total, 3000ull); + + /* temporal tuner uses the measured PSI delta */ + damos_set_psi_current_val(3250, &goal, &s); + KUNIT_EXPECT_EQ(test, goal.current_value, 250ul); + KUNIT_EXPECT_EQ(test, goal.last_psi_total, 3250ull); +} + static void damos_test_commit_quota_goals_for(struct kunit *test, struct damos_quota_goal *dst_goals, int nr_dst_goals, struct damos_quota_goal *src_goals, int nr_src_goals) @@ -2413,6 +2455,7 @@ static struct kunit_case damon_test_cases[] = { KUNIT_CASE(damos_test_new_filter), KUNIT_CASE(damos_test_new_scheme_keeps_src_quota), KUNIT_CASE(damos_test_commit_quota_goal), + KUNIT_CASE(damos_test_set_psi_current_val), KUNIT_CASE(damos_test_commit_quota_goals), KUNIT_CASE(damos_test_commit_quota), KUNIT_CASE(damos_test_commit_dests), From f1eb8a4e715f33b33904a5ab96707b1fbc66a2cc Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Mon, 28 Sep 2026 16:15:54 +0800 Subject: [PATCH 0865/1012] mm/vmalloc: fix vmalloc_dump_obj address alignment for last-page lookups Patch series "mm/vmalloc: fix vmalloc_dump_obj VA lookup", v4. vmalloc_dump_obj() has two bugs that cause it to miss vmalloc allocations. 1. PAGE_ALIGN() rounds up, pushing last-page addresses to va_end and outside the VA lookup range. 2. The function searches only one vmap node, but allocations larger than 64 KiB may span multiple vmap zones whose VA is stored in a different node's rb-tree. Patch 1 fixes the alignment, patch 2 fixes the cross-zone search. This patch (of 2): vmalloc_dump_obj() uses PAGE_ALIGN() to normalize the input address before looking it up in the per-node busy tree. PAGE_ALIGN() rounds up, which can push an address in the last page of a vmalloc allocation to va_end -- outside the [va_start, va_end) range that __find_vmap_area() searches. This causes the lookup to miss the VA and return false, degrading diagnostic output in OOM dumps and KASAN reports to the less informative "vmalloc memory" fallback. The upward alignment can also change the addr_to_node() mapping when the page boundary crosses a vmap zone boundary, causing the search to hit the wrong node entirely. Use PAGE_ALIGN_DOWN() instead, which rounds down to the page containing the address. This keeps the address within the VA range and preserves the correct node mapping. Link: https://lore.kernel.org/20260928-vmalloc_dump_obj-v4-0-6f288a431edc@linux.dev Link: https://lore.kernel.org/20260928-vmalloc_dump_obj-v4-1-6f288a431edc@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti --- mm/vmalloc.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index 4e4cb8d785bdb4..f5da00821f3fdc 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -5281,7 +5281,7 @@ bool vmalloc_dump_obj(void *object) unsigned long addr; unsigned long nr_pages; - addr = PAGE_ALIGN((unsigned long) object); + addr = PAGE_ALIGN_DOWN((unsigned long) object); vn = addr_to_node(addr); if (!spin_trylock(&vn->busy.lock)) From e046ae9e94780fb46f241eba4575fb344d126ad5 Mon Sep 17 00:00:00 2001 From: Ye Liu Date: Mon, 28 Sep 2026 16:15:55 +0800 Subject: [PATCH 0866/1012] mm/vmalloc: fix vmalloc_dump_obj cross-zone VA lookup vmalloc_dump_obj() searches only one vmap node (addr_to_node(addr)), but a vmalloc allocation may span multiple vmap zones. The VA is stored in only one node's rb-tree (addr_to_node(va_start)), so an object pointer in a different zone than va_start maps to a different node and the search misses. This affects any allocation larger than vmap_zone_size (64 KiB) on multi-CPU systems. Extract find_vmap_area_lock() from find_vmap_area() to share the cross-node iteration logic. The helper supports both spin_lock and spin_trylock, the latter for atomic dump contexts (OOM, KASAN, RCU). Link: https://lore.kernel.org/20260928-vmalloc_dump_obj-v4-2-6f288a431edc@linux.dev Signed-off-by: Ye Liu Signed-off-by: Andrew Morton Reviewed-by: Uladzislau Rezki (Sony) Cc: Paul Walmsley Cc: Palmer Dabbelt Cc: Albert Ou Cc: Alexandre Ghiti --- mm/vmalloc.c | 108 +++++++++++++++++++++++++++++++-------------------- 1 file changed, 66 insertions(+), 42 deletions(-) diff --git a/mm/vmalloc.c b/mm/vmalloc.c index f5da00821f3fdc..a9fce7efe4f496 100644 --- a/mm/vmalloc.c +++ b/mm/vmalloc.c @@ -2516,67 +2516,94 @@ static void free_unmap_vmap_area(struct vmap_area *va) free_vmap_area_noflush(va); } -struct vmap_area *find_vmap_area(unsigned long addr) +static inline int next_vmap_node_id(int i) +{ + return (i + nr_vmap_nodes - 1) % nr_vmap_nodes; +} + +enum vmap_lock_mode { + VMAP_LOCK, + VMAP_TRYLOCK, +}; + +/* + * Search for a vmap_area at @addr across all vmap nodes. An + * addr_to_node_id(addr) converts an address to a node index where + * a VA is located. If VA spans several zones and passed addr is not + * the same as va->va_start, what is not common, we may need to scan + * extra nodes. See an example: + * + * <----va----> + * -|-----|-----|-----|-----|- + * 1 2 0 1 + * + * VA resides in node 1 whereas it spans 1, 2 an 0. If passed addr + * is within 2 or 0 nodes we should do extra work. + * + * Returns the VA with @locked_vn->busy.lock held; the caller must + * release it. If @mode is VMAP_TRYLOCK, nodes that cannot be locked + * are skipped. + */ +static struct vmap_area * +find_vmap_area_lock(unsigned long addr, struct vmap_node **locked_vn, + enum vmap_lock_mode mode) { struct vmap_node *vn; struct vmap_area *va; int i, j; + *locked_vn = NULL; + if (unlikely(!vmap_initialized)) return NULL; - /* - * An addr_to_node_id(addr) converts an address to a node index - * where a VA is located. If VA spans several zones and passed - * addr is not the same as va->va_start, what is not common, we - * may need to scan extra nodes. See an example: - * - * <----va----> - * -|-----|-----|-----|-----|- - * 1 2 0 1 - * - * VA resides in node 1 whereas it spans 1, 2 an 0. If passed - * addr is within 2 or 0 nodes we should do extra work. - */ i = j = addr_to_node_id(addr); do { vn = &vmap_nodes[i]; - spin_lock(&vn->busy.lock); - va = __find_vmap_area(addr, &vn->busy.root); - spin_unlock(&vn->busy.lock); + if (mode == VMAP_LOCK) { + spin_lock(&vn->busy.lock); + } else { + if (!spin_trylock(&vn->busy.lock)) + continue; + } - if (va) + va = __find_vmap_area(addr, &vn->busy.root); + if (va) { + *locked_vn = vn; return va; - } while ((i = (i + nr_vmap_nodes - 1) % nr_vmap_nodes) != j); + } + + spin_unlock(&vn->busy.lock); + } while ((i = next_vmap_node_id(i)) != j); return NULL; } -static struct vmap_area *find_unlink_vmap_area(unsigned long addr) +struct vmap_area *find_vmap_area(unsigned long addr) { struct vmap_node *vn; struct vmap_area *va; - int i, j; - - /* - * Check the comment in the find_vmap_area() about the loop. - */ - i = j = addr_to_node_id(addr); - do { - vn = &vmap_nodes[i]; - spin_lock(&vn->busy.lock); - va = __find_vmap_area(addr, &vn->busy.root); - if (va) - unlink_va(va, &vn->busy.root); + va = find_vmap_area_lock(addr, &vn, VMAP_LOCK); + if (va) spin_unlock(&vn->busy.lock); - if (va) - return va; - } while ((i = (i + nr_vmap_nodes - 1) % nr_vmap_nodes) != j); + return va; +} - return NULL; +static struct vmap_area *find_unlink_vmap_area(unsigned long addr) +{ + struct vmap_node *vn; + struct vmap_area *va; + + va = find_vmap_area_lock(addr, &vn, VMAP_LOCK); + if (va) { + unlink_va(va, &vn->busy.root); + spin_unlock(&vn->busy.lock); + } + + return va; } /*** Per cpu kva allocator ***/ @@ -5282,14 +5309,11 @@ bool vmalloc_dump_obj(void *object) unsigned long nr_pages; addr = PAGE_ALIGN_DOWN((unsigned long) object); - vn = addr_to_node(addr); - - if (!spin_trylock(&vn->busy.lock)) - return false; - va = __find_vmap_area(addr, &vn->busy.root); + va = find_vmap_area_lock(addr, &vn, VMAP_TRYLOCK); if (!va || !va->vm) { - spin_unlock(&vn->busy.lock); + if (va) + spin_unlock(&vn->busy.lock); return false; } From 6c3a68ce1682576310cfdef2a75b8863ca1f603d Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:24 +0800 Subject: [PATCH 0867/1012] gpu: ipu-v3: don't use GFP_DMA when calling dma_alloc_coherent() Patch series "Don't use GFP_DMA when calling dma_alloc_coherent". This series picks up the first part of an earlier cleanup series [1] that was prepared back in the year of 2022, but for various reasons never made it merged and has been sitting in a local tree since then. This subset only touches the call sites where GFP_DMA is passed to dma_alloc_coherent() (and its dma_alloc_wc()/dmam_alloc_coherent() variants), which is the most self-contained and least risky slice of that work. That GFP_DMA is simply redundant here: the DMA API derives the allocation zone from the device's coherent_dma_mask (together with bus_dma_limit) and ignores the GFP_DMA flag passed by the caller. Removing the redundant GFP_DMA won't harm anything, while keeps it from being blindly copied into new code. This patch (of 13): dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-1-hebaoquan@kylinos.cn Link: https://lore.kernel.org/20260903111836.1777265-2-hebaoquan@kylinos.cn Link: https://lore.kernel.org/all/20220219005221.634-1-bhe@redhat.com/T/#u [1] Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/gpu/ipu-v3/ipu-image-convert.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/gpu/ipu-v3/ipu-image-convert.c b/drivers/gpu/ipu-v3/ipu-image-convert.c index 29b6d36c9bb674..493be4d4b091e3 100644 --- a/drivers/gpu/ipu-v3/ipu-image-convert.c +++ b/drivers/gpu/ipu-v3/ipu-image-convert.c @@ -371,7 +371,7 @@ static int alloc_dma_buf(struct ipu_image_convert_priv *priv, { buf->len = PAGE_ALIGN(size); buf->virt = dma_alloc_coherent(priv->ipu->dev, buf->len, &buf->phys, - GFP_DMA | GFP_KERNEL); + GFP_KERNEL); if (!buf->virt) { dev_err(priv->ipu->dev, "failed to alloc dma buffer\n"); return -ENOMEM; From 646f67a9be8fb345d6110bd24e832a4692652d7b Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:25 +0800 Subject: [PATCH 0868/1012] drm/sti: don't use GFP_DMA when calling dma_alloc_wc() dma_alloc_wc() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-3-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/gpu/drm/sti/sti_cursor.c | 4 ++-- drivers/gpu/drm/sti/sti_hqvdp.c | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/drivers/gpu/drm/sti/sti_cursor.c b/drivers/gpu/drm/sti/sti_cursor.c index d0b89b28e50cca..a0ec671477fdf2 100644 --- a/drivers/gpu/drm/sti/sti_cursor.c +++ b/drivers/gpu/drm/sti/sti_cursor.c @@ -240,7 +240,7 @@ static int sti_cursor_atomic_check(struct drm_plane *drm_plane, cursor->pixmap.base = dma_alloc_wc(cursor->dev, cursor->pixmap.size, &cursor->pixmap.paddr, - GFP_KERNEL | GFP_DMA); + GFP_KERNEL); if (!cursor->pixmap.base) { DRM_ERROR("Failed to allocate memory for pixmap\n"); return -EINVAL; @@ -380,7 +380,7 @@ struct drm_plane *sti_cursor_create(struct drm_device *drm_dev, /* Allocate clut buffer */ size = 0x100 * sizeof(unsigned short); cursor->clut = dma_alloc_wc(dev, size, &cursor->clut_paddr, - GFP_KERNEL | GFP_DMA); + GFP_KERNEL); if (!cursor->clut) { DRM_ERROR("Failed to allocate memory for cursor clut\n"); diff --git a/drivers/gpu/drm/sti/sti_hqvdp.c b/drivers/gpu/drm/sti/sti_hqvdp.c index cf1ed8a33b8e18..b0d66834bcbe8b 100644 --- a/drivers/gpu/drm/sti/sti_hqvdp.c +++ b/drivers/gpu/drm/sti/sti_hqvdp.c @@ -860,7 +860,7 @@ static void sti_hqvdp_init(struct sti_hqvdp *hqvdp) size = NB_VDP_CMD * sizeof(struct sti_hqvdp_cmd); hqvdp->hqvdp_cmd = dma_alloc_wc(hqvdp->dev, size, &dma_addr, - GFP_KERNEL | GFP_DMA); + GFP_KERNEL); if (!hqvdp->hqvdp_cmd) { DRM_ERROR("Failed to allocate memory for VDP cmd\n"); return; From 7b00efb9b10a376fbebf1ef22a91e3f6a4f3b144 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:26 +0800 Subject: [PATCH 0869/1012] ALSA: n64: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-4-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- sound/mips/snd-n64.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sound/mips/snd-n64.c b/sound/mips/snd-n64.c index f17e63f2ff5a67..e2cf9df15d485c 100644 --- a/sound/mips/snd-n64.c +++ b/sound/mips/snd-n64.c @@ -298,7 +298,7 @@ static int __init n64audio_probe(struct platform_device *pdev) priv->card = card; priv->ring_base = dma_alloc_coherent(card->dev, 32 * 1024, &priv->ring_base_dma, - GFP_DMA|GFP_KERNEL); + GFP_KERNEL); if (!priv->ring_base) { err = -ENOMEM; goto fail_card; From 41db9968e577b67f17cd34950489748bf56b9648 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:27 +0800 Subject: [PATCH 0870/1012] spi: spi-ti-qspi: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-5-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/spi/spi-ti-qspi.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/spi/spi-ti-qspi.c b/drivers/spi/spi-ti-qspi.c index 34b154922ff24d..fb0d940883fb93 100644 --- a/drivers/spi/spi-ti-qspi.c +++ b/drivers/spi/spi-ti-qspi.c @@ -855,7 +855,7 @@ static int ti_qspi_probe(struct platform_device *pdev) qspi->rx_bb_addr = dma_alloc_coherent(qspi->dev, QSPI_DMA_BUFFER_SIZE, &qspi->rx_bb_dma_addr, - GFP_KERNEL | GFP_DMA); + GFP_KERNEL); if (!qspi->rx_bb_addr) { dev_err(qspi->dev, "dma_alloc_coherent failed, using PIO mode\n"); From 7e4c182103918f82933a60f10f0c43649d35614f Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:28 +0800 Subject: [PATCH 0871/1012] fbdev: fsl-diu-fb: don't use GFP_DMA when calling dmam_alloc_coherent() dmam_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-6-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/video/fbdev/fsl-diu-fb.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/video/fbdev/fsl-diu-fb.c b/drivers/video/fbdev/fsl-diu-fb.c index b71d15794ce8b8..d7ed007915c189 100644 --- a/drivers/video/fbdev/fsl-diu-fb.c +++ b/drivers/video/fbdev/fsl-diu-fb.c @@ -1690,7 +1690,7 @@ static int fsl_diu_probe(struct platform_device *pdev) int ret; data = dmam_alloc_coherent(&pdev->dev, sizeof(struct fsl_diu_data), - &dma_addr, GFP_DMA | __GFP_ZERO); + &dma_addr, __GFP_ZERO); if (!data) return -ENOMEM; data->dma_addr = dma_addr; From 3fd530b05742f362dd95c2c548259ccba39b6b59 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:29 +0800 Subject: [PATCH 0872/1012] usb: gadget: lpc32xx_udc: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-7-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/usb/gadget/udc/lpc32xx_udc.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/usb/gadget/udc/lpc32xx_udc.c b/drivers/usb/gadget/udc/lpc32xx_udc.c index 044c31869cfb26..90d1eef528770e 100644 --- a/drivers/usb/gadget/udc/lpc32xx_udc.c +++ b/drivers/usb/gadget/udc/lpc32xx_udc.c @@ -3080,7 +3080,7 @@ static int lpc32xx_udc_probe(struct platform_device *pdev) /* Allocate memory for the UDCA */ udc->udca_v_base = dma_alloc_coherent(&pdev->dev, UDCA_BUFF_SIZE, &dma_handle, - (GFP_KERNEL | GFP_DMA)); + GFP_KERNEL); if (!udc->udca_v_base) { dev_err(udc->dev, "error getting UDCA region\n"); retval = -ENOMEM; From fda64418aefe46cc9764168d0a2af62a7d2f42a0 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:30 +0800 Subject: [PATCH 0873/1012] usb: cdns3: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Use GFP_KERNEL instead so the allocation may reclaim as usual for probe-time allocations. Link: https://lore.kernel.org/20260903111836.1777265-8-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/usb/cdns3/cdns3-gadget.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/usb/cdns3/cdns3-gadget.c b/drivers/usb/cdns3/cdns3-gadget.c index 42311c1bfada1d..484d3128cf081c 100644 --- a/drivers/usb/cdns3/cdns3-gadget.c +++ b/drivers/usb/cdns3/cdns3-gadget.c @@ -3376,7 +3376,7 @@ static int cdns3_gadget_start(struct cdns *cdns) /* allocate memory for setup packet buffer */ priv_dev->setup_buf = dma_alloc_coherent(priv_dev->sysdev, 8, - &priv_dev->setup_dma, GFP_DMA); + &priv_dev->setup_dma, GFP_KERNEL); if (!priv_dev->setup_buf) { ret = -ENOMEM; goto err2; From 2690d3c6dc9784e183b045a528da0b76f242d0ed Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:31 +0800 Subject: [PATCH 0874/1012] media: staging: imx: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-9-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/staging/media/imx/imx-media-utils.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/staging/media/imx/imx-media-utils.c b/drivers/staging/media/imx/imx-media-utils.c index f119477cac6b16..7bdcb88a686b1c 100644 --- a/drivers/staging/media/imx/imx-media-utils.c +++ b/drivers/staging/media/imx/imx-media-utils.c @@ -588,7 +588,7 @@ int imx_media_alloc_dma_buf(struct device *dev, buf->len = PAGE_ALIGN(size); buf->virt = dma_alloc_coherent(dev, buf->len, &buf->phys, - GFP_DMA | GFP_KERNEL); + GFP_KERNEL); if (!buf->virt) return -ENOMEM; From 55e6d9efe500cb5d6ca37a38797f37218672a98e Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:32 +0800 Subject: [PATCH 0875/1012] spi: atmel: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-10-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/spi/spi-atmel.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/spi/spi-atmel.c b/drivers/spi/spi-atmel.c index c8012c82c3a788..e91e84fdbe8f00 100644 --- a/drivers/spi/spi-atmel.c +++ b/drivers/spi/spi-atmel.c @@ -624,7 +624,7 @@ static int atmel_spi_configure_dma(struct spi_controller *host, if (IS_ENABLED(CONFIG_SOC_SAM_V4_V5)) { as->addr_tx_bbuf = dma_alloc_coherent(dev, SPI_MAX_DMA_XFER, &as->dma_addr_tx_bbuf, - GFP_KERNEL | GFP_DMA); + GFP_KERNEL); if (!as->addr_tx_bbuf) { err = -ENOMEM; goto err_release_dma; @@ -632,7 +632,7 @@ static int atmel_spi_configure_dma(struct spi_controller *host, as->addr_rx_bbuf = dma_alloc_coherent(dev, SPI_MAX_DMA_XFER, &as->dma_addr_rx_bbuf, - GFP_KERNEL | GFP_DMA); + GFP_KERNEL); if (!as->addr_rx_bbuf) { err = -ENOMEM; goto err_release_dma; From 7c64b8b9b947d885f016d23a5f09e34bc14d3b80 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:33 +0800 Subject: [PATCH 0876/1012] media: imx7-media-csi: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-11-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Reviewed-by: Frank Li Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/media/platform/nxp/imx7-media-csi.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/media/platform/nxp/imx7-media-csi.c b/drivers/media/platform/nxp/imx7-media-csi.c index 7ddc7ba06e3d4e..22c0cbdc92bfb4 100644 --- a/drivers/media/platform/nxp/imx7-media-csi.c +++ b/drivers/media/platform/nxp/imx7-media-csi.c @@ -466,7 +466,7 @@ static int imx7_csi_alloc_dma_buf(struct imx7_csi *csi, buf->len = PAGE_ALIGN(size); buf->virt = dma_alloc_coherent(csi->dev, buf->len, &buf->dma_addr, - GFP_DMA | GFP_KERNEL); + GFP_KERNEL); if (!buf->virt) return -ENOMEM; From 613be618f31786b6ded4636cb80203148a38d15b Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:34 +0800 Subject: [PATCH 0877/1012] media: nxp: imx8-isi: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Remove the redundant GFP_DMA flag. Link: https://lore.kernel.org/20260903111836.1777265-12-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Reviewed-by: Frank Li Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/media/platform/nxp/imx8-isi/imx8-isi-video.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/media/platform/nxp/imx8-isi/imx8-isi-video.c b/drivers/media/platform/nxp/imx8-isi/imx8-isi-video.c index f45c2aae59ce90..fc907d357149ee 100644 --- a/drivers/media/platform/nxp/imx8-isi/imx8-isi-video.c +++ b/drivers/media/platform/nxp/imx8-isi/imx8-isi-video.c @@ -773,7 +773,7 @@ static int mxc_isi_video_alloc_discard_buffers(struct mxc_isi_video *video) buf->size = PAGE_ALIGN(video->pix.plane_fmt[i].sizeimage); buf->addr = dma_alloc_coherent(video->pipe->isi->dev, buf->size, - &buf->dma, GFP_DMA | GFP_KERNEL); + &buf->dma, GFP_KERNEL); if (!buf->addr) { mxc_isi_video_free_discard_buffers(video); return -ENOMEM; From 18f2a0d784955a1ea2edeba40ecfedd90d5cf4e1 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:35 +0800 Subject: [PATCH 0878/1012] mtd: rawnand: gpmi: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Use GFP_KERNEL instead so the allocation may reclaim as usual for probe-time allocations. Link: https://lore.kernel.org/20260903111836.1777265-13-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c b/drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c index 527165ccc839da..0d91950ae9fde6 100644 --- a/drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c +++ b/drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c @@ -1387,7 +1387,7 @@ static int gpmi_alloc_dma_buffer(struct gpmi_nand_data *this) goto error_alloc; this->auxiliary_virt = dma_alloc_coherent(dev, geo->auxiliary_size, - &this->auxiliary_phys, GFP_DMA); + &this->auxiliary_phys, GFP_KERNEL); if (!this->auxiliary_virt) goto error_alloc; From c5d06644a7525baec96f05918c413dfbabcde297 Mon Sep 17 00:00:00 2001 From: Baoquan He Date: Thu, 3 Sep 2026 19:18:36 +0800 Subject: [PATCH 0879/1012] usb: cdns2: don't use GFP_DMA when calling dma_alloc_coherent() dma_alloc_coherent() allocates the DMA buffer with the device's addressing limitation in mind; the DMA core picks the zone from the device's coherent DMA mask and ignores GFP_DMA passed by the caller. Use GFP_KERNEL instead so the allocation may reclaim as usual for probe-time allocations. Link: https://lore.kernel.org/20260903111836.1777265-14-hebaoquan@kylinos.cn Signed-off-by: Baoquan He Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: Harry Yoo --- drivers/usb/gadget/udc/cdns2/cdns2-gadget.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/drivers/usb/gadget/udc/cdns2/cdns2-gadget.c b/drivers/usb/gadget/udc/cdns2/cdns2-gadget.c index 308d3c468ab197..8719f1f86e6106 100644 --- a/drivers/usb/gadget/udc/cdns2/cdns2-gadget.c +++ b/drivers/usb/gadget/udc/cdns2/cdns2-gadget.c @@ -2341,7 +2341,7 @@ static int cdns2_gadget_start(struct cdns2_device *pdev) /* Allocate memory for setup packet buffer. */ buf = dma_alloc_coherent(pdev->dev, 8, &pdev->ep0_preq.request.dma, - GFP_DMA); + GFP_KERNEL); pdev->ep0_preq.request.buf = buf; if (!pdev->ep0_preq.request.buf) { From 5e34812cd7fef7642a481b466aa63296fa1a3167 Mon Sep 17 00:00:00 2001 From: DAI RENJIE Date: Fri, 21 Aug 2026 14:08:17 +0000 Subject: [PATCH 0880/1012] resource: fix lost wakeup when waiting for a muxed region A task waiting for a muxed region can sleep forever in TASK_UNINTERRUPTIBLE even though the region it waits for is already free. __request_region_locked() queues itself on muxed_resource_wait and drops resource_lock before setting TASK_UNINTERRUPTIBLE, while __release_region() wakes the queue after dropping the same lock. A wakeup landing in between finds TASK_RUNNING, does not match TASK_NORMAL and is discarded; callers hold a muxed region only across a bounded transaction, so no further release is coming. The task is unkillable and its caller never returns. The window is one store wide, but an interrupt is enough to hold the waiter in it, and the machine this was seen on runs PREEMPT_DYNAMIC in its voluntary default. Since v6.11 spd5118 exports the DDR5 sensors of AMD boards through i2c-piix4, which takes a muxed region per SMBus transaction; a third of the in-tree users of request_muxed_region() are hwmon drivers, so reading a world-readable attribute is all an unprivileged user needs to drive the contention. The blocked task sleeps holding the i2c adapter bus lock, and 27 more piled up behind it. Reproduced by building a kernel with the two orderings selectable at runtime and a 2ms delay inside the window. Switching only that knob, a two-thread barriered reproducer loses the wakeup 200 times out of 200 before the fix and 0 out of 200 after it; without the delay it goes 20000 times through the wait path and loses none. Fix it by setting the task state before dropping resource_lock, as prepare_to_wait() does: the releasing side needs resource_lock to unlink the resource, so it cannot reach the wakeup before the state is published. Link: https://lore.kernel.org/20260821-b4-resource-muxed-lost-wakeup-v1-1-37eb6473a76c@gmail.com Fixes: 8b6d043b7ee2 ("resource: shared I/O region support") Signed-off-by: DAI RENJIE Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Reviewed-by: Andrew Morton Assisted-by: Claude:claude-opus-5 Cc: Mark Brown Cc: Kees Cook Cc: Bjorn Helgaas Cc: --- kernel/resource.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/resource.c b/kernel/resource.c index e60539a55541df..54d7199695fbe9 100644 --- a/kernel/resource.c +++ b/kernel/resource.c @@ -1350,8 +1350,8 @@ static int __request_region_locked(struct resource *res, struct resource *parent } if (conflict->flags & flags & IORESOURCE_MUXED) { add_wait_queue(&muxed_resource_wait, &wait); - write_unlock(&resource_lock); set_current_state(TASK_UNINTERRUPTIBLE); + write_unlock(&resource_lock); schedule(); remove_wait_queue(&muxed_resource_wait, &wait); write_lock(&resource_lock); From 0b10592201912de623c475e80d620440d6603fe9 Mon Sep 17 00:00:00 2001 From: OGAWA Hirofumi Date: Tue, 25 Aug 2026 21:11:32 +0900 Subject: [PATCH 0881/1012] fat: fix fat_ent_write() for reverting the value commit 64d9183203ee ("fat: restore original value when fat_ent_write failed") try to revert the fatent value to old value when got the error on mirror FAT. However it didn't work if the error is when writing the fatent bh. In that case, the bh is cleared the uptodate flag, so reuse bh is invalid. Fix this by reverting the fatent only if got the error on mirror FAT. Link: https://lore.kernel.org/87ik4yz9fv.fsf_-_@mail.parknet.co.jp Fixes: 64d9183203ee ("fat: restore original value when fat_ent_write failed") Signed-off-by: OGAWA Hirofumi Signed-off-by: Andrew Morton Reported-by: syzbot+e64c6472a3d96a75172a@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=e64c6472a3d96a75172a Reported-by: syzbot+26461e903494e689c24f@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=26461e903494e689c24f Cc: Yemu Lu Cc: Ren Wei Cc: Yuan Tan Cc: Yifan Wu Cc: Juefei Pu Cc: Xin Liu Cc: --- fs/fat/fat.h | 2 +- fs/fat/fatent.c | 21 ++++++++++++++++++--- fs/fat/file.c | 3 ++- fs/fat/misc.c | 6 ++---- 4 files changed, 23 insertions(+), 9 deletions(-) diff --git a/fs/fat/fat.h b/fs/fat/fat.h index 61338413d9f3e4..fbd207c55859e7 100644 --- a/fs/fat/fat.h +++ b/fs/fat/fat.h @@ -392,7 +392,7 @@ extern void fat_ent_access_init(struct super_block *sb); extern int fat_ent_read(struct inode *inode, struct fat_entry *fatent, int entry); extern int fat_ent_write(struct inode *inode, struct fat_entry *fatent, - int new, int wait); + int new, int old, int wait); extern int fat_alloc_clusters(struct inode *inode, int *cluster, int nr_cluster); extern int fat_free_clusters(struct inode *inode, int cluster); diff --git a/fs/fat/fatent.c b/fs/fat/fatent.c index f0801d99dd62ae..df23fc85f31307 100644 --- a/fs/fat/fatent.c +++ b/fs/fat/fatent.c @@ -413,7 +413,7 @@ static int fat_mirror_bhs(struct super_block *sb, struct buffer_head **bhs, } int fat_ent_write(struct inode *inode, struct fat_entry *fatent, - int new, int wait) + int new, int old, int wait) { struct super_block *sb = inode->i_sb; const struct fatent_operations *ops = MSDOS_SB(sb)->fatent_ops; @@ -422,10 +422,25 @@ int fat_ent_write(struct inode *inode, struct fat_entry *fatent, ops->ent_put(fatent, new); if (wait) { err = fat_sync_bhs(fatent->bhs, fatent->nr_bhs); - if (err) + if (err) { + /* + * bhs are not uptodate after I/O error. So we + * can't simply re-dirty to revert. And it + * would not have value to write again on I/O + * error. + */ return err; + } } - return fat_mirror_bhs(sb, fatent->bhs, fatent->nr_bhs); + + err = fat_mirror_bhs(sb, fatent->bhs, fatent->nr_bhs); + if (err) { + /* Try to revert if got the error on mirror FAT */ + ops->ent_put(fatent, old); + if (wait) + fat_sync_bhs(fatent->bhs, fatent->nr_bhs); + } + return err; } static inline int fat_ent_next(struct msdos_sb_info *sbi, diff --git a/fs/fat/file.c b/fs/fat/file.c index 1c835ca5f21a51..6c475c53334c6c 100644 --- a/fs/fat/file.c +++ b/fs/fat/file.c @@ -363,7 +363,8 @@ static int fat_free(struct inode *inode, int skip) __func__, MSDOS_I(inode)->i_pos); ret = -EIO; } else if (ret > 0) { - err = fat_ent_write(inode, &fatent, FAT_ENT_EOF, wait); + err = fat_ent_write(inode, &fatent, FAT_ENT_EOF, ret, + wait); if (err) ret = err; } diff --git a/fs/fat/misc.c b/fs/fat/misc.c index e79762cf19754d..c44296756eae61 100644 --- a/fs/fat/misc.c +++ b/fs/fat/misc.c @@ -133,11 +133,9 @@ int fat_chain_add(struct inode *inode, int new_dclus, int nr_cluster) ret = fat_ent_read(inode, &fatent, last); if (ret >= 0) { int wait = inode_needs_sync(inode); - int old = ret; - ret = fat_ent_write(inode, &fatent, new_dclus, wait); - if (ret < 0) - fat_ent_write(inode, &fatent, old, wait); + ret = fat_ent_write(inode, &fatent, new_dclus, ret, + wait); fatent_brelse(&fatent); } if (ret < 0) From 844c67369f72de6c89ad783b3fd1d59f8e8f4902 Mon Sep 17 00:00:00 2001 From: Michael Liang Date: Fri, 21 Aug 2026 12:15:27 -0600 Subject: [PATCH 0882/1012] fault-inject: fix dentry leak fault_create_debugfs_attr() has always taken an extra dentry reference on the created directory (attr->dname = dget(dir)) so that fail_dump() could print the name via %pd from any context. Nothing anywhere in the tree ever calls dput() on attr->dname. For callers with a matching teardown, that unmatched reference causes one dentry plus its attached inode to leak per fault_create_debugfs_attr / debugfs_remove_recursive cycle. simple_recursive_removal() drops debugfs's own +1 ref on the child dentry, but the dget()'d ref keeps its refcount at 1: the dentry ends up unhashed but pinned, and its inode is never freed. Boot-once callers (mm/failslab, block/blk-core, etc.) leak exactly once at init and never destroy the tree, so the impact there is bounded. But per-lifecycle callers (drivers/nvme, drivers/infiniband/hw/hfi1, drivers/mmc, drivers/iommu/iommufd, drivers/media, drivers/misc, drivers/gpu/drm/msm, drivers/crypto, net/sunrpc) leak on every create/destroy cycle. We observed this in production: an NVMe/RDMA host repeatedly reconnecting to a target that rejected the CRTO Property Get went through ~50 nvme controller create/destroy cycles per second, and dentry and inode_cache grew by ~13k pinned objects per 240 s -- unrecoverable through drop_caches. Byte math matched a per-cycle 1-dentry / 1-inode leak from the "fault_inject" directory dentry. Fix this by not holding any external reference in fault_attr. Embed the directory name as a fixed-size char array (FAULT_ATTR_DNAME_LEN, 64 bytes) inside struct fault_attr, copied by strscpy() at fault_create_debugfs_attr() time. fail_dump() prints it via %s. Advantages of an embedded array over kstrdup() + kfree() paired with a new destroy API: - Zero API footprint. No new export and no caller changes required: callers already own their fault_attr's memory and free it when they are done, and now that suffices. - No allocation on the create path. - fault_create_debugfs_attr() cannot fail from the name-copy step. - No lifetime coupling between attr->dname and debugfs; the string is valid for exactly as long as the containing struct. The 64-byte length accommodates every in-tree caller with generous headroom (the longest current name is "fail_dma_array_full", 19 chars). The user-visible fail_dump() format changes from "name %pd" to "name %s", but the printed content is identical -- %pd on the created directory renders the same string that was passed in as @name. drivers/infiniband/hw/hfi1/fault.c drops a now-invalid "attr.dname = NULL" statement; the surrounding kzalloc() already zero-initialises the array. Link: https://lore.kernel.org/20260821181527.3271414-1-mliang@purestorage.com Fixes: 6adc4a22f20b ("fault-inject: add ratelimit option") Signed-off-by: Michael Liang Signed-off-by: Andrew Morton Reviewed-by: Andrew Morton Cc: Akinbou Mita Cc: Dennis Dalessandro Cc: Jason Gunthorpe Cc: Leon Romanovsky Cc: Vlastimil Babka Cc: --- drivers/infiniband/hw/hfi1/fault.c | 1 - include/linux/fault-inject.h | 10 ++++++++-- lib/fault-inject.c | 7 +++++-- 3 files changed, 13 insertions(+), 5 deletions(-) diff --git a/drivers/infiniband/hw/hfi1/fault.c b/drivers/infiniband/hw/hfi1/fault.c index 4ab72ef03ba11b..941a0b96590b60 100644 --- a/drivers/infiniband/hw/hfi1/fault.c +++ b/drivers/infiniband/hw/hfi1/fault.c @@ -216,7 +216,6 @@ int hfi1_fault_init_debugfs(struct hfi1_ibdev *ibd) ibd->fault->attr.interval = 1; ibd->fault->attr.require_end = ULONG_MAX; ibd->fault->attr.stacktrace_depth = 32; - ibd->fault->attr.dname = NULL; ibd->fault->attr.verbose = 0; ibd->fault->enable = false; ibd->fault->opcode = false; diff --git a/include/linux/fault-inject.h b/include/linux/fault-inject.h index 58fd14c8227080..5c74748a53f38f 100644 --- a/include/linux/fault-inject.h +++ b/include/linux/fault-inject.h @@ -18,6 +18,13 @@ enum fault_flags { #include #include +/* + * Length of the debugfs directory name embedded in struct fault_attr. + * Chosen to accommodate every in-tree caller of fault_create_debugfs_attr() + * (the longest is "fail_dma_array_full", 19 chars) with generous headroom. + */ +#define FAULT_ATTR_DNAME_LEN 64 + /* * For explanation of the elements of this struct, see * Documentation/fault-injection/fault-injection.rst @@ -37,7 +44,7 @@ struct fault_attr { unsigned long count; struct ratelimit_state ratelimit_state; - struct dentry *dname; + char dname[FAULT_ATTR_DNAME_LEN]; }; #define FAULT_ATTR_INITIALIZER { \ @@ -47,7 +54,6 @@ struct fault_attr { .stacktrace_depth = 32, \ .ratelimit_state = RATELIMIT_STATE_INIT_DISABLED, \ .verbose = 2, \ - .dname = NULL, \ } #define DECLARE_FAULT_ATTR(name) struct fault_attr name = FAULT_ATTR_INITIALIZER diff --git a/lib/fault-inject.c b/lib/fault-inject.c index 999053fa133e3f..02916ef2761c60 100644 --- a/lib/fault-inject.c +++ b/lib/fault-inject.c @@ -5,6 +5,7 @@ #include #include #include +#include #include #include #include @@ -64,7 +65,7 @@ static void fail_dump(struct fault_attr *attr) { if (attr->verbose > 0 && __ratelimit(&attr->ratelimit_state)) { printk(KERN_NOTICE "FAULT_INJECTION: forcing a failure.\n" - "name %pd, interval %lu, probability %lu, " + "name %s, interval %lu, probability %lu, " "space %d, times %d\n", attr->dname, attr->interval, attr->probability, atomic_read(&attr->space), @@ -261,7 +262,9 @@ struct dentry *fault_create_debugfs_attr(const char *name, debugfs_create_xul("reject-end", mode, dir, &attr->reject_end); #endif /* CONFIG_FAULT_INJECTION_STACKTRACE_FILTER */ - attr->dname = dget(dir); + if (strscpy(attr->dname, name, sizeof(attr->dname)) == -E2BIG) + pr_warn("FAULT_INJECTION: name '%s' truncated to '%s'\n", + name, attr->dname); return dir; } EXPORT_SYMBOL_GPL(fault_create_debugfs_attr); From f6a5549e3dbae746437d0cd19db686e02287a858 Mon Sep 17 00:00:00 2001 From: Andrew Morton Date: Tue, 26 May 2026 15:14:09 -0700 Subject: [PATCH 0883/1012] drivers/media/v4l2-core/v4l2-vp9.c: reduce inlining csky allmodconfig, gcc-15.2.0: drivers/media/v4l2-core/v4l2-vp9.c: In function 'v4l2_vp9_adapt_noncoef_probs': drivers/media/v4l2-core/v4l2-vp9.c:1834:1: error: the frame size of 1436 bytes is larger than 1280 bytes [-Werror=frame-larger-than=] The amount of inlining in there is simply nuts. This patch semi-randomly uninlines various things and fixes the above. Ad the .text size reduction is tremendous: ts:/usr/src/25> size drivers/media/v4l2-core/v4l2-vp9.o text data bss dec hex filename 22450 36 0 22486 57d6 drivers/media/v4l2-core/v4l2-vp9.o-before 16144 36 0 16180 3f34 drivers/media/v4l2-core/v4l2-vp9.o-after Reviewed-by: Daniel Almeida Cc: Mauro Carvalho Chehab Signed-off-by: Andrew Morton --- drivers/media/v4l2-core/v4l2-vp9.c | 30 +++++++++++++++--------------- 1 file changed, 15 insertions(+), 15 deletions(-) diff --git a/drivers/media/v4l2-core/v4l2-vp9.c b/drivers/media/v4l2-core/v4l2-vp9.c index 859589f1fd35f5..e965ffbd9b8aa2 100644 --- a/drivers/media/v4l2-core/v4l2-vp9.c +++ b/drivers/media/v4l2-core/v4l2-vp9.c @@ -1582,25 +1582,25 @@ static inline u8 noncoef_merge_prob(u8 pre_prob, u32 ct0, u32 ct1) * merge_prob(p[9], c[9], [10]) */ -static inline void merge_probs_variant_a(u8 *p, const u32 *c, u16 count_sat, u32 update_factor) +static noinline_for_stack void merge_probs_variant_a(u8 *p, const u32 *c, u16 count_sat, u32 update_factor) { p[1] = merge_prob(p[1], c[0], c[1] + c[2], count_sat, update_factor); p[2] = merge_prob(p[2], c[1], c[2], count_sat, update_factor); } -static inline void merge_probs_variant_b(u8 *p, const u32 *c, u16 count_sat, u32 update_factor) +static noinline_for_stack void merge_probs_variant_b(u8 *p, const u32 *c, u16 count_sat, u32 update_factor) { p[0] = merge_prob(p[0], c[0], c[1], count_sat, update_factor); } -static inline void merge_probs_variant_c(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_c(u8 *p, const u32 *c) { p[0] = noncoef_merge_prob(p[0], c[2], c[1] + c[0] + c[3]); p[1] = noncoef_merge_prob(p[1], c[0], c[1] + c[3]); p[2] = noncoef_merge_prob(p[2], c[1], c[3]); } -static void merge_probs_variant_d(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_d(u8 *p, const u32 *c) { u32 sum = 0, s2; @@ -1624,20 +1624,20 @@ static void merge_probs_variant_d(u8 *p, const u32 *c) p[8] = noncoef_merge_prob(p[8], c[6], c[7]); } -static inline void merge_probs_variant_e(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_e(u8 *p, const u32 *c) { p[0] = noncoef_merge_prob(p[0], c[0], c[1] + c[2] + c[3]); p[1] = noncoef_merge_prob(p[1], c[1], c[2] + c[3]); p[2] = noncoef_merge_prob(p[2], c[2], c[3]); } -static inline void merge_probs_variant_f(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_f(u8 *p, const u32 *c) { p[0] = noncoef_merge_prob(p[0], c[0], c[1] + c[2]); p[1] = noncoef_merge_prob(p[1], c[1], c[2]); } -static void merge_probs_variant_g(u8 *p, const u32 *c) +static noinline_for_stack void merge_probs_variant_g(u8 *p, const u32 *c) { u32 sum; @@ -1659,12 +1659,12 @@ static void merge_probs_variant_g(u8 *p, const u32 *c) } /* 8.4.3 Coefficient probability adaptation process */ -static inline void adapt_probs_variant_a_coef(u8 *p, const u32 *c, u32 update_factor) +static noinline_for_stack void adapt_probs_variant_a_coef(u8 *p, const u32 *c, u32 update_factor) { merge_probs_variant_a(p, c, 24, update_factor); } -static inline void adapt_probs_variant_b_coef(u8 *p, const u32 *c, u32 update_factor) +static noinline_for_stack void adapt_probs_variant_b_coef(u8 *p, const u32 *c, u32 update_factor) { merge_probs_variant_b(p, c, 24, update_factor); } @@ -1724,33 +1724,33 @@ static inline void adapt_probs_variant_b(u8 *p, const u32 *c) merge_probs_variant_b(p, c, 20, 128); } -static inline void adapt_probs_variant_c(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_c(u8 *p, const u32 *c) { merge_probs_variant_c(p, c); } -static inline void adapt_probs_variant_d(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_d(u8 *p, const u32 *c) { merge_probs_variant_d(p, c); } -static inline void adapt_probs_variant_e(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_e(u8 *p, const u32 *c) { merge_probs_variant_e(p, c); } -static inline void adapt_probs_variant_f(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_f(u8 *p, const u32 *c) { merge_probs_variant_f(p, c); } -static inline void adapt_probs_variant_g(u8 *p, const u32 *c) +static noinline_for_stack void adapt_probs_variant_g(u8 *p, const u32 *c) { merge_probs_variant_g(p, c); } /* 8.4.4 Non coefficient probability adaptation process, adapt_prob() */ -static inline u8 adapt_prob(u8 prob, const u32 counts[2]) +static noinline_for_stack u8 adapt_prob(u8 prob, const u32 counts[2]) { return noncoef_merge_prob(prob, counts[0], counts[1]); } From cd457dd6a0bbacbe374344fb6f26d4450c6ae1de Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Tue, 4 Aug 2026 09:34:00 +0000 Subject: [PATCH 0884/1012] taskstats: copy signal->stats under siglock in taskstats_exit taskstats_exit() copies tsk->signal->stats into the exit reply without taking any lock. Every other writer of this struct holds sighand->siglock before touching it, and this copy does not. The copy happens on the last thread of a thread group that exits. group_dead being 1 only says that every thread has dropped signal->live, it does not say how far the other threads got in do_exit(). One of them can still be inside fill_tgid_exit() adding its counters to the struct while the last thread copies it out, so the copy can read the struct in the middle of an update. The commit that added the copy assumed no locking was needed because the group was dead: /* No locking needed for tsk->signal->stats since group is dead */ but at that point the other threads have not necessarily finished their exit path. cpu0 (thread A, not last) cpu1 (thread B, last) =========================== ============================== atomic_dec(&signal->live) atomic_dec(&signal->live) -> 0 group_dead = 0 group_dead = 1 ... taskstats_exit(tsk, 1) taskstats_exit(tsk, 0) fill_tgid_exit(tsk) [siglock] fill_tgid_exit(tsk) memcpy(stats, signal->stats) spin_lock(siglock) reads ac_utime (new) stats->ac_utime += x reads ac_stime (old) stats->ac_stime += y torn snapshot -> netlink spin_unlock(siglock) The listeners receive a partially updated tgid snapshot, with some fields from before the concurrent update and some from after. There is no crash or splat, which is likely why this went unnoticed since 2006. A userspace model of the same shape, writer under a lock and a lockless memcpy reader, produces millions of torn reads in a few seconds. Take siglock around the copy like every other access does. sighand is still alive here because taskstats_exit() runs before exit_notify(), and fill_tgid_exit() already takes this same lock earlier in this function. Link: https://lore.kernel.org/20260804093400.3922-1-include@grrlz.net Fixes: ad4ecbcba728 ("[PATCH] delay accounting taskstats interface send tgid once") Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Acked-by: Oleg Nesterov Cc: Balbir Singh --- kernel/taskstats.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/kernel/taskstats.c b/kernel/taskstats.c index f31df72f0e9df1..9a48827e22bce3 100644 --- a/kernel/taskstats.c +++ b/kernel/taskstats.c @@ -590,6 +590,7 @@ void taskstats_exit(struct task_struct *tsk, int group_dead) struct sk_buff *rep_skb; size_t size; int is_thread_group; + unsigned long flags; if (!family_registered) return; @@ -635,7 +636,10 @@ void taskstats_exit(struct task_struct *tsk, int group_dead) if (!stats) goto err; + /* This was racy before, copy the stats under siglock. */ + spin_lock_irqsave(&tsk->sighand->siglock, flags); memcpy(stats, tsk->signal->stats, sizeof(*stats)); + spin_unlock_irqrestore(&tsk->sighand->siglock, flags); stats->version = TASKSTATS_VERSION; send: From 45eac1ae2d4b72ff14e5978387f658b881efe303 Mon Sep 17 00:00:00 2001 From: Petr Vorel Date: Mon, 10 Aug 2026 18:11:59 +0200 Subject: [PATCH 0885/1012] checkpatch: skip CamelCase cache for --no-tree without root Running outside tree (--no-tree) without git root (--root DIR) is not doable because we have no include/ directory which could be cached. But 3445686af721 expected that we are always in Linux tree (w/a git). But --no-tree does not require --root. Therefore skip whole caching in that case. This fixes perl and find errors when running checkpatch.pl *with* --no-tree --strict and *without* --root: No structs that should be const will be found - file 'scripts/const_structs.checkpatch': No such file or directory Use of uninitialized value $root in concatenation (.) or string at scripts/checkpatch.pl line 1213. find: `/include': No such file or directory Link: https://lore.kernel.org/20260810161159.1044160-1-pvorel@suse.cz Fixes: 3445686af721 ("checkpatch: ignore existing CamelCase uses from include/...") Signed-off-by: Petr Vorel Signed-off-by: Andrew Morton Cc: Andy Whitcroft Cc: Dwaipayan Ray Cc: Joe Perches Cc: Lukas Bulwahn --- scripts/checkpatch.pl | 2 ++ 1 file changed, 2 insertions(+) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index 8a7787d228a63d..f424dafce5bce7 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -1206,6 +1206,8 @@ sub seed_camelcase_includes { my $git_last_include_commit = `${git_command} log --no-merges --pretty=format:"%h%n" -1 -- include`; chomp $git_last_include_commit; $camelcase_cache = ".checkpatch-camelcase.git.$git_last_include_commit"; + } elsif (not defined $root) { + return; } else { my $last_mod_date = 0; $files = `find $root/include -name "*.h"`; From cbabf8b8bbf6a50ad49699411329341f042ca200 Mon Sep 17 00:00:00 2001 From: Zi Yan Date: Thu, 20 Aug 2026 09:19:12 -0400 Subject: [PATCH 0886/1012] USB: gadgetfs: do not WARN about excessively large memory allocations GadgetFS passes an excessively large user input len to kmalloc and kmalloc gives a WARN (see below for details). Suppress it by passing __GFP_NOWARN to kmalloc used by both ep_write_iter() and ep_read_iter(). Follow the same method as commit 4f2629ea67e72 ("USB: usbfs: Don't WARN about excessively large memory allocations"). kmalloc is used to allocate physically contiguous memory for kernel allocations. For requests larger than KMALLOC_MAX_CACHE_SIZE, kmalloc uses the page allocator and can only support up to KMALLOC_MAX_SIZE. For request sizes bigger than KMALLOC_MAX_SIZE, the page allocator can emit a WARN because kmalloc allocates an order greater than MAX_PAGE_ORDER. Link: https://lore.kernel.org/DKTTMAS94IMH.2C6ERY0ZIVWVZ@nvidia.com Fixes: b3c466ce5129 ("page allocator: do not sanity check order in the fast path") Signed-off-by: Zi Yan Signed-off-by: Andrew Morton Reported-by: syzbot+805630f1453e490427fa@syzkaller.appspotmail.com Closes: https://lore.kernel.org/all/6a820ebc.9ebadd4d.20b15e.001b.GAE@google.com/ Tested-by: syzbot+805630f1453e490427fa@syzkaller.appspotmail.com Acked-by: Alan Stern Acked-by: Vlastimil Babka (SUSE) Cc: Greg Kroah-Hartman Cc: --- drivers/usb/gadget/legacy/inode.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/usb/gadget/legacy/inode.c b/drivers/usb/gadget/legacy/inode.c index 67c6ffaf4f72d6..7e383e3e0f664d 100644 --- a/drivers/usb/gadget/legacy/inode.c +++ b/drivers/usb/gadget/legacy/inode.c @@ -613,7 +613,7 @@ ep_read_iter(struct kiocb *iocb, struct iov_iter *to) return -EBADMSG; } - buf = kmalloc(len, GFP_KERNEL); + buf = kmalloc(len, GFP_KERNEL | __GFP_NOWARN); if (unlikely(!buf)) { mutex_unlock(&epdata->lock); return -ENOMEM; @@ -675,7 +675,7 @@ ep_write_iter(struct kiocb *iocb, struct iov_iter *from) return -EBADMSG; } - buf = kmalloc(len, GFP_KERNEL); + buf = kmalloc(len, GFP_KERNEL | __GFP_NOWARN); if (unlikely(!buf)) { mutex_unlock(&epdata->lock); return -ENOMEM; From 9c9117acb0b20bab1229fb487b78c85112b56ee7 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Tue, 25 Aug 2026 15:11:57 +0000 Subject: [PATCH 0887/1012] mailmap: update email address for Bradley Morgan I switched to brads@mainlining.org for kernel work, so map the old include@grrlz.net address over to keep shortlog and blame from splitting commits between the two. Link: https://lore.kernel.org/20260825151157.4533-1-brads@mainlining.org Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton --- .mailmap | 1 + 1 file changed, 1 insertion(+) diff --git a/.mailmap b/.mailmap index 389b94a0124e31..3940b0a12c2820 100644 --- a/.mailmap +++ b/.mailmap @@ -171,6 +171,7 @@ Boris Brezillon Boris Brezillon Boris Brezillon Boris Brezillon +Bradley Morgan Brendan Higgins Brendan Jackman Brian Avery From 30bae35e0f7553a110cd64ef28bc28b67b3cc9df Mon Sep 17 00:00:00 2001 From: Thorsten Blum Date: Wed, 26 Aug 2026 12:09:03 +0200 Subject: [PATCH 0888/1012] init: fix early boot crash with bare hostname parameter When a bare hostname parameter is specified on the kernel command line without the '=' separator, early parameter parsing passes NULL to early_hostname(), which dereferences it in strscpy() and can crash the system during early boot. Reject NULL values in early_hostname() and return -EINVAL instead. Link: https://lore.kernel.org/20260826100904.296151-2-blum@kernel.org Fixes: 5a704629f2c1 ("init: add "hostname" kernel parameter") Signed-off-by: Thorsten Blum Signed-off-by: Andrew Morton Cc: Dan Moulding Cc: --- init/version.c | 3 +++ 1 file changed, 3 insertions(+) diff --git a/init/version.c b/init/version.c index 94c96f6fbfe6a2..0bd5c45aabc463 100644 --- a/init/version.c +++ b/init/version.c @@ -23,6 +23,9 @@ static int __init early_hostname(char *arg) size_t maxlen = bufsize - 1; ssize_t arglen; + if (!arg) + return -EINVAL; + arglen = strscpy(init_uts_ns.name.nodename, arg, bufsize); if (arglen < 0) { pr_warn("hostname parameter exceeds %zd characters and will be truncated", From 89ca9a9f16735bb4f9f24aaa7000bf1206a91434 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Wed, 26 Aug 2026 19:26:59 +0800 Subject: [PATCH 0889/1012] ocfs2: fix deadlock in inline-data truncate transactions Updating an inode xattr can cause an ABBA deadlock with inline file truncation: ocfs2_truncate_file() down_write(&oi->ip_alloc_sem) ocfs2_truncate_inline() ocfs2_start_trans() ocfs2_xattr_set() ocfs2_start_trans() ocfs2_xattr_ibody_set() down_write(&oi->ip_alloc_sem) The xattr set path starts the merged transaction before the inode-body xattr helper acquires ip_alloc_sem, reversing the ip_alloc_sem -> transaction order used by the allocation and truncate paths. The transaction merge in commit 85db90e77806 ("ocfs2/xattr: Merge xattr set transaction.") introduced this ordering. Fix it by acquiring ip_alloc_sem once in ocfs2_xattr_set(), before xattr preparation, allocation reservations and ocfs2_start_trans(), and removing the per-helper acquisition from ocfs2_xattr_ibody_find(), ocfs2_xattr_ibody_set() and ocfs2_xattr_create_index_block(). These helpers now assert via lockdep that the caller holds ip_alloc_sem. ocfs2_xattr_set_handle(), which only sets initial ACL or security xattrs on unpublished inodes inside the create transaction, takes ip_alloc_sem under a dedicated lockdep subclass so that the assertions hold without creating a transaction -> ip_alloc_sem cycle against the ip_alloc_sem -> transaction order. The inode is unpublished, so the acquisition can never contend. This keeps the established ip_alloc_sem -> transaction order and makes the locking unconditional, so lockdep can verify a single plain ordering instead of conditional acquisitions. Link: https://lore.kernel.org/20260826112659.246574-1-joseph.qi@linux.alibaba.com Fixes: 85db90e77806 ("ocfs2/xattr: Merge xattr set transaction.") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: ZhengYuan Huang Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao --- fs/ocfs2/xattr.c | 79 ++++++++++++++++++++++++++++++++---------------- 1 file changed, 53 insertions(+), 26 deletions(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index bfafe059bedff0..3f83a75d29a6da 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -2909,6 +2909,9 @@ static int ocfs2_xattr_has_space_inline(struct inode *inode, * * Find extended attribute in inode block and * fill search info into struct ocfs2_xattr_search. + * + * The inline free-space check races with truncate and allocation, so + * callers must hold ip_alloc_sem for writing. */ static int ocfs2_xattr_ibody_find(struct inode *inode, int name_index, @@ -2920,13 +2923,13 @@ static int ocfs2_xattr_ibody_find(struct inode *inode, int ret; int has_space = 0; + lockdep_assert_held_write(&oi->ip_alloc_sem); + if (inode->i_sb->s_blocksize == OCFS2_MIN_BLOCKSIZE) return 0; if (!(oi->ip_dyn_features & OCFS2_INLINE_XATTR_FL)) { - down_read(&oi->ip_alloc_sem); has_space = ocfs2_xattr_has_space_inline(inode, di); - up_read(&oi->ip_alloc_sem); if (!has_space) return 0; } @@ -3007,6 +3010,7 @@ static int ocfs2_xattr_ibody_init(struct inode *inode, * * Set, replace or remove an extended attribute into inode block. * + * Callers must hold ip_alloc_sem for writing. */ static int ocfs2_xattr_ibody_set(struct inode *inode, struct ocfs2_xattr_info *xi, @@ -3017,16 +3021,17 @@ static int ocfs2_xattr_ibody_set(struct inode *inode, struct ocfs2_inode_info *oi = OCFS2_I(inode); struct ocfs2_xa_loc loc; + lockdep_assert_held_write(&oi->ip_alloc_sem); + if (inode->i_sb->s_blocksize == OCFS2_MIN_BLOCKSIZE) return -ENOSPC; - down_write(&oi->ip_alloc_sem); if (!(oi->ip_dyn_features & OCFS2_INLINE_XATTR_FL)) { ret = ocfs2_xattr_ibody_init(inode, xs->inode_bh, ctxt); if (ret) { if (ret != -ENOSPC) mlog_errno(ret); - goto out; + return ret; } } @@ -3036,13 +3041,10 @@ static int ocfs2_xattr_ibody_set(struct inode *inode, if (ret) { if (ret != -ENOSPC) mlog_errno(ret); - goto out; + return ret; } xs->here = loc.xl_entry; -out: - up_write(&oi->ip_alloc_sem); - return ret; } @@ -3686,6 +3688,18 @@ static int __ocfs2_xattr_set_handle(struct inode *inode, return ret; } +/* + * ip_alloc_sem subclass for inodes being initialized before publication. + * ocfs2_xattr_set_handle() runs inside the create transaction, so taking + * ip_alloc_sem there adds a transaction -> ip_alloc_sem order that would + * form a lockdep cycle with the ip_alloc_sem -> transaction order used + * elsewhere, if not for this separate subclass. The inode is unpublished + * so the acquisition can never contend. + */ +enum { + OCFS2_IP_ALLOC_SEM_UNPUBLISHED = 1, +}; + /* * This helper is only for setting initial ACL or security xattrs on an inode * that is still unpublished, unhashed, and unattached to a dentry. @@ -3747,6 +3761,13 @@ int ocfs2_xattr_set_handle(handle_t *handle, xis.inode_bh = xbs.inode_bh = di_bh; di = (struct ocfs2_dinode *)di_bh->b_data; + /* + * The inode is unpublished and cannot contend, but take the + * semaphore anyway so the helpers' lockdep assertions hold. + */ + down_write_nested(&OCFS2_I(inode)->ip_alloc_sem, + OCFS2_IP_ALLOC_SEM_UNPUBLISHED); + ret = ocfs2_xattr_ibody_find(inode, name_index, name, &xis); if (ret) goto cleanup; @@ -3759,6 +3780,7 @@ int ocfs2_xattr_set_handle(handle_t *handle, ret = __ocfs2_xattr_set_handle(inode, di, &xi, &xis, &xbs, &ctxt); cleanup: + up_write(&OCFS2_I(inode)->ip_alloc_sem); brelse(xbs.xattr_bh); ocfs2_xattr_bucket_free(xbs.bucket); @@ -3827,30 +3849,38 @@ int ocfs2_xattr_set(struct inode *inode, di = (struct ocfs2_dinode *)di_bh->b_data; down_write(&OCFS2_I(inode)->ip_xattr_sem); + /* + * The allocation and truncate paths take ip_alloc_sem before + * starting a transaction, so take it here before xattr + * preparation, allocation reservations and ocfs2_start_trans() + * to keep that order. The xattr helpers below no longer take + * it themselves. + */ + down_write(&OCFS2_I(inode)->ip_alloc_sem); /* * Scan inode and external block to find the same name * extended attribute and collect search information. */ ret = ocfs2_xattr_ibody_find(inode, name_index, name, &xis); if (ret) - goto cleanup; + goto out_free_ac; if (xis.not_found) { ret = ocfs2_xattr_block_find(inode, name_index, name, &xbs); if (ret) - goto cleanup; + goto out_free_ac; } if (xis.not_found && xbs.not_found) { ret = -ENODATA; if (flags & XATTR_REPLACE) - goto cleanup; + goto out_free_ac; ret = 0; if (!value) - goto cleanup; + goto out_free_ac; } else { ret = -EEXIST; if (flags & XATTR_CREATE) - goto cleanup; + goto out_free_ac; } /* Check whether the value is refcounted and do some preparation. */ @@ -3861,7 +3891,7 @@ int ocfs2_xattr_set(struct inode *inode, &ref_meta, &ref_credits); if (ret) { mlog_errno(ret); - goto cleanup; + goto out_free_ac; } } @@ -3872,7 +3902,7 @@ int ocfs2_xattr_set(struct inode *inode, if (ret < 0) { inode_unlock(tl_inode); mlog_errno(ret); - goto cleanup; + goto out_free_ac; } } inode_unlock(tl_inode); @@ -3881,7 +3911,7 @@ int ocfs2_xattr_set(struct inode *inode, &xbs, &ctxt, ref_meta, &credits); if (ret) { mlog_errno(ret); - goto cleanup; + goto out_free_ac; } /* we need to update inode's ctime field, so add credit for it. */ @@ -3899,6 +3929,7 @@ int ocfs2_xattr_set(struct inode *inode, ocfs2_commit_trans(osb, ctxt.handle); out_free_ac: + up_write(&OCFS2_I(inode)->ip_alloc_sem); if (ctxt.data_ac) ocfs2_free_alloc_context(ctxt.data_ac); if (ctxt.meta_ac) @@ -3907,7 +3938,6 @@ int ocfs2_xattr_set(struct inode *inode, ocfs2_schedule_truncate_log_flush(osb, 1); ocfs2_run_deallocs(osb, &ctxt.dealloc); -cleanup: if (ref_tree) ocfs2_unlock_refcount_tree(osb, ref_tree, 1); up_write(&OCFS2_I(inode)->ip_xattr_sem); @@ -4508,6 +4538,10 @@ static void ocfs2_xattr_update_xattr_search(struct inode *inode, xs->here = &xs->header->xh_entries[i]; } +/* + * Caller must hold ip_alloc_sem for writing, since a new xattr block + * is allocated and the xattr block header is rewritten. + */ static int ocfs2_xattr_create_index_block(struct inode *inode, struct ocfs2_xattr_search *xs, struct ocfs2_xattr_set_ctxt *ctxt) @@ -4523,19 +4557,14 @@ static int ocfs2_xattr_create_index_block(struct inode *inode, struct ocfs2_xattr_tree_root *xr; u16 xb_flags = le16_to_cpu(xb->xb_flags); + lockdep_assert_held_write(&oi->ip_alloc_sem); + trace_ocfs2_xattr_create_index_block_begin( (unsigned long long)xb_bh->b_blocknr); BUG_ON(xb_flags & OCFS2_XATTR_INDEXED); BUG_ON(!xs->bucket); - /* - * XXX: - * We can use this lock for now, and maybe move to a dedicated mutex - * if performance becomes a problem later. - */ - down_write(&oi->ip_alloc_sem); - ret = ocfs2_journal_access_xb(handle, INODE_CACHE(inode), xb_bh, OCFS2_JOURNAL_ACCESS_WRITE); if (ret) { @@ -4597,8 +4626,6 @@ static int ocfs2_xattr_create_index_block(struct inode *inode, ocfs2_journal_dirty(handle, xb_bh); out: - up_write(&oi->ip_alloc_sem); - return ret; } From 45fca805089c7238cecf9886515487af78958d29 Mon Sep 17 00:00:00 2001 From: ZhengYuan Huang Date: Thu, 6 Aug 2026 16:50:12 +0800 Subject: [PATCH 0890/1012] ocfs2: reject inconsistent local xattr entries [BUG] A corrupt OCFS2 xattr entry can set OCFS2_XATTR_ENTRY_LOCAL while keeping xe_value_size larger than OCFS2_XATTR_INLINE_SIZE. When that entry reaches namevalue_size_xe(), the filesystem hits its BUG_ON: kernel BUG at fs/ocfs2/xattr.c:231! Oops: invalid opcode: 0000 [#1] SMP KASAN NOPTI RIP: 0010:namevalue_size_xe fs/ocfs2/xattr.c:231 [inline] RIP: 0010:ocfs2_xa_block_wipe_namevalue+0x2e4/0x330 fs/ocfs2/xattr.c:1638 Call Trace: ocfs2_xa_wipe_namevalue fs/ocfs2/xattr.c:1470 [inline] ocfs2_xa_remove_entry+0xae/0x1d0 fs/ocfs2/xattr.c:1941 ocfs2_xa_remove fs/ocfs2/xattr.c:2043 [inline] ocfs2_xa_set+0x11a8/0x30a0 fs/ocfs2/xattr.c:2247 ocfs2_xattr_ibody_set+0x302/0xc50 fs/ocfs2/xattr.c:2795 __ocfs2_xattr_set_handle+0x7e6/0xdb0 fs/ocfs2/xattr.c:3416 ocfs2_xattr_set+0x1447/0x2610 fs/ocfs2/xattr.c:3650 ocfs2_xattr_security_set+0x37/0x50 fs/ocfs2/xattr.c:7241 __vfs_removexattr+0x14d/0x1d0 fs/xattr.c:518 cap_inode_killpriv+0x29/0x50 security/commoncap.c:355 security_inode_killpriv+0x105/0x220 security/security.c:2724 setattr_prepare+0x147/0x8a0 fs/attr.c:219 ocfs2_setattr+0x504/0x1fd0 fs/ocfs2/file.c:1148 notify_change+0x4b5/0x1030 fs/attr.c:546 do_truncate+0x1d2/0x230 fs/open.c:68 handle_truncate fs/namei.c:3596 [inline] do_open fs/namei.c:3979 [inline] path_openat+0x260f/0x2ce0 fs/namei.c:4134 do_filp_open+0x1f6/0x430 fs/namei.c:4161 do_sys_openat2+0x117/0x1c0 fs/open.c:1437 ... [CAUSE] namevalue_size_xe() assumes that local entries contain an inline value no larger than OCFS2_XATTR_INLINE_SIZE. Existing xattr metadata validation only checks whether the value fits the storage region, and cached entries can reach lookup and bucket maintenance paths without a semantic check. A corrupt entry can therefore be passed to namevalue_size_xe(). [FIX] Validate the local/value-size invariant in the existing flat and bucket metadata validators and before accepting matched entries or traversing bucket entries in paths that call namevalue_size_xe(). Return an OCFS2 corruption error instead of firing the assertion. Link: https://lore.kernel.org/20260806085012.2650042-1-gality369@gmail.com Signed-off-by: ZhengYuan Huang Signed-off-by: Andrew Morton Reviewed-by: Joseph Qi Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao --- fs/ocfs2/xattr.c | 53 +++++++++++++++++++++++++++++++++++++++++------- 1 file changed, 46 insertions(+), 7 deletions(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index 3f83a75d29a6da..740d4bb3890f9c 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -237,6 +237,21 @@ static int namevalue_size_xe(struct ocfs2_xattr_entry *xe) return namevalue_size(xe->xe_name_len, value_len); } +static int ocfs2_validate_xattr_entry(struct super_block *sb, u64 blkno, + struct ocfs2_xattr_entry *xe) +{ + u64 value_len = le64_to_cpu(xe->xe_value_size); + + if (value_len > OCFS2_XATTR_INLINE_SIZE && + ocfs2_xattr_is_local(xe)) + return ocfs2_error(sb, + "Invalid local xattr in block %llu: value size %llu\n", + (unsigned long long)blkno, + (unsigned long long)value_len); + + return 0; +} + static int ocfs2_xattr_bucket_get_name_value(struct super_block *sb, struct ocfs2_xattr_header *xh, @@ -989,7 +1004,7 @@ static int ocfs2_validate_xattr_entries_flat(struct super_block *sb, u64 blkno, size_t entries_limit = region_size; size_t nv_limit = region_size; size_t max_entries; - int i; + int i, ret; if (region_size < sizeof(*xh)) return ocfs2_error(sb, @@ -1009,6 +1024,11 @@ static int ocfs2_validate_xattr_entries_flat(struct super_block *sb, u64 blkno, struct ocfs2_xattr_entry *xe = &xh->xh_entries[i]; size_t name_offset = le16_to_cpu(xe->xe_name_offset); size_t value_offset; + u64 value_len = le64_to_cpu(xe->xe_value_size); + + ret = ocfs2_validate_xattr_entry(sb, blkno, xe); + if (ret) + return ret; if (name_offset > nv_limit || xe->xe_name_len > nv_limit - name_offset) @@ -1023,8 +1043,7 @@ static int ocfs2_validate_xattr_entries_flat(struct super_block *sb, u64 blkno, (unsigned long long)blkno, i); if (ocfs2_xattr_is_local(xe)) { - if (le64_to_cpu(xe->xe_value_size) > - nv_limit - value_offset) + if (value_len > nv_limit - value_offset) return ocfs2_error(sb, "Invalid xattr in block %llu: entry %d value is out of bounds\n", (unsigned long long)blkno, @@ -1109,7 +1128,7 @@ static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, size_t entries_limit = sb->s_blocksize; size_t nv_limit = sb->s_blocksize; size_t max_entries; - int i; + int i, ret; if (region_size < sizeof(*xh)) return ocfs2_error(sb, @@ -1137,6 +1156,11 @@ static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, size_t block_off = name_offset >> sb->s_blocksize_bits; size_t block_offset = name_offset % nv_limit; size_t value_offset; + u64 value_len = le64_to_cpu(xe->xe_value_size); + + ret = ocfs2_validate_xattr_entry(sb, blkno, xe); + if (ret) + return ret; if (name_offset >= region_size || block_off >= bucket->bu_blocks) return ocfs2_error(sb, @@ -1155,8 +1179,7 @@ static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, (unsigned long long)blkno, i); if (ocfs2_xattr_is_local(xe)) { - if (le64_to_cpu(xe->xe_value_size) > - nv_limit - value_offset) + if (value_len > nv_limit - value_offset) return ocfs2_error(sb, "Invalid xattr bucket %llu: entry %d value is out of bounds\n", (unsigned long long)blkno, @@ -1304,7 +1327,7 @@ static int ocfs2_xattr_find_entry(struct inode *inode, int name_index, { struct ocfs2_xattr_entry *entry; size_t name_len; - int i, name_offset, cmp = 1; + int i, name_offset, cmp = 1, ret; if (name == NULL) return -EINVAL; @@ -1327,6 +1350,12 @@ static int ocfs2_xattr_find_entry(struct inode *inode, int name_index, return -EFSCORRUPTED; } cmp = memcmp(name, (xs->base + name_offset), name_len); + if (!cmp) { + ret = ocfs2_validate_xattr_entry(inode->i_sb, + OCFS2_I(inode)->ip_blkno, entry); + if (ret) + return ret; + } } if (cmp == 0) break; @@ -4068,6 +4097,10 @@ static int ocfs2_find_xe_in_bucket(struct inode *inode, xe_name = bucket_block(bucket, block_off) + new_offset; if (!memcmp(name, xe_name, name_len)) { + ret = ocfs2_validate_xattr_entry(inode->i_sb, + OCFS2_I(inode)->ip_blkno, xe); + if (ret) + break; *xe_index = i; *found = 1; ret = 0; @@ -4705,6 +4738,9 @@ static int ocfs2_defrag_xattr_bucket(struct inode *inode, xe = xh->xh_entries; end = OCFS2_XATTR_BUCKET_SIZE; for (i = 0; i < le16_to_cpu(xh->xh_count); i++, xe++) { + ret = ocfs2_validate_xattr_entry(inode->i_sb, blkno, xe); + if (ret) + goto out; offset = le16_to_cpu(xe->xe_name_offset); len = namevalue_size_xe(xe); @@ -4987,6 +5023,9 @@ static int ocfs2_divide_xattr_bucket(struct inode *inode, name_value_len = 0; for (i = 0; i < start; i++) { xe = &xh->xh_entries[i]; + ret = ocfs2_validate_xattr_entry(inode->i_sb, blk, xe); + if (ret) + goto out; name_value_len += namevalue_size_xe(xe); if (le16_to_cpu(xe->xe_name_offset) < name_offset) name_offset = le16_to_cpu(xe->xe_name_offset); From a6dc8fb0802c33909b43d9bec2f1dedc6a99f46e Mon Sep 17 00:00:00 2001 From: Geert Uytterhoeven Date: Mon, 24 Aug 2026 17:14:01 +0200 Subject: [PATCH 0891/1012] raid/kunit: enable RAID6 PQ and XOR benchmarks if KUNIT_ALL_TESTS=m Enabling the (possibly long-running benchmarks) by default may cause a big delay in boot time in case of built-in tests. However, they can still safely be enabled by default if all tests are modular, as they would only run when requested explicitly by the system administrator. Link: https://lore.kernel.org/64c6e0191bd8ccef0074ffbb09bd0584680d710b.1787584360.git.geert@linux-m68k.org Signed-off-by: Geert Uytterhoeven Signed-off-by: Andrew Morton Reviewed-by: Christoph Hellwig Acked-by: Ard Biesheuvel Cc: Eric Biggers --- lib/raid/Kconfig | 2 ++ 1 file changed, 2 insertions(+) diff --git a/lib/raid/Kconfig b/lib/raid/Kconfig index 01f007b2522cfc..563a178aa930c4 100644 --- a/lib/raid/Kconfig +++ b/lib/raid/Kconfig @@ -32,6 +32,7 @@ config XOR_KUNIT_TEST config XOR_BENCHMARK bool "Benchmark for xor_gen" depends on XOR_KUNIT_TEST + default y if KUNIT_ALL_TESTS=m help Include benchmarks in the KUnit test suite for xor_gen. @@ -63,6 +64,7 @@ config RAID6_PQ_KUNIT_TEST config RAID6_PQ_KUNIT_BENCHMARK bool "Benchmark for RAID6 PQ" depends on RAID6_PQ_KUNIT_TEST + default y if KUNIT_ALL_TESTS=m help Include benchmarks in the KUnit test suite for raid P/Q generation. From a56663d4071255317763d3e2bae1319cdf531cea Mon Sep 17 00:00:00 2001 From: Daehyeon Ko <4ncienth@gmail.com> Date: Mon, 24 Aug 2026 13:23:59 +0900 Subject: [PATCH 0892/1012] ipc/mqueue: release notification resources during inode eviction mqueue_flush_file() removes an mq_notify() registration only when the closing task belongs to the thread group stored in notify_owner. A task in a separate thread group created with CLONE_FILES can register SIGEV_THREAD notification and exit without closing the shared file table. If another thread group then unlinks and last-closes the queue, ->flush() skips the registration and inode eviction loses the only pointers to its resources. The orphaned registration permanently retains the notification skb, its netlink socket, a pid reference and a user namespace reference. An unprivileged process can repeat the sequence with new queues and sockets. No inode users remain during eviction. Remove any stale registration there after dropping info->lock, since netlink_sendskb() may release the final socket reference. This bug creates an unkillable kernel resource leak by failing to free netlink socket, PID, and user namespace references when a POSIX message queue is evicted. An unprivileged process can exploit this leak repeatedly to cause kernel memory exhaustion and lead to a DoS. Link: https://lore.kernel.org/20260824042359.925145-1-4ncienth@gmail.com Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Signed-off-by: Daehyeon Ko <4ncienth@gmail.com> Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Davidlohr Bueso Cc: Al Viro Cc: Christian Brauner Cc: Jan Kara Cc: Manfred Spraul Cc: NeilBrown Cc: --- ipc/mqueue.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/ipc/mqueue.c b/ipc/mqueue.c index d1dd36a651b0d0..d1a1965c98118c 100644 --- a/ipc/mqueue.c +++ b/ipc/mqueue.c @@ -528,6 +528,13 @@ static void mqueue_evict_inode(struct inode *inode) list_add_tail(&msg->m_list, &tmp_msg); kfree(info->node_cache); spin_unlock(&info->lock); + /* + * A shared file table can let the notification owner exit without + * running ->flush(). No users of the inode remain during eviction, so + * tear down any stale notification after dropping info->lock because + * netlink_sendskb() may release the final socket reference. + */ + remove_notification(info); list_for_each_entry_safe(msg, nmsg, &tmp_msg, m_list) { list_del(&msg->m_list); From 8479f4bf1f3c0220a71a8f31e54e135070d3b95b Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Thu, 27 Aug 2026 06:17:08 +0200 Subject: [PATCH 0893/1012] klist: avoid accesses after waking klist_remove() klist_remove() waits until a node is unreferenced so that its caller can free the containing object. klist_release() currently publishes waiter->woken and wakes the waiter before its final accesses to the waiter and node. klist_remove() can then return, allowing its stack waiter and the containing object to be freed or reused while klist_release() is still running. In particular, bus_remove_driver() can free drv->p while __device_attach() walks the same bus klist with bus_for_each_drv(). On an arm64 Cortex-A72 system, an unpatched 7.2.0-rc3 kernel with CONFIG_PREEMPT_RT=y and CONFIG_KASAN=y reproduced the bug through the in-tree I2C/at24 path. KASAN reported a use-after-free in klist_dec_and_del() reached from klist_next()/bus_for_each_drv() while at24 was being unregistered. Clear n_klist and take a task reference before publishing woken. Use release/acquire accesses for that publication and wake the referenced task. The task reference keeps the waiter task alive if it returns and exits before wake_up_process(). Link: https://lore.kernel.org/20260827041708.31682-1-kmehltretter@gmail.com Fixes: 8b0c250be489 ("[PATCH] add klist_node_attached() to determine if a node is on a list or not.") Fixes: 210272a28465 ("driver core: Remove completion from struct klist_node") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Danilo Krummrich Cc: Greg Kroah-Hartman Cc: Matthew Wilcox (Oracle) Cc: "Rafael J. Wysocki" Cc: --- lib/klist.c | 16 ++++++++++++---- 1 file changed, 12 insertions(+), 4 deletions(-) diff --git a/lib/klist.c b/lib/klist.c index 332a4fbf18ff08..f133740b1c2cb7 100644 --- a/lib/klist.c +++ b/lib/klist.c @@ -36,6 +36,7 @@ #include #include #include +#include /* * Use the lowest bit of n_klist to mark deleted nodes and exclude @@ -187,18 +188,24 @@ static void klist_release(struct kref *kref) WARN_ON(!knode_dead(n)); list_del(&n->n_node); + knode_set_klist(n, NULL); spin_lock(&klist_remove_lock); list_for_each_entry_safe(waiter, tmp, &klist_remove_waiters, list) { + struct task_struct *p; + if (waiter->node != n) continue; + p = waiter->process; + get_task_struct(p); list_del(&waiter->list); - waiter->woken = 1; + /* Publish only after the final waiter and n accesses */ + smp_store_release(&waiter->woken, 1); mb(); - wake_up_process(waiter->process); + wake_up_process(p); + put_task_struct(p); } spin_unlock(&klist_remove_lock); - knode_set_klist(n, NULL); } static int klist_dec_and_del(struct klist_node *n) @@ -250,7 +257,8 @@ void klist_remove(struct klist_node *n) for (;;) { set_current_state(TASK_UNINTERRUPTIBLE); - if (waiter.woken) + /* Pairs with the release store in klist_release() */ + if (smp_load_acquire(&waiter.woken)) break; schedule(); } From c699651b206c2bae73a927ec6bc81711e96ea185 Mon Sep 17 00:00:00 2001 From: Vishal Badole Date: Wed, 26 Aug 2026 22:45:37 +0530 Subject: [PATCH 0894/1012] lib/group_cpus: snapshot cluster masks to keep grouping hotplug invariant group_cpus_evenly() builds the managed-IRQ affinity spread used by multi-queue devices such as NVMe. That spread is meant to be a property of the static CPU topology: it walks cpu_present_mask and then cpu_possible_mask so every hardware queue owns a fixed set of CPUs, including CPUs that are offline at the time. A driver depends on that partition staying stable across re-computation - the CPUs a queue is given at probe must still describe the same queue after the device is later reset and its affinity recomputed. On an AMD system that stability breaks across an s2idle cycle. With CPUs 3-11 offlined and only CPUs 0-2 left online, the machine is suspended to s2idle and resumed. The NVMe controller uses the simple-suspend quirk, so resume fully re-initialises it and recomputes the affinity spread. The system then hangs for roughly two minutes and stays sluggish afterwards, the controller only making progress through its command-timeout poll: nvme nvme0: I/O tag 898 (3382) QID 9 timeout, completion polled nvme nvme0: I/O tag 398 (618e) QID 11 timeout, completion polled QID 9 and QID 11 are the queues whose CPUs were offline when the spread was recomputed. "completion polled" means the commands did finish in hardware, but their interrupts were never delivered to a CPU that was watching the queue, so nothing reaped them until the timeout fired. It happens because commit 89802ca36c96 ("lib/group_cpus: make group CPU cluster aware") derives the cluster groups from topology_cluster_cpumask(), which lists only the cluster siblings that are online when it is called. The resulting partition therefore depends on the transient online mask rather than on the topology alone. Recomputed on resume while the non-boot CPUs are still offline, it no longer matches the boot-time partition, and a queue is left with an affinity that does not cover the CPU it is meant to serve once that CPU comes back online. The dependence is on the online mask, not on any AMD-specific behaviour, so the same stall is reproducible on Intel platforms as well. Make the cluster grouping depend on the complete cluster topology rather than on whichever CPUs happen to be online. Snapshot the cluster masks once while every present CPU is online and reuse that view for every later spread. Every spread then groups from the same masks, so the partition computed when the controller is reset matches the one computed at probe and each queue's IRQ still covers the CPUs it serves. If the snapshot was never taken, the cluster path is skipped and the plain present/possible spread is used. Link: https://lore.kernel.org/20260826171537.4167367-1-Vishal.Badole@amd.com Fixes: 89802ca36c96 ("lib/group_cpus: make group CPU cluster aware") Signed-off-by: Vishal Badole Signed-off-by: Andrew Morton Cc: "Borislav Petkov (AMD)" Cc: Radu Rendec Cc: Thomas Gleixner Cc: Tim Chen Cc: Wangyang Guo Cc: Tianyou Li Cc: Dan Liang Cc: --- lib/group_cpus.c | 87 ++++++++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 85 insertions(+), 2 deletions(-) diff --git a/lib/group_cpus.c b/lib/group_cpus.c index e6e18d7a49bba6..3c2229feb9c3d2 100644 --- a/lib/group_cpus.c +++ b/lib/group_cpus.c @@ -6,6 +6,7 @@ #include #include #include +#include #include #include @@ -286,6 +287,77 @@ static void assign_cpus_to_groups(unsigned int ncpus, } } +/* + * topology_cluster_cpumask() only lists the cluster siblings that are online, + * so group_cpus_evenly() would compute a different managed-IRQ partition when + * recomputed with CPUs offline (e.g. an NVMe reset across s2idle), steering a + * queue's IRQ away from the CPU it serves. + * + * Snapshot the cluster masks once, on the first spread seen with every + * present CPU online, and reuse it so the grouping stays stable. If no + * snapshot exists (partial boot via maxcpus=/nosmp, or allocation failure) + * the cluster path is skipped and the plain present/possible spread is + * used. Only the cluster path is stabilised; grp_spread_init_one()'s + * sibling mask is unchanged. The snapshot lives for the system lifetime + * and is not refreshed for CPUs hot-added after boot. + */ +static cpumask_var_t *cluster_snapshot; +static bool cluster_snapshot_ready; +static DEFINE_MUTEX(cluster_snapshot_lock); + +static void capture_cluster_snapshot(void) +{ + cpumask_var_t *snapshot; + unsigned int cpu; + + /* Pairs with the smp_store_release() below. */ + if (smp_load_acquire(&cluster_snapshot_ready)) + return; + + /* Only capture when all present CPUs are online. */ + if (!data_race(cpumask_equal(cpu_present_mask, cpu_online_mask))) + return; + + mutex_lock(&cluster_snapshot_lock); + if (cluster_snapshot_ready) + goto out; + + snapshot = kcalloc(nr_cpu_ids, sizeof(*snapshot), GFP_KERNEL); + if (!snapshot) + goto out; + + for_each_possible_cpu(cpu) + if (!zalloc_cpumask_var(&snapshot[cpu], GFP_KERNEL)) + goto free_snapshot; + + /* Trylock: a caller may hold a lock the hotplug writer needs. */ + if (!cpus_read_trylock()) + goto free_snapshot; + + /* Recheck under the lock, which also pins the cluster masks. */ + if (!data_race(cpumask_equal(cpu_present_mask, cpu_online_mask))) { + cpus_read_unlock(); + goto free_snapshot; + } + + for_each_possible_cpu(cpu) + cpumask_copy(snapshot[cpu], topology_cluster_cpumask(cpu)); + cpus_read_unlock(); + + cluster_snapshot = snapshot; + /* Publish the filled snapshot before the ready flag. */ + smp_store_release(&cluster_snapshot_ready, true); + goto out; + +free_snapshot: + /* Unallocated entries are NULL, which free_cpumask_var() ignores. */ + for_each_possible_cpu(cpu) + free_cpumask_var(snapshot[cpu]); + kfree(snapshot); +out: + mutex_unlock(&cluster_snapshot_lock); +} + static int alloc_cluster_groups(unsigned int ncpus, unsigned int ngroups, struct cpumask *node_cpumask, @@ -299,6 +371,17 @@ static int alloc_cluster_groups(unsigned int ncpus, const struct cpumask **clusters; struct node_groups *cluster_groups; + /* + * Capture on the first spread with every present CPU online (normally + * the first device probe); later spreads reuse it. Sample the ready + * flag once so both loops below use one consistent source. + */ + capture_cluster_snapshot(); + + /* Pairs with the smp_store_release() in capture_cluster_snapshot(). */ + if (!smp_load_acquire(&cluster_snapshot_ready)) + goto no_cluster; + cpumask_copy(msk, node_cpumask); /* Probe how many clusters in this node. */ @@ -307,7 +390,7 @@ static int alloc_cluster_groups(unsigned int ncpus, if (cpu >= nr_cpu_ids) break; - cluster_mask = topology_cluster_cpumask(cpu); + cluster_mask = cluster_snapshot[cpu]; if (!cpumask_weight(cluster_mask)) goto no_cluster; /* Clean out CPUs on the same cluster. */ @@ -331,7 +414,7 @@ static int alloc_cluster_groups(unsigned int ncpus, cpumask_copy(msk, node_cpumask); for (n = 0; n < ncluster; n++) { cpu = cpumask_first(msk); - cluster_mask = topology_cluster_cpumask(cpu); + cluster_mask = cluster_snapshot[cpu]; nc = cpumask_weight_and(cluster_mask, node_cpumask); clusters[n] = cluster_mask; cluster_groups[n].id = n; From e615ed8dad01ecb1aee6b3fe9c72302eea831f21 Mon Sep 17 00:00:00 2001 From: Chris Gellermann Date: Mon, 3 Aug 2026 14:48:59 +0200 Subject: [PATCH 0895/1012] selftests/membarrier: introduce helper to get membarrier command registrations Patch series "selftests/membarrier: Skip an unregistered memory barrier test on Musl". The membarrier test "membarrier MEMBARRIER_CMD_PRIVATE_EXPEDITED not registered failure" fails in the multithreaded test scenario when using Musl libc as the command gets preregistered implicitly during thread creation. Skip the test if command registration is detected. This patch (of 2): Add a new membarrier_get_registrations() for reusage. Link: https://lore.kernel.org/20260803124900.3328789-1-christian.gellermann@codasip.com Link: https://lore.kernel.org/20260803124900.3328789-2-christian.gellermann@codasip.com Signed-off-by: Chris Gellermann Signed-off-by: Andrew Morton Tested-by: Michael Jeanson Cc: Ben Segall Cc: Dietmar Eggemann Cc: Ingo Molnar Cc: Juri Lelli Cc: K Prateek Nayak Cc: Mathieu Desnoyers Cc: Mel Gorman Cc: "Paul E . McKenney" Cc: Peter Zijlstra Cc: Shuah Khan Cc: Steven Rostedt Cc: Valentin Schneider Cc: Vincent Guittot Cc: Wei Yang --- tools/testing/selftests/membarrier/membarrier_test_impl.h | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/tools/testing/selftests/membarrier/membarrier_test_impl.h b/tools/testing/selftests/membarrier/membarrier_test_impl.h index f6d7c44b2288da..29aac3bc498871 100644 --- a/tools/testing/selftests/membarrier/membarrier_test_impl.h +++ b/tools/testing/selftests/membarrier/membarrier_test_impl.h @@ -16,6 +16,11 @@ static int sys_membarrier(int cmd, int flags) return syscall(__NR_membarrier, cmd, flags); } +static int membarrier_get_registrations(void) +{ + return sys_membarrier(MEMBARRIER_CMD_GET_REGISTRATIONS, 0); +} + static int test_membarrier_get_registrations(int cmd) { int ret, flags = 0; @@ -24,7 +29,7 @@ static int test_membarrier_get_registrations(int cmd) registrations |= cmd; - ret = sys_membarrier(MEMBARRIER_CMD_GET_REGISTRATIONS, 0); + ret = membarrier_get_registrations(); if (ret < 0) { ksft_exit_fail_msg( "%s test: flags = %d, errno = %d\n", From 5ebabf5aa51f4732b2fba7ed68d13decae641398 Mon Sep 17 00:00:00 2001 From: Chris Gellermann Date: Mon, 3 Aug 2026 14:49:00 +0200 Subject: [PATCH 0896/1012] selftests/membarrier: skip unpermitted membarrier command test if preregistered by libc On thread creation, Musl registers the private expedited memory barrier, see pthread_create [1]. Thus, invoking the barrier command will no longer be rejected by the kernel with EPERM. The test checking this will fail. Check if the memory barrier command has been registered and skip the test in this case. Link: https://git.musl-libc.org/cgit/musl/tree/src/thread/pthread_create.c#n260 [1] Link: https://lore.kernel.org/20260803124900.3328789-3-christian.gellermann@codasip.com Signed-off-by: Chris Gellermann Signed-off-by: Andrew Morton Tested-by: Michael Jeanson Cc: Ben Segall Cc: Dietmar Eggemann Cc: Ingo Molnar Cc: Juri Lelli Cc: K Prateek Nayak Cc: Mathieu Desnoyers Cc: Mel Gorman Cc: "Paul E . McKenney" Cc: Peter Zijlstra Cc: Shuah Khan Cc: Steven Rostedt Cc: Valentin Schneider Cc: Vincent Guittot Cc: Wei Yang --- .../selftests/membarrier/membarrier_test_impl.h | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tools/testing/selftests/membarrier/membarrier_test_impl.h b/tools/testing/selftests/membarrier/membarrier_test_impl.h index 29aac3bc498871..b4dcbb32538d47 100644 --- a/tools/testing/selftests/membarrier/membarrier_test_impl.h +++ b/tools/testing/selftests/membarrier/membarrier_test_impl.h @@ -113,6 +113,16 @@ static int test_membarrier_private_expedited_fail(void) int cmd = MEMBARRIER_CMD_PRIVATE_EXPEDITED, flags = 0; const char *test_name = "sys membarrier MEMBARRIER_CMD_PRIVATE_EXPEDITED not registered failure"; + /* + * Some C libraries, like Musl, register the private expedited barrier + * command when creating a thread. Expecting an EPERM on an unregistered + * command will therefore no longer work. Skip the test in this case. + */ + if (MEMBARRIER_CMD_REGISTER_PRIVATE_EXPEDITED & membarrier_get_registrations()) { + ksft_test_result_skip("%s test: Command already registered\n", test_name); + return 0; + } + if (sys_membarrier(cmd, flags) != -1) { ksft_exit_fail_msg( "%s test: flags = %d. Should fail, but passed\n", From cb80bc52be80e299f2fffa6d162e5ffef8f24a06 Mon Sep 17 00:00:00 2001 From: Florian Schmaus Date: Fri, 28 Aug 2026 17:54:07 +0200 Subject: [PATCH 0897/1012] selftests/epoll: fix race condition in multi-waiter wakeup tests In tests with multiple concurrent waiters on edge-triggered epoll instances where an emitter writes to multiple sockets (epoll16, epoll56, epoll58): When the emitter performs its first write(), ep_poll_callback() fires and wakes up both waiters because one waiter uses epoll_wait() and the other one uses poll(). This translates to different wait queues, ep->wq for epoll and ep->poll_wait for poll/select, which are both awoken by the kernel because of that single write. Next, both waiter threads invoke epoll_wait(), but since there is only one event, only one epoll_wait() will return non-zero because of the edge-triggered mode being used (in level-triggered mode, the kernel would re-queue the event because of remaining unread data). Since the second waiter sees an empty ready list, it does not increment ctx.count and the test fails spuriously with ctx.count == 1 instead of 2. Emitter (CPU 0) Thread 0 (CPU 1) Thread 1 (CPU 2) =============== ================ ================ epoll_wait(e0, -1) poll(e0, -1) [on e0->wq] [on e0->poll_wait] write(sfd[1]) | +--(Kernel wakes BOTH e0->wq and e0->poll_wait via callback)--+ | | | wakes up wakes up | | epoll_wait() reaps e1 poll() returns 1 | | (e1 removed via ET) (wants event) | | e0->rdllist is EMPTY | | | count++ (count = 1) v | | epoll_wait(e0, 0) | | sees EMPTY list! | | returns 0! | | thread exits | v | write(sfd[3]) | (event arrives too late!) v EXPECT_EQ(count, 2) <-- SPURIOUS FAILURE! Introduce waiter_entry1ap_loop() to retry poll() if the initial epoll_wait(..., 0) yielded no events. This ensures the thread waits for the subsequent write rather than failing immediately. Apply this helper in epoll16, epoll56, and for both waiter threads in epoll58. Link: https://lore.kernel.org/20260828-selftest-epoll-fix-race-v2-1-953ab57fd60a@codasip.com Fixes: f2728fe80cef ("selftests: add epoll selftests") Signed-off-by: Florian Schmaus Signed-off-by: Andrew Morton Cc: Heiher Cc: Roman Penyaev Cc: Shuah Khan Cc: Christian Brauner --- .../filesystems/epoll/epoll_wakeup_test.c | 32 +++++++++++++------ 1 file changed, 22 insertions(+), 10 deletions(-) diff --git a/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c b/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c index 81a994943e121b..b4dcbcd79773a7 100644 --- a/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c +++ b/tools/testing/selftests/filesystems/epoll/epoll_wakeup_test.c @@ -74,6 +74,24 @@ static void *waiter_entry1ap(void *data) return NULL; } +static void *waiter_entry1ap_loop(void *data) +{ + struct pollfd pfd; + struct epoll_event e; + struct epoll_mtcontext *ctx = data; + + pfd.fd = ctx->efd[0]; + pfd.events = POLLIN; + while (poll(&pfd, 1, 2000) > 0) { + if (epoll_wait(ctx->efd[0], &e, 1, 0) > 0) { + __sync_fetch_and_add(&ctx->count, 1); + break; + } + } + + return NULL; +} + static void *waiter_entry1o(void *data) { struct epoll_event e; @@ -809,7 +827,7 @@ TEST(epoll16) ASSERT_EQ(epoll_ctl(ctx.efd[0], EPOLL_CTL_ADD, ctx.sfd[2], events), 0); ctx.main = pthread_self(); - ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap, &ctx), 0); + ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap_loop, &ctx), 0); ASSERT_EQ(pthread_create(&emitter, NULL, emitter_entry2, &ctx), 0); if (epoll_wait(ctx.efd[0], events, 1, -1) > 0) @@ -2925,7 +2943,7 @@ TEST(epoll56) ASSERT_EQ(epoll_ctl(ctx.efd[0], EPOLL_CTL_ADD, ctx.efd[2], &e), 0); ctx.main = pthread_self(); - ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap, &ctx), 0); + ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap_loop, &ctx), 0); ASSERT_EQ(pthread_create(&emitter, NULL, emitter_entry2, &ctx), 0); if (epoll_wait(ctx.efd[0], &e, 1, -1) > 0) @@ -3030,7 +3048,6 @@ TEST(epoll57) TEST(epoll58) { pthread_t emitter; - struct pollfd pfd; struct epoll_event e; struct epoll_mtcontext ctx = { 0 }; @@ -3061,15 +3078,10 @@ TEST(epoll58) ASSERT_EQ(epoll_ctl(ctx.efd[0], EPOLL_CTL_ADD, ctx.efd[2], &e), 0); ctx.main = pthread_self(); - ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap, &ctx), 0); + ASSERT_EQ(pthread_create(&ctx.waiter, NULL, waiter_entry1ap_loop, &ctx), 0); ASSERT_EQ(pthread_create(&emitter, NULL, emitter_entry2, &ctx), 0); - pfd.fd = ctx.efd[0]; - pfd.events = POLLIN; - if (poll(&pfd, 1, -1) > 0) { - if (epoll_wait(ctx.efd[0], &e, 1, 0) > 0) - __sync_fetch_and_add(&ctx.count, 1); - } + waiter_entry1ap_loop(&ctx); ASSERT_EQ(pthread_join(ctx.waiter, NULL), 0); EXPECT_EQ(ctx.count, 2); From fd85c51f206a0673f177d780c00a90fa6a81ec80 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Fri, 28 Aug 2026 19:28:23 +0800 Subject: [PATCH 0898/1012] ocfs2: exit recovery thread on mount error path When a mount fails after the cluster connection has been established, e.g. in ocfs2_mount_volume(), ocfs2_fill_super() unwinds via out_debugfs/out_super and frees the osb without disabling recovery. A node failure event can concurrently launch the recovery thread, which blocks in __ocfs2_wait_on_mount() waiting for the volume state to become VOLUME_MOUNTED or VOLUME_DISABLED. As the mount error path neither sets VOLUME_DISABLED nor wakes osb_mount_event, the thread can never make progress: the kthread leaks and stays blocked on the wait queue embedded in the freed osb, which may then be accessed as freed memory. Fix it by setting VOLUME_DISABLED and waking osb_mount_event on this path so the thread bails out, and replace the plain kfree(osb->recovery_map) with ocfs2_recovery_exit(), which waits for a running recovery thread to exit before the recovery map is freed. Link: https://lore.kernel.org/20260828112825.666097-1-joseph.qi@linux.alibaba.com Fixes: f1e75d128b46 ("ocfs2: rewrite error handling of ocfs2_fill_super") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reviewed-by: Heming Zhao Cc: Changwei Ge Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh Cc: --- fs/ocfs2/super.c | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/fs/ocfs2/super.c b/fs/ocfs2/super.c index c62e389d4dd659..f785c39d1fb8a5 100644 --- a/fs/ocfs2/super.c +++ b/fs/ocfs2/super.c @@ -1169,8 +1169,17 @@ static int ocfs2_fill_super(struct super_block *sb, struct fs_context *fc) out_debugfs: debugfs_remove_recursive(osb->osb_debug_root); out_super: + /* + * A recovery thread launched by a node failure event may still be + * waiting for the volume to be mounted. Set VOLUME_DISABLED and + * wake it up, then wait for it to exit before osb is freed, + * otherwise the kthread would leak and stay blocked on the wait + * queue embedded in the freed osb. + */ + atomic_set(&osb->vol_state, VOLUME_DISABLED); + wake_up(&osb->osb_mount_event); ocfs2_release_system_inodes(osb); - kfree(osb->recovery_map); + ocfs2_recovery_exit(osb); ocfs2_delete_osb(osb); kfree(osb); out: From b18fb6db7ffbd6c3038781481dd359cc30c0095b Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Fri, 28 Aug 2026 19:28:24 +0800 Subject: [PATCH 0899/1012] ocfs2: free replay slots in ocfs2_recovery_exit() Commit ce2fcf1516d6 ("ocfs2: fix memory leak in ocfs2_mount_volume()") added ocfs2_free_replay_slots() calls to the mount error paths out_dismount and out_check_volume to fix a leak of osb->replay_map. However these calls are unlocked while the bail path of the recovery thread, which is woken up by out_dismount right before the call, also frees the replay slots under osb->recovery_lock. Both sides can thus observe a non-NULL osb->replay_map and trigger a double free. Fix this by moving ocfs2_free_replay_slots() into ocfs2_recovery_exit() after ocfs2_recovery_disable(), which waits for a running recovery thread to exit under osb->recovery_lock, and drop the unlocked call sites. Both ocfs2_dismount_volume() and the out_super path of ocfs2_fill_super() call ocfs2_recovery_exit(), so the replay slots are freed on every path. Since super.c no longer references it, make ocfs2_free_replay_slots() static again. Link: https://lore.kernel.org/20260828112825.666097-2-joseph.qi@linux.alibaba.com Fixes: ce2fcf1516d6 ("ocfs2: fix memory leak in ocfs2_mount_volume()") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reviewed-by: Heming Zhao Cc: Changwei Ge Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh Cc: --- fs/ocfs2/journal.c | 3 ++- fs/ocfs2/journal.h | 1 - fs/ocfs2/super.c | 5 +---- 3 files changed, 3 insertions(+), 6 deletions(-) diff --git a/fs/ocfs2/journal.c b/fs/ocfs2/journal.c index d8afbc1a76bb8a..a3938a03e93bf7 100644 --- a/fs/ocfs2/journal.c +++ b/fs/ocfs2/journal.c @@ -156,7 +156,7 @@ static void ocfs2_queue_replay_slots(struct ocfs2_super *osb, replay_map->rm_state = REPLAY_DONE; } -void ocfs2_free_replay_slots(struct ocfs2_super *osb) +static void ocfs2_free_replay_slots(struct ocfs2_super *osb) { struct ocfs2_replay_map *replay_map = osb->replay_map; @@ -243,6 +243,7 @@ void ocfs2_recovery_exit(struct ocfs2_super *osb) /* XXX: Should we bug if there are dirty entries? */ kfree(rm); + ocfs2_free_replay_slots(osb); } static int __ocfs2_recovery_map_test(struct ocfs2_super *osb, diff --git a/fs/ocfs2/journal.h b/fs/ocfs2/journal.h index f8b3b2a3d6309e..19fc920d26b1cd 100644 --- a/fs/ocfs2/journal.h +++ b/fs/ocfs2/journal.h @@ -151,7 +151,6 @@ void ocfs2_recovery_exit(struct ocfs2_super *osb); void ocfs2_recovery_disable_quota(struct ocfs2_super *osb); int ocfs2_compute_replay_slots(struct ocfs2_super *osb); -void ocfs2_free_replay_slots(struct ocfs2_super *osb); /* * Journal Control: * Initialize, Load, Shutdown, Wipe a journal. diff --git a/fs/ocfs2/super.c b/fs/ocfs2/super.c index f785c39d1fb8a5..6a8092b65bb558 100644 --- a/fs/ocfs2/super.c +++ b/fs/ocfs2/super.c @@ -1162,7 +1162,6 @@ static int ocfs2_fill_super(struct super_block *sb, struct fs_context *fc) out_dismount: atomic_set(&osb->vol_state, VOLUME_DISABLED); wake_up(&osb->osb_mount_event); - ocfs2_free_replay_slots(osb); ocfs2_dismount_volume(sb, 1); goto out; @@ -1776,14 +1775,12 @@ static int ocfs2_mount_volume(struct super_block *sb) status = ocfs2_truncate_log_init(osb); if (status < 0) { mlog_errno(status); - goto out_check_volume; + goto out_system_inodes; } ocfs2_super_unlock(osb, 1); return 0; -out_check_volume: - ocfs2_free_replay_slots(osb); out_system_inodes: if (osb->local_alloc_state == OCFS2_LA_ENABLED) ocfs2_shutdown_local_alloc(osb); From 8bce2bf0b00bcdc6296f28191c8048145c5f1491 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Fri, 28 Aug 2026 19:28:25 +0800 Subject: [PATCH 0900/1012] ocfs2: defer suballocator block group reclaim to workqueue When the last bit in a suballocator block group is freed, _ocfs2_free_suballoc_bits() reclaims the group back to the global bitmap. The reclaim takes inode_lock() on the global bitmap inode while running inside the freeing transaction, adding a lock dependency of j_trans_barrier -> global bitmap inode i_rwsem This forms a circular dependency with paths such as ocfs2_shutdown_local_alloc(), which take the global bitmap inode lock before starting a transaction: Task1 (dealloc): ocfs2_run_deallocs ocfs2_free_cached_blocks ocfs2_start_trans down_read(j_trans_barrier) _ocfs2_free_suballoc_bits _ocfs2_reclaim_suballoc_to_main inode_lock(main_bm_inode) <- wait on Task2 Task2 (dismount): ocfs2_shutdown_local_alloc inode_lock(main_bm_inode) ocfs2_start_trans down_read(j_trans_barrier) <- wait on Task3 Task3 (ocfs2cmt): ocfs2_commit_cache down_write(j_trans_barrier) <- wait on Task1's handle jbd2_journal_flush Task1 waits for Task2's inode_lock(), Task2 waits for the j_trans_barrier down_write() held by ocfs2cmt, and ocfs2cmt waits for Task1's running transaction to commit - a real deadlock, observed with aio-stress direct IO writes racing dismount. Fix it by deferring the reclaim to the per-superblock ocfs2_wq workqueue, so the freeing transaction no longer takes the global bitmap inode lock. The worker re-checks under the suballocator locks that the block group is still fully freed (it may have been allocated from again in the meantime), takes the global bitmap inode locks before starting its own transaction, and performs the same suballocator cleanup and space return. The inode lock order (suballocator inode -> global bitmap inode) is consistent with the existing "inode lock before transaction" order, breaking the cycle. Reclaim work can still be queued late in dismount, e.g. when the truncate log is flushed or orphan dir recovery frees inode bits, so both ocfs2_dismount_volume() and the mount error path flush ocfs2_wq right before the system inodes are released, while the journal is still alive, to make sure no reclaim work is left running. The worker also bails out if the journal is already gone. Tested with the ocfs2 testsuite (including aio-stress direct IO) and umount/mount cycles on a CONFIG_PROVE_LOCKING kernel: the circular locking dependency is gone and freed block groups are still returned to the global bitmap. Link: https://lore.kernel.org/20260828112825.666097-3-joseph.qi@linux.alibaba.com Fixes: 4a54331616b3 ("ocfs2: give ocfs2 the ability to reclaim suballocator free bg") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reviewed-by: Heming Zhao Assisted-by: Qoder:Qwen3.8-Max Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao --- fs/ocfs2/ocfs2.h | 5 ++ fs/ocfs2/suballoc.c | 202 +++++++++++++++++++++++++++++++++++++------- fs/ocfs2/suballoc.h | 2 +- fs/ocfs2/super.c | 16 ++++ 4 files changed, 195 insertions(+), 30 deletions(-) diff --git a/fs/ocfs2/ocfs2.h b/fs/ocfs2/ocfs2.h index 62cad6522c7a31..b747cdec178758 100644 --- a/fs/ocfs2/ocfs2.h +++ b/fs/ocfs2/ocfs2.h @@ -502,6 +502,11 @@ struct ocfs2_super */ struct workqueue_struct *ocfs2_wq; + /* deferred reclaim of fully freed suballocator block groups */ + spinlock_t os_suballoc_reclaim_lock; + struct list_head os_suballoc_reclaim_list; + struct work_struct os_suballoc_reclaim_work; + /* sysfs directory per partition */ struct kset *osb_dev_kset; diff --git a/fs/ocfs2/suballoc.c b/fs/ocfs2/suballoc.c index 20c3aec6b9873c..453b56be9624c6 100644 --- a/fs/ocfs2/suballoc.c +++ b/fs/ocfs2/suballoc.c @@ -2687,16 +2687,24 @@ static int ocfs2_block_group_clear_bits(handle_t *handle, * cleanup rec/alloc_inode job, then switches to the main bitmap * to reclaim released space. * + * Callers must hold inode_lock() and ocfs2_inode_lock() on + * main_bm_inode, i.e. the global bitmap inode locks must be taken + * before starting the transaction. + * * handle: The transaction handle * alloc_inode: The suballoc inode * alloc_bh: The buffer_head of suballoc inode * group_bh: The group descriptor buffer_head of suballocator managed. - * Caller should release the input group_bh. + * This function takes ownership of it and will release it. + * main_bm_inode: The global bitmap inode + * main_bm_bh: The buffer_head of the global bitmap inode */ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, struct inode *alloc_inode, struct buffer_head *alloc_bh, - struct buffer_head *group_bh) + struct buffer_head *group_bh, + struct inode *main_bm_inode, + struct buffer_head *main_bm_bh) { int idx, status = 0; int i, next_free_rec, len = 0; @@ -2706,8 +2714,6 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, u64 bg_blkno, start_blk; unsigned int count; struct ocfs2_chain_rec *rec; - struct buffer_head *main_bm_bh = NULL; - struct inode *main_bm_inode = NULL; struct ocfs2_super *osb = OCFS2_SB(alloc_inode->i_sb); struct ocfs2_dinode *fe = (struct ocfs2_dinode *) alloc_bh->b_data; struct ocfs2_chain_list *cl = &fe->id2.i_chain; @@ -2794,24 +2800,12 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, ocfs2_remove_from_cache(INODE_CACHE(alloc_inode), group_bh); memset(group, 0, sizeof(struct ocfs2_group_desc)); - /* prepare job for reclaim clusters */ - main_bm_inode = ocfs2_get_system_file_inode(osb, - GLOBAL_BITMAP_SYSTEM_INODE, - OCFS2_INVALID_SLOT); - if (!main_bm_inode) - goto bail; /* ignore the error in reclaim path */ - - inode_lock(main_bm_inode); - - status = ocfs2_inode_lock(main_bm_inode, &main_bm_bh, 1); - if (status < 0) - goto free_bm_inode; /* ignore the error in reclaim path */ - ocfs2_block_to_cluster_group(main_bm_inode, start_blk, &bg_blkno, &start_bit); fe = (struct ocfs2_dinode *) main_bm_bh->b_data; cl = &fe->id2.i_chain; - /* reuse group_bh, caller will release the input group_bh */ + /* release the suballocator group descriptor before reuse */ + brelse(group_bh); group_bh = NULL; /* reclaim clusters to global_bitmap */ @@ -2819,7 +2813,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, &group_bh); if (status < 0) { mlog_errno(status); - goto free_bm_bh; + goto bail; } group = (struct ocfs2_group_desc *) group_bh->b_data; @@ -2827,7 +2821,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, ocfs2_error(alloc_inode->i_sb, "reclaim length (%d) beyands block group length (%d)", count + start_bit, le16_to_cpu(group->bg_bits)); - goto free_group_bh; + goto bail; } old_bg_contig_free_bits = group->bg_contig_free_bits; @@ -2837,7 +2831,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, _ocfs2_clear_bit); if (status < 0) { mlog_errno(status); - goto free_group_bh; + goto bail; } status = ocfs2_journal_access_di(handle, INODE_CACHE(main_bm_inode), @@ -2847,7 +2841,7 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, ocfs2_block_group_set_bits(handle, main_bm_inode, group, group_bh, start_bit, count, le16_to_cpu(old_bg_contig_free_bits), 1); - goto free_group_bh; + goto bail; } idx = le16_to_cpu(group->bg_chain); @@ -2858,19 +2852,168 @@ static int _ocfs2_reclaim_suballoc_to_main(handle_t *handle, fe->id1.bitmap1.i_used = cpu_to_le32(tmp_used - count); ocfs2_journal_dirty(handle, main_bm_bh); -free_group_bh: +bail: brelse(group_bh); + return status; +} + +/* + * When a suballocator block group becomes fully freed, its space is + * reclaimed back to the global bitmap. Taking the global bitmap inode + * lock inside the freeing transaction would create a lock dependency + * of "j_trans_barrier -> global bitmap inode i_rwsem", which forms a + * circular dependency with paths like ocfs2_shutdown_local_alloc() that + * take the inode lock before starting a transaction, and can lead to a + * real deadlock with the ocfs2cmt journal commit thread. So queue the + * reclaim to the workqueue and let it run outside the freeing + * transaction. + */ +struct ocfs2_suballoc_reclaim_work { + struct list_head list; + struct inode *alloc_inode; + u64 bg_blkno; +}; + +static void ocfs2_queue_suballoc_reclaim(struct ocfs2_super *osb, + struct inode *alloc_inode, + u64 bg_blkno) +{ + struct ocfs2_suballoc_reclaim_work *reclaim_work; + + reclaim_work = kmalloc_obj(*reclaim_work, GFP_NOFS); + if (!reclaim_work) { + /* + * Reclaim is only a space return optimization. If we can't + * queue it, the freed block group just stays owned by the + * suballocator. + */ + return; + } + + igrab(alloc_inode); + reclaim_work->alloc_inode = alloc_inode; + reclaim_work->bg_blkno = bg_blkno; + + spin_lock(&osb->os_suballoc_reclaim_lock); + list_add_tail(&reclaim_work->list, &osb->os_suballoc_reclaim_list); + spin_unlock(&osb->os_suballoc_reclaim_lock); -free_bm_bh: + queue_work(osb->ocfs2_wq, &osb->os_suballoc_reclaim_work); +} + +static void ocfs2_do_suballoc_reclaim(struct ocfs2_super *osb, + struct ocfs2_suballoc_reclaim_work *reclaim_work) +{ + int status, i; + handle_t *handle; + struct inode *alloc_inode = reclaim_work->alloc_inode; + struct inode *main_bm_inode; + struct buffer_head *alloc_bh = NULL, *group_bh = NULL; + struct buffer_head *main_bm_bh = NULL; + struct ocfs2_dinode *fe; + struct ocfs2_chain_list *cl; + struct ocfs2_chain_rec *rec; + + /* journal already gone, e.g. during dismount cleanup */ + if (!osb->journal) + return; + + inode_lock(alloc_inode); + status = ocfs2_inode_lock(alloc_inode, &alloc_bh, 1); + if (status < 0) + goto out_alloc; + + fe = (struct ocfs2_dinode *) alloc_bh->b_data; + cl = &fe->id2.i_chain; + + /* + * The block group may have been allocated from again since the + * reclaim work was queued, re-check that it is still fully freed. + * A stale work item can also reference a group that is no longer + * chained, whose descriptor would fail validation and trigger a + * spurious ocfs2_error(), so verify chain membership first. + */ + for (i = 0; i < le16_to_cpu(cl->cl_next_free_rec); i++) { + rec = &cl->cl_recs[i]; + if (le64_to_cpu(rec->c_blkno) == reclaim_work->bg_blkno) + break; + } + if (i == le16_to_cpu(cl->cl_next_free_rec) || + ocfs2_is_cluster_bitmap(alloc_inode) || + (le32_to_cpu(rec->c_free) != (le32_to_cpu(rec->c_total) - 1)) || + (le16_to_cpu(cl->cl_next_free_rec) == 1)) + goto out_alloc_unlock; + + status = ocfs2_read_group_descriptor(alloc_inode, fe, + reclaim_work->bg_blkno, &group_bh); + if (status < 0) + goto out_alloc_unlock; + + main_bm_inode = ocfs2_get_system_file_inode(osb, + GLOBAL_BITMAP_SYSTEM_INODE, + OCFS2_INVALID_SLOT); + if (!main_bm_inode) + goto out_group; + + inode_lock(main_bm_inode); + status = ocfs2_inode_lock(main_bm_inode, &main_bm_bh, 1); + if (status < 0) + goto out_main; + + handle = ocfs2_start_trans(osb, OCFS2_SUBALLOC_FREE); + if (IS_ERR(handle)) { + status = PTR_ERR(handle); + mlog_errno(status); + goto out_main_unlock; + } + + status = _ocfs2_reclaim_suballoc_to_main(handle, alloc_inode, + alloc_bh, group_bh, + main_bm_inode, main_bm_bh); + /* group_bh ownership passed to _ocfs2_reclaim_suballoc_to_main() */ + group_bh = NULL; + if (status < 0) + mlog_errno(status); + + ocfs2_commit_trans(osb, handle); + +out_main_unlock: ocfs2_inode_unlock(main_bm_inode, 1); brelse(main_bm_bh); - -free_bm_inode: +out_main: inode_unlock(main_bm_inode); iput(main_bm_inode); +out_group: + brelse(group_bh); +out_alloc_unlock: + ocfs2_inode_unlock(alloc_inode, 1); + brelse(alloc_bh); +out_alloc: + inode_unlock(alloc_inode); +} -bail: - return status; +void ocfs2_suballoc_reclaim_worker(struct work_struct *work) +{ + struct ocfs2_super *osb = container_of(work, struct ocfs2_super, + os_suballoc_reclaim_work); + struct ocfs2_suballoc_reclaim_work *reclaim_work; + + while (1) { + spin_lock(&osb->os_suballoc_reclaim_lock); + if (list_empty(&osb->os_suballoc_reclaim_list)) { + spin_unlock(&osb->os_suballoc_reclaim_lock); + break; + } + reclaim_work = list_first_entry(&osb->os_suballoc_reclaim_list, + struct ocfs2_suballoc_reclaim_work, + list); + list_del(&reclaim_work->list); + spin_unlock(&osb->os_suballoc_reclaim_lock); + + ocfs2_do_suballoc_reclaim(osb, reclaim_work); + iput(reclaim_work->alloc_inode); + kfree(reclaim_work); + } } /* @@ -2955,7 +3098,8 @@ static int _ocfs2_free_suballoc_bits(handle_t *handle, goto bail; } - _ocfs2_reclaim_suballoc_to_main(handle, alloc_inode, alloc_bh, group_bh); + ocfs2_queue_suballoc_reclaim(OCFS2_SB(alloc_inode->i_sb), alloc_inode, + bg_blkno); bail: brelse(group_bh); diff --git a/fs/ocfs2/suballoc.h b/fs/ocfs2/suballoc.h index bcf2ed4a86310b..6042abc032f96e 100644 --- a/fs/ocfs2/suballoc.h +++ b/fs/ocfs2/suballoc.h @@ -206,7 +206,7 @@ int ocfs2_lock_allocators(struct inode *inode, struct ocfs2_extent_tree *et, int ocfs2_test_inode_bit(struct ocfs2_super *osb, u64 blkno, int *res); - +void ocfs2_suballoc_reclaim_worker(struct work_struct *work); /* * The following two interfaces are for ocfs2_create_inode_in_orphan(). diff --git a/fs/ocfs2/super.c b/fs/ocfs2/super.c index 6a8092b65bb558..1e76b1d9fe0af4 100644 --- a/fs/ocfs2/super.c +++ b/fs/ocfs2/super.c @@ -1784,6 +1784,9 @@ static int ocfs2_mount_volume(struct super_block *sb) out_system_inodes: if (osb->local_alloc_state == OCFS2_LA_ENABLED) ocfs2_shutdown_local_alloc(osb); + /* Drain pending suballoc reclaim work before the journal goes away */ + if (osb->ocfs2_wq) + flush_workqueue(osb->ocfs2_wq); ocfs2_release_system_inodes(osb); /* before journal shutdown, we should release slot_info */ ocfs2_free_slot_info(osb); @@ -1854,6 +1857,14 @@ static void ocfs2_dismount_volume(struct super_block *sb, int mnt_err) if (osb->cconn) ocfs2_super_unlock(osb, 1); + /* + * Drain pending suballoc reclaim work while the system inodes and + * the journal are still alive, since the worker needs to look up + * the global bitmap inode and start a transaction. + */ + if (osb->ocfs2_wq) + flush_workqueue(osb->ocfs2_wq); + ocfs2_release_system_inodes(osb); ocfs2_journal_shutdown(osb); @@ -2140,6 +2151,11 @@ static int ocfs2_initialize_super(struct super_block *sb, INIT_WORK(&osb->dquot_drop_work, ocfs2_drop_dquot_refs); init_llist_head(&osb->dquot_drop_list); + spin_lock_init(&osb->os_suballoc_reclaim_lock); + INIT_LIST_HEAD(&osb->os_suballoc_reclaim_list); + INIT_WORK(&osb->os_suballoc_reclaim_work, + ocfs2_suballoc_reclaim_worker); + /* get some pseudo constants for clustersize bits */ osb->s_clustersize_bits = le32_to_cpu(di->id2.i_super.s_clustersize_bits); From c38b018482d326a4f5c4b068678b337a2e791b8e Mon Sep 17 00:00:00 2001 From: Thorsten Blum Date: Fri, 28 Aug 2026 19:43:37 +0200 Subject: [PATCH 0901/1012] init: simplify early_hostname() Inline the strscpy() check and remove the redundant arglen variable. Use %zu to format the unsigned maxlen argument and add a newline after the truncation warning. Link: https://lore.kernel.org/20260828174337.609333-2-blum@kernel.org Signed-off-by: Thorsten Blum Signed-off-by: Andrew Morton --- init/version.c | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/init/version.c b/init/version.c index 0bd5c45aabc463..cbae11b6f7688b 100644 --- a/init/version.c +++ b/init/version.c @@ -21,16 +21,14 @@ static int __init early_hostname(char *arg) { size_t bufsize = sizeof(init_uts_ns.name.nodename); size_t maxlen = bufsize - 1; - ssize_t arglen; if (!arg) return -EINVAL; - arglen = strscpy(init_uts_ns.name.nodename, arg, bufsize); - if (arglen < 0) { - pr_warn("hostname parameter exceeds %zd characters and will be truncated", + if (strscpy(init_uts_ns.name.nodename, arg, bufsize) < 0) + pr_warn("hostname parameter exceeds %zu characters and will be truncated\n", maxlen); - } + return 0; } early_param("hostname", early_hostname); From dc95cc2b48c64a64b9a3d1077412216936ebb13a Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 22 Aug 2026 16:33:27 +0200 Subject: [PATCH 0902/1012] squashfs: fix fragment index table sizing overflow on 32-bit Patch series "squashfs: harden fragment index table sizing". Two integer overflows undermine fragment index table handling. One is in the original fragment sizing macros. The other is in a bounds check added by commit 1cac63cc9b2f ("Squashfs: add sanity checks to fragment reading at mount time"). Patch 1: the fragment byte count wraps on 32-bit, so the index table is allocated too small and squashfs_frag_lookup() reads out of bounds. A crafted image triggers a KASAN out-of-bounds read on a 32-bit build. With the fix the same image fails cleanly at mount. Patch 2: the check that the table fits before the next one adds two u64 values controlled by the filesystem image and can wrap. This patch (of 2): SQUASHFS_FRAGMENT_BYTES() multiplies the on-disk fragment count (an unsigned int) by sizeof(struct squashfs_fragment_entry), a size_t. On a 32-bit kernel that product is 32-bit and can wrap. squashfs_read_fragment_index_table() sizes the fragment index table from it, but squashfs_frag_lookup() bounds the fragment number against msblk->fragments, the unwrapped superblock value. The two disagree: an image declaring 0x10000001 fragments wraps the product to 16, so a single index entry is allocated, yet the lookup still accepts fragment 0x0fffffff: if (fragment >= msblk->fragments) return -EIO; block = SQUASHFS_FRAGMENT_INDEX(fragment); ... start_block = le64_to_cpu(msblk->fragment_index[block]); block is then 524287 and the read lands ~4MB past an 8-byte allocation. On a 32-bit build KASAN catches it when the crafted image is mounted and the file is stat'd. Cast to u64 in the macro so the multiplication is 64-bit on all targets. After conversion to index-table entries, SQUASHFS_FRAGMENT_INDEX_BYTES() is at most 64 MiB for any u32 count, so it fits both the unsigned int local and the int argument it feeds. 64-bit builds are unchanged. Link: https://lore.kernel.org/20260822143328.68867-1-kmehltretter@gmail.com Link: https://lore.kernel.org/20260822143328.68867-2-kmehltretter@gmail.com Fixes: ffae2cd73a9e ("Squashfs: header files") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Assisted-by: Claude:claude-opus-5 Cc: Phillip Lougher Cc: --- fs/squashfs/squashfs_fs.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/fs/squashfs/squashfs_fs.h b/fs/squashfs/squashfs_fs.h index a955d9369749f2..93436c7d80c973 100644 --- a/fs/squashfs/squashfs_fs.h +++ b/fs/squashfs/squashfs_fs.h @@ -136,7 +136,7 @@ static inline int squashfs_block_size(__le32 raw) /* fragment and fragment table defines */ #define SQUASHFS_FRAGMENT_BYTES(A) \ - ((A) * sizeof(struct squashfs_fragment_entry)) + ((u64)(A) * sizeof(struct squashfs_fragment_entry)) #define SQUASHFS_FRAGMENT_INDEX(A) (SQUASHFS_FRAGMENT_BYTES(A) / \ SQUASHFS_METADATA_SIZE) From 3394c023a177c0a6680869bb55cb4823b81c7587 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 22 Aug 2026 16:33:28 +0200 Subject: [PATCH 0903/1012] squashfs: make the fragment index table bounds check overflow-safe squashfs_read_fragment_index_table() checks that the table fits before the next one with: if (fragment_table_start + length > next_table) return ERR_PTR(-EINVAL); fragment_table_start comes from the superblock and is not validated before this point. A start of 2^64 - length wraps the sum to zero, so the check passes regardless of next_table and fails to reject the invalid table ordering. length then reaches kmalloc() through squashfs_read_table(). A fragment count of 0xffffffff asks for 64MB, order 14. GFP_KERNEL does not include __GFP_NOWARN, so the page allocator warns before the mount fails with -ENOMEM. With panic_on_warn, the warning panics the kernel. Compare the operands instead of adding them. id.c and export.c avoid the same wrap with an exact-size check. Keep the inequality here because a gap before the next table is still allowed. Link: https://lore.kernel.org/20260822143328.68867-3-kmehltretter@gmail.com Fixes: 1cac63cc9b2f ("Squashfs: add sanity checks to fragment reading at mount time") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Assisted-by: Claude:claude-opus-5 Cc: Phillip Lougher Cc: --- fs/squashfs/fragment.c | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/fs/squashfs/fragment.c b/fs/squashfs/fragment.c index 49602b9a42e19e..c46e946fa47447 100644 --- a/fs/squashfs/fragment.c +++ b/fs/squashfs/fragment.c @@ -69,9 +69,11 @@ __le64 *squashfs_read_fragment_index_table(struct super_block *sb, /* * Sanity check, length bytes should not extend into the next table - * this check also traps instances where fragment_table_start is - * incorrectly larger than the next table start + * incorrectly larger than the next table start. Both values are read + * from the filesystem image, so compare without adding them. */ - if (fragment_table_start + length > next_table) + if (fragment_table_start > next_table || + length > next_table - fragment_table_start) return ERR_PTR(-EINVAL); table = squashfs_read_table(sb, fragment_table_start, length); From e3ac687f95487453d8c4e861bdcc978ef7a7226b Mon Sep 17 00:00:00 2001 From: Aaron Tomlin Date: Sat, 29 Aug 2026 10:53:20 -0400 Subject: [PATCH 0904/1012] hung_task: reset warning budget when problem gets resolved Patch series "hung_task: Improve warning budget handling and task reporting", v10. The hung_task watchdog detects tasks stuck in TASK_UNINTERRUPTIBLE (D) state for longer than CONFIG_DEFAULT_HUNG_TASK_TIMEOUT seconds. To prevent log spam during system spikes, sysctl_hung_task_warnings enforces a budget on the number of logged warnings. However, the current implementation has two major limitations: 1. Permanent exhaustion of warning budget sysctl_hung_task_warnings is decremented directly when printing warnings. Once this budget hits zero, no further warnings are reported until an administrator manually updates the sysctl value or reboots the system. Consequently, a single temporary hang episode permanently blinds the kernel watchdog to any subsequent hung tasks after system recovery. 2. Total log suppression when budget is exhausted Once the warning budget reaches zero, hung_task_info() completely suppresses all output, including the basic single-line alert. While suppressing verbose stack dumps and lock debugging is desirable to prevent dmesg flooding, hiding basic task alerts leaves administrators entirely unaware that tasks are hanging. This patch series resolves both limitations by decoupling the configured warning limit from the active runtime budget, automatically resetting the budget upon system recovery or sysctl updates, and emitting a single aggregate summary line when hung tasks are detected under an exhausted warning budget. Patch 1 separates the configured sysctl hung_task_warnings from the runtime budget, making khungtaskd the sole owner of runtime budget updates. The budget is reloaded directly when a scan finds zero hung tasks, or via an atomic reset request published on sysctl write. Patch 2 prevents dmesg flooding during system-wide hangs by keeping non-panic per-task stack dumps budgeted, while providing ongoing visibility by logging a single aggregate summary line at the end of each scan iteration when the warning budget is exhausted. This patch (of 2): The sysctl hung_task_warnings currently holds both the configured warning limit and the remaining budget. Each detailed report decrements the sysctl, so once it reaches zero, the configured limit is lost and cannot be restored automatically. Keep sysctl_hung_task_warnings as the configured warning limit and make khungtaskd the sole owner of the remaining budget. A check that finds no hung tasks reloads the budget directly from the configured limit. A successful sysctl write publishes an atomic reset request, which khungtaskd consumes at the start of the next check. Link: https://lore.kernel.org/20260829145321.18423-1-atomlin@atomlin.com Link: https://lore.kernel.org/20260829145321.18423-2-atomlin@atomlin.com Signed-off-by: Aaron Tomlin Signed-off-by: Andrew Morton Suggested-by: Petr Mladek Suggested-by: Lance Yang Tested-by: Lance Yang Reviewed-by: Lance Yang Reviewed-by: Bradley Morgan Cc: David Laight Cc: "Masami Hiramatsu (Google)" --- Documentation/admin-guide/sysctl/kernel.rst | 5 ++- kernel/hung_task.c | 49 +++++++++++++++++---- 2 files changed, 44 insertions(+), 10 deletions(-) diff --git a/Documentation/admin-guide/sysctl/kernel.rst b/Documentation/admin-guide/sysctl/kernel.rst index ffea61d448ebb1..c03369e234a928 100644 --- a/Documentation/admin-guide/sysctl/kernel.rst +++ b/Documentation/admin-guide/sysctl/kernel.rst @@ -459,8 +459,9 @@ hung_task_warnings ================== The maximum number of warnings to report. During a check interval -if a hung task is detected, this value is decreased by 1. -When this value reaches 0, no more warnings will be reported. +if a hung task is detected, the internal warning budget is decreased by 1. +When this budget reaches 0, no more detailed warnings will be reported. The +warning budget is reset to the configured limit when no hung task is found. This file shows up if ``CONFIG_DETECT_HUNG_TASK`` is enabled. -1: report an infinite number of warnings. diff --git a/kernel/hung_task.c b/kernel/hung_task.c index 6fcc94ce4ca9d2..a5043188456d42 100644 --- a/kernel/hung_task.c +++ b/kernel/hung_task.c @@ -57,8 +57,20 @@ unsigned long __read_mostly sysctl_hung_task_timeout_secs = CONFIG_DEFAULT_HUNG_ */ static unsigned long __read_mostly sysctl_hung_task_check_interval_secs; +/* + * Limit the number of printed hung tasks to prevent printing + * the same or similar backtraces repeatedly. + */ static int __read_mostly sysctl_hung_task_warnings = 10; +/* + * The number of hung tasks which still can be reported. + * The budget gets restored to the original limit when + * the previous stall is resolved. + */ +static int hung_task_warnings_budget = 10; +static atomic_t reset_hung_task_warnings = ATOMIC_INIT(0); + static int __read_mostly did_panic; static bool hung_task_call_panic; @@ -245,11 +257,11 @@ static void hung_task_info(struct task_struct *t, unsigned long timeout, /* * The given task did not get scheduled for more than * CONFIG_DEFAULT_HUNG_TASK_TIMEOUT. Therefore, complain - * accordingly + * accordingly with full details if the budget is not exhausted. */ - if (sysctl_hung_task_warnings || hung_task_call_panic) { - if (sysctl_hung_task_warnings > 0) - sysctl_hung_task_warnings--; + if (hung_task_warnings_budget || hung_task_call_panic) { + if (hung_task_warnings_budget > 0) + hung_task_warnings_budget--; pr_err("INFO: task %s:%d blocked%s for more than %ld seconds.\n", t->comm, t->pid, t->in_iowait ? " in I/O wait" : "", (jiffies - t->last_switch_time) / HZ); @@ -264,7 +276,7 @@ static void hung_task_info(struct task_struct *t, unsigned long timeout, sched_show_task(t); debug_show_blocker(t, timeout); - if (!sysctl_hung_task_warnings) + if (!hung_task_warnings_budget) pr_info("Future hung task reports are suppressed, see sysctl kernel.hung_task_warnings\n"); } @@ -304,7 +316,7 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout) unsigned long last_break = jiffies; struct task_struct *g, *t; unsigned long this_round_count; - int need_warning = sysctl_hung_task_warnings; + int need_warning; unsigned long si_mask = hung_task_si_mask; /* @@ -314,6 +326,11 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout) if (test_taint(TAINT_DIE) || did_panic) return; + if (atomic_xchg_acquire(&reset_hung_task_warnings, 0)) + hung_task_warnings_budget = + READ_ONCE(sysctl_hung_task_warnings); + need_warning = hung_task_warnings_budget; + this_round_count = 0; rcu_read_lock(); for_each_process_thread(g, t) { @@ -340,8 +357,11 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout) unlock: rcu_read_unlock(); - if (!this_round_count) + if (!this_round_count) { + hung_task_warnings_budget = + READ_ONCE(sysctl_hung_task_warnings); return; + } if (need_warning || hung_task_call_panic) { si_mask |= SYS_INFO_LOCKS; @@ -425,6 +445,19 @@ static int proc_dohung_task_timeout_secs(const struct ctl_table *table, int writ return ret; } +static int proc_dohung_task_warnings(const struct ctl_table *table, int write, + void *buffer, + size_t *lenp, loff_t *ppos) +{ + int ret; + + ret = proc_dointvec_minmax(table, write, buffer, lenp, ppos); + if (!ret && write) + atomic_set_release(&reset_hung_task_warnings, 1); + + return ret; +} + /* * This is needed for proc_doulongvec_minmax of sysctl_hung_task_timeout_secs * and hung_task_check_interval_secs @@ -480,7 +513,7 @@ static const struct ctl_table hung_task_sysctls[] = { .data = &sysctl_hung_task_warnings, .maxlen = sizeof(int), .mode = 0644, - .proc_handler = proc_dointvec_minmax, + .proc_handler = proc_dohung_task_warnings, .extra1 = SYSCTL_NEG_ONE, }, { From 2cf4b2405e34433f947a7e47656079f956f5c3b5 Mon Sep 17 00:00:00 2001 From: Aaron Tomlin Date: Sat, 29 Aug 2026 10:53:21 -0400 Subject: [PATCH 0905/1012] hung_task: log summary line when warning budget is exhausted Once the warning budget is exhausted, hung_task_info() normally stops printing per-task details. When panic is triggered, full details are still printed so diagnostics remain available before panic. To retain visibility without restoring per-task output after budget exhaustion, emit a single aggregate summary line at the end of each watchdog scan that detects hung tasks with an exhausted budget. This keeps non-panic per-task reports budgeted during system-wide hangs. Link: https://lore.kernel.org/20260829145321.18423-3-atomlin@atomlin.com Signed-off-by: Aaron Tomlin Signed-off-by: Andrew Morton Suggested-by: Petr Mladek Suggested-by: Lance Yang Reviewed-by: Petr Mladek Reviewed-by: Lance Yang Reviewed-by: Bradley Morgan Cc: David Laight Cc: "Masami Hiramatsu (Google)" --- kernel/hung_task.c | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/kernel/hung_task.c b/kernel/hung_task.c index a5043188456d42..49d47ae475ab1c 100644 --- a/kernel/hung_task.c +++ b/kernel/hung_task.c @@ -277,7 +277,7 @@ static void hung_task_info(struct task_struct *t, unsigned long timeout, debug_show_blocker(t, timeout); if (!hung_task_warnings_budget) - pr_info("Future hung task reports are suppressed, see sysctl kernel.hung_task_warnings\n"); + pr_info("hung_task: further per-task details suppressed until warning budget is reset or panic is triggered (see sysctl kernel.hung_task_warnings)\n"); } touch_nmi_watchdog(); @@ -363,6 +363,10 @@ static void check_hung_uninterruptible_tasks(unsigned long timeout) return; } + if (!hung_task_warnings_budget && !hung_task_call_panic) + pr_info("hung_task: %lu hung tasks detected, warning budget exhausted\n", + this_round_count); + if (need_warning || hung_task_call_panic) { si_mask |= SYS_INFO_LOCKS; From 7796868d5243c48cc36829f122f50303eb9d4fe7 Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Tue, 18 Aug 2026 23:16:32 +0000 Subject: [PATCH 0906/1012] init, arch: make CONFIG_COMMAND_LINE_SIZE globally configurable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Currently, s390 has the ability to configure the maximum kernel command line size via Kconfig (CONFIG_COMMAND_LINE_SIZE). Other architectures define a hardcoded COMMAND_LINE_SIZE macro in their setup.h headers. In some use cases, such as netboot kernels, rootfs configurations, or larger initramfs setups, a larger command line size is required. While for embedded workloads, it can be reduced to save memory. Move CONFIG_COMMAND_LINE_SIZE out of arch/s390/Kconfig and into init/Kconfig under General setup, and update every architecture's setup.h header to define COMMAND_LINE_SIZE as CONFIG_COMMAND_LINE_SIZE. For user-space API (uapi) headers, wrap the definition in an `#ifdef __KERNEL__` guard and retain the historical hardcoded default in the `#else` block. When user-space headers are installed via `make headers_install`, unifdef strips out the kernel section, ensuring the same value as before for user-space applications including ``. For S390, the range is kept the same, but other architectures have varying constraints. S390 requires a minimum of 896 bytes to protect legacy bootloaders from overwriting the .text section. ARM, M68K, and NIOS2 allocate the command line directly on severely constrained decompressor stacks, so their ranges are strictly capped at 2048 bytes to prevent deterministic stack exhaustion and boot panics. PowerPC (PPC) boot wrappers silently truncate arguments past 2048 bytes, so it is also capped at 2048 to prevent silent parameter loss. The SuperH (SUPERH) boot parameter page allocates exactly PAGE_SIZE (typically 4096 bytes), and placing a 4096-byte command line starting at offset 256 would cause strscpy() to read out of bounds; it is capped at 3840 bytes. Alpha physically limits its boot parameter block to 256 bytes, so its limit is strictly locked to 256. All other architectures are capped at 4096 bytes to prevent unreasonable allocations. Link: https://lore.kernel.org/20260818231646.804507-2-wfelipe@google.com Signed-off-by: Maciej Żenczykowski Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Arnd Bergmann Cc: Christian Borntraeger Cc: Heiko Carstens Cc: Palmer Dabbelt Cc: Sven Schnelle Cc: Vasily Gorbik --- arch/alpha/include/uapi/asm/setup.h | 4 ++++ arch/arc/include/asm/setup.h | 2 +- arch/arm/include/uapi/asm/setup.h | 6 +++++- arch/arm64/include/uapi/asm/setup.h | 4 ++++ arch/loongarch/include/uapi/asm/setup.h | 4 ++++ arch/m68k/include/uapi/asm/setup.h | 6 +++++- arch/microblaze/include/uapi/asm/setup.h | 4 ++++ arch/mips/include/uapi/asm/setup.h | 4 ++++ arch/parisc/include/uapi/asm/setup.h | 4 ++++ arch/powerpc/include/uapi/asm/setup.h | 4 ++++ arch/riscv/include/uapi/asm/setup.h | 4 ++++ arch/s390/Kconfig | 8 -------- arch/sparc/include/uapi/asm/setup.h | 10 +++++++--- arch/um/include/asm/setup.h | 2 +- arch/x86/include/asm/setup.h | 2 +- arch/xtensa/include/uapi/asm/setup.h | 4 ++++ include/uapi/asm-generic/setup.h | 4 ++++ init/Kconfig | 16 ++++++++++++++++ 18 files changed, 76 insertions(+), 16 deletions(-) diff --git a/arch/alpha/include/uapi/asm/setup.h b/arch/alpha/include/uapi/asm/setup.h index f881ea5947cbc1..169f743ef7658d 100644 --- a/arch/alpha/include/uapi/asm/setup.h +++ b/arch/alpha/include/uapi/asm/setup.h @@ -2,6 +2,10 @@ #ifndef _UAPI__ALPHA_SETUP_H #define _UAPI__ALPHA_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 256 +#endif #endif /* _UAPI__ALPHA_SETUP_H */ diff --git a/arch/arc/include/asm/setup.h b/arch/arc/include/asm/setup.h index 1c6db599e1fcc9..60e158d58cec77 100644 --- a/arch/arc/include/asm/setup.h +++ b/arch/arc/include/asm/setup.h @@ -9,7 +9,7 @@ #include #include -#define COMMAND_LINE_SIZE 256 +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE /* * Data structure to map a ID to string diff --git a/arch/arm/include/uapi/asm/setup.h b/arch/arm/include/uapi/asm/setup.h index 8e50e034fec73a..4aa93558af1e7f 100644 --- a/arch/arm/include/uapi/asm/setup.h +++ b/arch/arm/include/uapi/asm/setup.h @@ -17,7 +17,11 @@ #include -#define COMMAND_LINE_SIZE 1024 +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else +#define COMMAND_LINE_SIZE 1024 +#endif /* The list ends with an ATAG_NONE node. */ #define ATAG_NONE 0x00000000 diff --git a/arch/arm64/include/uapi/asm/setup.h b/arch/arm64/include/uapi/asm/setup.h index 5d703888f35110..2236890175a5ab 100644 --- a/arch/arm64/include/uapi/asm/setup.h +++ b/arch/arm64/include/uapi/asm/setup.h @@ -22,6 +22,10 @@ #include +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 2048 +#endif #endif diff --git a/arch/loongarch/include/uapi/asm/setup.h b/arch/loongarch/include/uapi/asm/setup.h index d46363ce3e024c..03c7bfa1e5d9f2 100644 --- a/arch/loongarch/include/uapi/asm/setup.h +++ b/arch/loongarch/include/uapi/asm/setup.h @@ -3,6 +3,10 @@ #ifndef _UAPI_ASM_LOONGARCH_SETUP_H #define _UAPI_ASM_LOONGARCH_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 4096 +#endif #endif /* _UAPI_ASM_LOONGARCH_SETUP_H */ diff --git a/arch/m68k/include/uapi/asm/setup.h b/arch/m68k/include/uapi/asm/setup.h index 25fe26d5597cc6..2d5b24a5345f92 100644 --- a/arch/m68k/include/uapi/asm/setup.h +++ b/arch/m68k/include/uapi/asm/setup.h @@ -12,6 +12,10 @@ #ifndef _UAPI_M68K_SETUP_H #define _UAPI_M68K_SETUP_H -#define COMMAND_LINE_SIZE 256 +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else +#define COMMAND_LINE_SIZE 256 +#endif #endif /* _UAPI_M68K_SETUP_H */ diff --git a/arch/microblaze/include/uapi/asm/setup.h b/arch/microblaze/include/uapi/asm/setup.h index 16c56807f86a2d..e4b253064c7d99 100644 --- a/arch/microblaze/include/uapi/asm/setup.h +++ b/arch/microblaze/include/uapi/asm/setup.h @@ -12,6 +12,10 @@ #ifndef _UAPI_ASM_MICROBLAZE_SETUP_H #define _UAPI_ASM_MICROBLAZE_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 256 +#endif #endif /* _UAPI_ASM_MICROBLAZE_SETUP_H */ diff --git a/arch/mips/include/uapi/asm/setup.h b/arch/mips/include/uapi/asm/setup.h index 7d48c433b0c27d..8d6c474835aece 100644 --- a/arch/mips/include/uapi/asm/setup.h +++ b/arch/mips/include/uapi/asm/setup.h @@ -2,7 +2,11 @@ #ifndef _UAPI_MIPS_SETUP_H #define _UAPI_MIPS_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 4096 +#endif #endif /* _UAPI_MIPS_SETUP_H */ diff --git a/arch/parisc/include/uapi/asm/setup.h b/arch/parisc/include/uapi/asm/setup.h index 78b2f4ec7d6522..cfa77e84205dc6 100644 --- a/arch/parisc/include/uapi/asm/setup.h +++ b/arch/parisc/include/uapi/asm/setup.h @@ -2,6 +2,10 @@ #ifndef _PARISC_SETUP_H #define _PARISC_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 1024 +#endif #endif /* _PARISC_SETUP_H */ diff --git a/arch/powerpc/include/uapi/asm/setup.h b/arch/powerpc/include/uapi/asm/setup.h index c54940b09d065c..daa15ac4e94c6e 100644 --- a/arch/powerpc/include/uapi/asm/setup.h +++ b/arch/powerpc/include/uapi/asm/setup.h @@ -2,6 +2,10 @@ #ifndef _UAPI_ASM_POWERPC_SETUP_H #define _UAPI_ASM_POWERPC_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 2048 +#endif #endif /* _UAPI_ASM_POWERPC_SETUP_H */ diff --git a/arch/riscv/include/uapi/asm/setup.h b/arch/riscv/include/uapi/asm/setup.h index eb4f0209c6960e..a6c1f4b0987e35 100644 --- a/arch/riscv/include/uapi/asm/setup.h +++ b/arch/riscv/include/uapi/asm/setup.h @@ -3,6 +3,10 @@ #ifndef _UAPI_ASM_RISCV_SETUP_H #define _UAPI_ASM_RISCV_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 2048 +#endif #endif /* _UAPI_ASM_RISCV_SETUP_H */ diff --git a/arch/s390/Kconfig b/arch/s390/Kconfig index 4b51bc6e8948d7..026ba041ca9509 100644 --- a/arch/s390/Kconfig +++ b/arch/s390/Kconfig @@ -521,14 +521,6 @@ endchoice config 64BIT def_bool y -config COMMAND_LINE_SIZE - int "Maximum size of kernel command line" - default 4096 - range 896 1048576 - help - This allows you to specify the maximum length of the kernel command - line. - config SMP def_bool y diff --git a/arch/sparc/include/uapi/asm/setup.h b/arch/sparc/include/uapi/asm/setup.h index 3c208a4dd46405..7054f5249a3a68 100644 --- a/arch/sparc/include/uapi/asm/setup.h +++ b/arch/sparc/include/uapi/asm/setup.h @@ -6,10 +6,14 @@ #ifndef _UAPI_SPARC_SETUP_H #define _UAPI_SPARC_SETUP_H -#if defined(__sparc__) && defined(__arch64__) -# define COMMAND_LINE_SIZE 2048 +#ifdef __KERNEL__ +# define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE #else -# define COMMAND_LINE_SIZE 256 +# if defined(__sparc__) && defined(__arch64__) +# define COMMAND_LINE_SIZE 2048 +# else +# define COMMAND_LINE_SIZE 256 +# endif #endif diff --git a/arch/um/include/asm/setup.h b/arch/um/include/asm/setup.h index 80ada899f25426..bc83dc4d467d30 100644 --- a/arch/um/include/asm/setup.h +++ b/arch/um/include/asm/setup.h @@ -6,6 +6,6 @@ * command line, so this choice is ok. */ -#define COMMAND_LINE_SIZE 4096 +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE #endif /* SETUP_H_INCLUDED */ diff --git a/arch/x86/include/asm/setup.h b/arch/x86/include/asm/setup.h index 895d09faaf832e..1b333bb091d7f2 100644 --- a/arch/x86/include/asm/setup.h +++ b/arch/x86/include/asm/setup.h @@ -4,7 +4,7 @@ #include -#define COMMAND_LINE_SIZE 2048 +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE #include #include diff --git a/arch/xtensa/include/uapi/asm/setup.h b/arch/xtensa/include/uapi/asm/setup.h index 5356a5fd4d1737..dcf33a403527a3 100644 --- a/arch/xtensa/include/uapi/asm/setup.h +++ b/arch/xtensa/include/uapi/asm/setup.h @@ -12,6 +12,10 @@ #ifndef _XTENSA_SETUP_H #define _XTENSA_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 256 +#endif #endif diff --git a/include/uapi/asm-generic/setup.h b/include/uapi/asm-generic/setup.h index 88ac5100df3598..b8d06e6d56bd7e 100644 --- a/include/uapi/asm-generic/setup.h +++ b/include/uapi/asm-generic/setup.h @@ -2,6 +2,10 @@ #ifndef __ASM_GENERIC_SETUP_H #define __ASM_GENERIC_SETUP_H +#ifdef __KERNEL__ +#define COMMAND_LINE_SIZE CONFIG_COMMAND_LINE_SIZE +#else #define COMMAND_LINE_SIZE 512 +#endif #endif /* __ASM_GENERIC_SETUP_H */ diff --git a/init/Kconfig b/init/Kconfig index 8583d9f06c522e..edc79893e1ac4f 100644 --- a/init/Kconfig +++ b/init/Kconfig @@ -1614,6 +1614,22 @@ config CMDLINE_FROM_BOOTCONFIG If unsure, say N. +config COMMAND_LINE_SIZE + int "Maximum size of kernel command line" + default 4096 if S390 || LOONGARCH || MIPS || UML + default 2048 if X86 || ARM64 || PPC || SPARC64 || RISCV + default 1024 if ARM || PARISC + default 256 if ALPHA || ARC || M68K || MICROBLAZE || SPARC32 || XTENSA + default 512 + range 896 1048576 if S390 + range 256 2048 if ARM || M68K || NIOS2 || PPC + range 256 3840 if SUPERH + range 256 256 if ALPHA + range 256 4096 + help + This allows you to specify the maximum length of the kernel command + line. + config CMDLINE_LOG_WRAP_IDEAL_LEN int "Length to try to wrap the cmdline when logged at boot" default 1021 From 51cfdbb934df10b8f88f83d1106bb532d32c37c7 Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Tue, 18 Aug 2026 23:16:33 +0000 Subject: [PATCH 0907/1012] init/Kconfig: make config INIT_ENV_ARG_LIMIT user-configurable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Patch series "init, arch: make command line size and init arg limit configurable", v2. This patch series promotes COMMAND_LINE_SIZE from arch/s390/Kconfig to init/Kconfig to be generally available to other architectures. In some use cases, such as netboot kernels, rootfs configurations, larger initramfs setups, require larger sizes. While for embedded workloads, it can be reduced to save memory. Since COMMAND_LINE_SIZE can be larger, it also makes sense to allow INIT_ENV_ARG_LIMIT to be configured. This patch (of 2): INIT_ENV_ARG_LIMIT is defined without a prompt string (`int`), making it a hidden Kconfig symbol that defaults to 32 (or 128 for UML) and cannot be configured in `make menuconfig`. Now that CONFIG_COMMAND_LINE_SIZE is configurable across all architectures, users who select larger kernel command lines (e.g., 4096 bytes) may pass more than 32 command-line arguments or environment variables (`foo=bar`) to `/sbin/init`. If INIT_ENV_ARG_LIMIT remains hardcoded at 32, any argument after the 32nd sets the panic_later flag and causes a hard kernel panic on boot. Add a prompt string ("Maximum number of kernel command line arguments") and a `range 32 4096` to `config INIT_ENV_ARG_LIMIT` so that users can configure their init argument and environment variable limit when needed, while preserving the existing default of 32 for standard builds. Link: https://lore.kernel.org/20260818231646.804507-1-wfelipe@google.com Link: https://lore.kernel.org/20260818231646.804507-3-wfelipe@google.com Signed-off-by: Maciej Żenczykowski Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton Cc: Albert Ou Cc: Alexander Gordeev Cc: Alexandre Ghiti Cc: Arnd Bergmann Cc: Christian Borntraeger Cc: Heiko Carstens Cc: Palmer Dabbelt Cc: Sven Schnelle Cc: Vasily Gorbik --- init/Kconfig | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/init/Kconfig b/init/Kconfig index edc79893e1ac4f..3b141ba427d5e0 100644 --- a/init/Kconfig +++ b/init/Kconfig @@ -231,9 +231,10 @@ config BROKEN_ON_SMP default y config INIT_ENV_ARG_LIMIT - int + int "Maximum number of kernel command line arguments" default 32 if !UML default 128 if UML + range 32 4096 help Maximum of each of the number of arguments and environment variables passed to init from the kernel command line. From 26c03d2c9dde1ab8532f5855efe4d4005e60d3c5 Mon Sep 17 00:00:00 2001 From: Zhan Xusheng Date: Mon, 17 Aug 2026 20:16:13 +0800 Subject: [PATCH 0908/1012] minmax.h: update the stale 'x' versus 'ux' comment Commit b280bb27a9f7 ("minmax.h: reduce the #define expansion of min(), max() and clamp()") made __sign_use(), __is_nonneg() and __types_ok() take only 'ux', and commit a5743f32baec ("minmax.h: use BUILD_BUG_ON_MSG() for the lo < hi test in clamp()") did the same for the clamp() limit test. The comment describing the old split was added one patch earlier and was never updated. 'ux' now carries the value check too, since __is_nonneg() tests it rather than the original expression, and nothing here looks at the value of 'x' any more: it is expanded only to initialise 'ux' and in the error message, as the first of those changes intended. Link: https://lore.kernel.org/20260817121613.3846511-1-zhanxusheng@xiaomi.com Signed-off-by: Zhan Xusheng Signed-off-by: Andrew Morton Cc: David Laight Cc: "H. Peter Anvin" --- include/linux/minmax.h | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/include/linux/minmax.h b/include/linux/minmax.h index a0158db54a0411..5ef4d58c0c4291 100644 --- a/include/linux/minmax.h +++ b/include/linux/minmax.h @@ -38,9 +38,9 @@ * Note that 'x' is the original expression, and 'ux' is the unique variable * that contains the value. * - * We use 'ux' for pure type checking, and 'x' for when we need to look at the - * value (but without evaluating it for side effects! - * Careful to only ever evaluate it with sizeof() or __builtin_constant_p() etc). + * We use 'ux' for both the type and the value checks, so 'x' itself is only + * expanded twice: once to initialise 'ux', and once quoted in the error + * message. * * Pointers end up being checked by the normal C type rules at the actual * comparison, and these expressions only need to be careful to not cause From 3658e85b634204fa367da76cb755ad52a892db17 Mon Sep 17 00:00:00 2001 From: Julian Braha Date: Mon, 17 Aug 2026 22:03:11 +0100 Subject: [PATCH 0909/1012] lib: cleanup "fake" tristates in Kconfig These 7 DECOMPRESS_ options (e.g. DECOMPRESS_GZIP) currently have the tristate type, but can never be set to M. Their only valid values are Y and N, making them effectively booleans. Let's make their types more accurate by changing them to 'bool'. Note that this is only a code cleanup, there is no functional change. These bistates were found by kconfirm, a static analysis tool for Kconfig. Link: https://lore.kernel.org/20260817210311.2142999-1-julianbraha@gmail.com Signed-off-by: Julian Braha Signed-off-by: Andrew Morton Cc: Arnd Bergmann Cc: Jani Nikula Cc: Julia Lawall --- lib/Kconfig | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/lib/Kconfig b/lib/Kconfig index 4e6b34c3346d56..1e42f5167c68bc 100644 --- a/lib/Kconfig +++ b/lib/Kconfig @@ -219,29 +219,29 @@ source "lib/xz/Kconfig" # config DECOMPRESS_GZIP select ZLIB_INFLATE - tristate + bool config DECOMPRESS_BZIP2 - tristate + bool config DECOMPRESS_LZMA - tristate + bool config DECOMPRESS_XZ select XZ_DEC - tristate + bool config DECOMPRESS_LZO select LZO_DECOMPRESS - tristate + bool config DECOMPRESS_LZ4 select LZ4_DECOMPRESS - tristate + bool config DECOMPRESS_ZSTD select ZSTD_DECOMPRESS - tristate + bool # # Generic allocator support is selected if needed From bcadb7bda59e53847f91aad9c13045044f5c83c0 Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Tue, 18 Aug 2026 04:53:46 +0000 Subject: [PATCH 0910/1012] init/main: fix off-by-one in argv_init cleanup Patch series "init: fix array boundary bugs in boot parameter parsing". This series fixes two distinct boundary logic edge-case bugs in `init/main.c` related to parsing boot command-line arguments and environment variables. Both bugs have been present since the early git history (Linux-2.6.12-rc2). 1. The first patch fixes an off-by-one error in `init_setup()` where the final slot of the `argv_init` array was left uncleared. This allowed a stale kernel parameter to leak into the `init` process's user-space command line if exactly `MAX_INIT_ARGS` unknown parameters were passed. 2. The second patch fixes a false-positive kernel panic in `unknown_bootoption()`. If a user filled the environment variable array up to its exact limit (32) and then attempted to overwrite the final variable, the kernel would panic before evaluating whether it was a harmless duplicate. Exact QEMU reproduction steps for both edge cases are documented inside their respective commit descriptions. This patch (of 2): When cleaning up argv_init in init_setup() and rdinit_setup(), the loop terminates one element early due to using '<' instead of '<='. Since argv_init is sized MAX_INIT_ARGS+2, index MAX_INIT_ARGS is a valid element that should be cleared to NULL. If exactly MAX_INIT_ARGS unknown arguments are passed before 'init=', the uncleared argv_init[MAX_INIT_ARGS] can act as a ghost argument to /sbin/init or cause a spurious kernel panic when later appended to. To verify the argument leak, boot a VM into a shell with 32 unknown kernel arguments, the init parameter, and 31 user arguments: STALE_ARGS=$(for i in {1..32}; do echo -n "stale$i "; done) USER_ARGS=$(for i in {1..31}; do echo -n "user$i "; done) qemu-system-x86_64 -kernel bzImage \ -append "$STALE_ARGS init=/bin/sh $USER_ARGS" Running `cat /proc/1/cmdline` inside the shell reveals that the 32nd kernel argument ('stale32') incorrectly leaked into the init process's command line. This patch zeroes the final slot, cleanly terminating the array. Link: https://lore.kernel.org/20260818045357.4123784-1-wfelipe@google.com Link: https://lore.kernel.org/20260818045357.4123784-2-wfelipe@google.com Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Fixes: ffdfc40976dd ("[PATCH] Add rdinit parameter to pick early userspace init") Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton --- init/main.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/init/main.c b/init/main.c index 31f2bf54976ab9..517b76447950f9 100644 --- a/init/main.c +++ b/init/main.c @@ -581,7 +581,7 @@ static int __init init_setup(char *str) * the shell think it should execute a script with such name. * So we ignore all arguments entered _before_ init=... [MJ] */ - for (i = 1; i < MAX_INIT_ARGS; i++) + for (i = 1; i <= MAX_INIT_ARGS; i++) argv_init[i] = NULL; return 1; } @@ -594,7 +594,7 @@ static int __init rdinit_setup(char *str) ramdisk_execute_command = str; ramdisk_execute_command_set = true; /* See "auto" comment in init_setup */ - for (i = 1; i < MAX_INIT_ARGS; i++) + for (i = 1; i <= MAX_INIT_ARGS; i++) argv_init[i] = NULL; return 1; } From 6ee97d8afae43144ed456da7216e776ff662224b Mon Sep 17 00:00:00 2001 From: Wilson Felipe Pereira Date: Tue, 18 Aug 2026 04:53:47 +0000 Subject: [PATCH 0911/1012] init/main: fix false-positive kernel panic on environment variable overwrite In unknown_bootoption(), the limit checking for environment variables sets panic_later *before* checking if the variable already exists in envp_init. If a user passes exactly MAX_INIT_ENVS custom variables and then overwrites the final variable by matching its key, it causes a false-positive hard panic on boot despite not actually exceeding the array bounds or increasing the total variable count. Swapping the order of these checks allows the duplicate check to break out of the loop before the panic flag is erroneously latched. To verify, boot a VM with 31 custom variables (filling the array up to its limit of 32) and then overwrite the very last variable: ENV_VARS=$(for i in {1..31}; do echo -n "var$i=$i "; done) qemu-system-x86_64 -kernel bzImage -append "$ENV_VARS var31=overwrite" Without this patch, the kernel crashes instantly with: Kernel panic - not syncing: Too many boot env vars at 'var31=overwrite' With this patch, the kernel safely overwrites the variable and boots. Link: https://lore.kernel.org/20260818045357.4123784-3-wfelipe@google.com Fixes: 1da177e4c3f4 ("Linux-2.6.12-rc2") Signed-off-by: Wilson Felipe Pereira Signed-off-by: Andrew Morton --- init/main.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/init/main.c b/init/main.c index 517b76447950f9..d1cd8efb823860 100644 --- a/init/main.c +++ b/init/main.c @@ -548,12 +548,12 @@ static int __init unknown_bootoption(char *param, char *val, /* Environment option */ unsigned int i; for (i = 0; envp_init[i]; i++) { + if (!strncmp(param, envp_init[i], len+1)) + break; if (i == MAX_INIT_ENVS) { panic_later = "env"; panic_param = param; } - if (!strncmp(param, envp_init[i], len+1)) - break; } envp_init[i] = param; } else { From de7869e2dfdcd257553b2d8996e7f1678bf2500b Mon Sep 17 00:00:00 2001 From: Andrei Vagin Date: Sun, 16 Aug 2026 16:12:15 +0000 Subject: [PATCH 0912/1012] proc: report SIGEV_NONE in /proc/pid/timers if target task has died When a posix timer is created targeting a specific thread (using SIGEV_SIGNAL | SIGEV_THREAD_ID), it takes a reference to the target struct pid in timer->it_pid. If the target thread subsequently terminates, its numeric tid is freed and can be recycled for an unrelated task. However, the timer holds its reference to the original struct pid. show_timer() in /proc/[pid]/timers previously called pid_nr_ns() directly on timer->it_pid without checking whether any task remained attached to that struct pid. As a result: 1. It reported the stale tid, which could mistakenly refer to a recycled pid. 2. In the kernel, expired signals for dead target threads are dropped by posixtimer_send_sigqueue() because posixtimer_get_target() returns NULL, so the timer functionally acts as SIGEV_NONE. 3. Checkpoint/restore tools (CRIU) parsing /proc/[pid]/timers would try to restore a timer with SIGEV_SIGNAL | SIGEV_THREAD_ID targeting a non-existent or unrelated thread. Check pid_has_task(timer->it_pid, timer->it_pid_type) in show_timer(). If the target task has died, override notify to SIGEV_NONE and report PID 0 (e.g., 'notify: none/pid.0'). Link: https://lore.kernel.org/20260816161216.984580-1-avagin@google.com Fixes: 57b8015e07a7 ("posix-timers: Show sigevent info in proc file") Signed-off-by: Andrei Vagin Signed-off-by: Andrew Morton Reviewed-by: Pavel Tikhomirov Cc: Thomas Gleixner --- fs/proc/base.c | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/fs/proc/base.c b/fs/proc/base.c index 6a39de424f62a1..1455de58e53cf0 100644 --- a/fs/proc/base.c +++ b/fs/proc/base.c @@ -2519,17 +2519,23 @@ static int show_timer(struct seq_file *m, void *v) struct k_itimer *timer = hlist_entry((struct hlist_node *)v, struct k_itimer, list); struct timers_private *tp = m->private; int notify = timer->it_sigev_notify; + pid_t nr = 0; guard(spinlock_irq)(&timer->it_lock); if (!posixtimer_valid(timer)) return 0; + if (timer->it_pid && pid_has_task(timer->it_pid, timer->it_pid_type)) + nr = pid_nr_ns(timer->it_pid, tp->ns); + else + notify = SIGEV_NONE; + seq_printf(m, "ID: %d\n", timer->it_id); seq_printf(m, "signal: %d/%px\n", timer->sigq.info.si_signo, timer->sigq.info.si_value.sival_ptr); seq_printf(m, "notify: %s/%s.%d\n", nstr[notify & ~SIGEV_THREAD_ID], (notify & SIGEV_THREAD_ID) ? "tid" : "pid", - pid_nr_ns(timer->it_pid, tp->ns)); + nr); seq_printf(m, "ClockID: %d\n", timer->it_clock); return 0; From 6d1e066e81ca4bf3a385866a5120ba21f48135c6 Mon Sep 17 00:00:00 2001 From: Konstantin Khorenko Date: Fri, 14 Aug 2026 18:57:09 +0200 Subject: [PATCH 0913/1012] selftests/core: fix unshare_test with large fs.nr_open The test assumes fs.nr_open is close to the default 1048576, but some systems set it much higher (e.g. 1073741816). This is systemd's doing: since systemd v240 (2018), PID 1 bumps fs.nr_open and fs.file-max to their largest possible values on boot, as file descriptors are already accounted for by memcg [1]. In that case, dup2() to nr_open + 64 requires the kernel to allocate a file descriptor table with ~1 billion entries, which fails with ENOMEM. On a kernel that already carries 04a2c4b4511d1, dup2() no longer fails with ENOMEM. The allocation is now rejected up front and the caller gets EMFILE instead, without the WARNING, but the test still fails. Cap the nr_open value used for the test's own arithmetic to a known reasonable base value (1048576) and restore the true original value once the test has completed. Link: https://lore.kernel.org/20260814165709.513263-1-khorenko@virtuozzo.com Link: https://github.com/systemd/systemd/commit/a8b627aaed409a15260c25988970c795bf963812 [1] Signed-off-by: Konstantin Khorenko Signed-off-by: Eva Kurchatova Signed-off-by: Andrew Morton Cc: Shuah Khan Cc: Wei Yang Cc: --- tools/testing/selftests/core/unshare_test.c | 19 +++++++++++++++++-- 1 file changed, 17 insertions(+), 2 deletions(-) diff --git a/tools/testing/selftests/core/unshare_test.c b/tools/testing/selftests/core/unshare_test.c index ffce75a6c228f1..d40e963dd5205e 100644 --- a/tools/testing/selftests/core/unshare_test.c +++ b/tools/testing/selftests/core/unshare_test.c @@ -40,6 +40,14 @@ TEST(unshare_EMFILE) ASSERT_EQ(sscanf(buf, "%d", &nr_open), 1); + /* + * Cap nr_open for the duration of the test to avoid ENOMEM from a + * huge fd table allocation; buf/n keep the real original value so + * fs.nr_open can be restored to it once the test is done. + */ + if (nr_open > 1024 * 1024) + nr_open = 1024 * 1024; + ASSERT_EQ(0, getrlimit(RLIMIT_NOFILE, &rlimit)); /* bump fs.nr_open */ @@ -73,10 +81,13 @@ TEST(unshare_EMFILE) if (pid == 0) { int err; + char buf3[32]; + ssize_t n3; - /* restore fs.nr_open */ + /* restore fs.nr_open to the (possibly capped) test baseline */ + n3 = sprintf(buf3, "%d\n", nr_open); lseek(fd, 0, SEEK_SET); - write(fd, buf, n); + write(fd, buf3, n3); /* ... and now unshare(CLONE_FILES) must fail with EMFILE */ err = unshare(CLONE_FILES); EXPECT_EQ(err, -1) @@ -89,6 +100,10 @@ TEST(unshare_EMFILE) EXPECT_EQ(waitpid(pid, &status, 0), pid); EXPECT_EQ(true, WIFEXITED(status)); EXPECT_EQ(0, WEXITSTATUS(status)); + + /* restore the real fs.nr_open value */ + lseek(fd, 0, SEEK_SET); + write(fd, buf, n); } TEST_HARNESS_MAIN From 92d33ae375870cbadecd46865852112cc39dd847 Mon Sep 17 00:00:00 2001 From: Thomas Maarseveen Date: Wed, 12 Aug 2026 20:55:33 +0200 Subject: [PATCH 0914/1012] lib/tests: add KUnit tests for errseq The errseq_t infrastructure (lib/errseq.c) underpins writeback error reporting but has no regression tests. Its semantics are subtle enough to have needed fixing before: commit b4678df184b3 ("errseq: Always report a writeback error once") changed how unseen errors reach new samplers. Add a KUnit suite covering the documented single-threaded semantics: - a zeroed errseq_t is the "no error yet" epoch - errors are recorded, overwrite one another, and both ends of the valid errno range round-trip exactly - an error nobody has seen samples as zero, so a check against a fresh sample still reports it - errseq_check_and_advance() reports a given error exactly once per cursor and leaves the cursor in place when nothing has changed - once an error has been seen, a fresh sample is current and a check against it reports nothing - the same error recorded again after being seen is reported again, even to a cursor that consumed the first occurrence while another cursor marked the repeat as seen - independent cursors each observe each error The lockless behaviour of errseq_t under concurrent updates and the WARN path for invalid error values are deliberately out of scope. Tested with ./tools/testing/kunit/kunit.py run, with a kunitconfig enabling CONFIG_KUNIT=y and CONFIG_ERRSEQ_KUNIT_TEST=y; all 13 tests pass under ARCH=um. Link: https://lore.kernel.org/20260812-errseq-kunit-v1-1-312be4c3aa0d@gmail.com Signed-off-by: Thomas Maarseveen Signed-off-by: Andrew Morton Acked-by: Jeff Layton Cc: David Gow --- MAINTAINERS | 1 + lib/Kconfig.debug | 15 +++ lib/tests/Makefile | 1 + lib/tests/errseq_kunit.c | 237 +++++++++++++++++++++++++++++++++++++++ 4 files changed, 254 insertions(+) create mode 100644 lib/tests/errseq_kunit.c diff --git a/MAINTAINERS b/MAINTAINERS index 360977678f707e..67c42e55d60afd 100644 --- a/MAINTAINERS +++ b/MAINTAINERS @@ -9700,6 +9700,7 @@ M: Jeff Layton S: Maintained F: include/linux/errseq.h F: lib/errseq.c +F: lib/tests/errseq_kunit.c ESD CAN NETWORK DRIVERS M: Stefan Mätje diff --git a/lib/Kconfig.debug b/lib/Kconfig.debug index 134b15a44625e8..c3f448f3b8f13c 100644 --- a/lib/Kconfig.debug +++ b/lib/Kconfig.debug @@ -2797,6 +2797,21 @@ config SYSCTL_KUNIT_TEST If unsure, say N. +config ERRSEQ_KUNIT_TEST + tristate "KUnit test for errseq" if !KUNIT_ALL_TESTS + depends on KUNIT + default KUNIT_ALL_TESTS + help + This builds the errseq KUnit test suite. + It tests the documented semantics of the errseq_t error-tracking + infrastructure (lib/errseq.c), which underpins writeback error + reporting. + + For more information on KUnit and unit tests in general please refer + to the KUnit documentation in Documentation/dev-tools/kunit/. + + If unsure, say N. + config KFIFO_KUNIT_TEST tristate "KUnit Test for the generic kernel FIFO implementation" if !KUNIT_ALL_TESTS depends on KUNIT diff --git a/lib/tests/Makefile b/lib/tests/Makefile index 3cac3b63a7522c..8e11b125433bf1 100644 --- a/lib/tests/Makefile +++ b/lib/tests/Makefile @@ -13,6 +13,7 @@ obj-$(CONFIG_BLACKHOLE_DEV_KUNIT_TEST) += blackhole_dev_kunit.o obj-$(CONFIG_CHECKSUM_KUNIT) += checksum_kunit.o obj-$(CONFIG_CMDLINE_KUNIT_TEST) += cmdline_kunit.o obj-$(CONFIG_CPUMASK_KUNIT_TEST) += cpumask_kunit.o +obj-$(CONFIG_ERRSEQ_KUNIT_TEST) += errseq_kunit.o obj-$(CONFIG_FFS_KUNIT_TEST) += ffs_kunit.o CFLAGS_fortify_kunit.o += $(call cc-disable-warning, unsequenced) CFLAGS_fortify_kunit.o += $(call cc-disable-warning, stringop-overread) diff --git a/lib/tests/errseq_kunit.c b/lib/tests/errseq_kunit.c new file mode 100644 index 00000000000000..8f39ebc4a2488e --- /dev/null +++ b/lib/tests/errseq_kunit.c @@ -0,0 +1,237 @@ +// SPDX-License-Identifier: GPL-2.0 +/* + * KUnit tests for the errseq_t error-tracking infrastructure. + * + * These exercise the documented single-threaded semantics of the errseq + * API (see Documentation/core-api/errseq.rst and lib/errseq.c): error + * recording and overwriting, the "seen" handoff between errseq_sample() + * and errseq_check_and_advance(), and the re-reporting of an error that + * is recorded again after it has been seen. + * + * The lockless properties of errseq_t under concurrent updates are + * outside the scope of these deterministic tests, as is the WARN path + * for invalid error values. + */ +#include + +#include +#include +#include + +/* + * A zeroed errseq_t is the "no error has ever occurred" epoch: it + * samples as zero and no check against it reports anything. + */ +static void errseq_test_zero_epoch_reports_no_error(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + + KUNIT_EXPECT_EQ(test, errseq_sample(&eseq), 0); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, 0), 0); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), 0); + KUNIT_EXPECT_EQ(test, since, 0); +} + +static void errseq_test_set_records_error(struct kunit *test) +{ + errseq_t eseq = 0; + + /* errseq_set() returns the previous value; the epoch is zero. */ + KUNIT_EXPECT_EQ(test, errseq_set(&eseq, -EIO), 0); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, 0), -EIO); +} + +/* Any error set always overwrites an existing error. */ +static void errseq_test_set_overwrites_error(struct kunit *test) +{ + errseq_t eseq = 0; + + errseq_set(&eseq, -EIO); + errseq_set(&eseq, -ENOSPC); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, 0), -ENOSPC); +} + +/* Both ends of the valid error range are recorded exactly. */ +static void errseq_test_errno_range_extremes(struct kunit *test) +{ + errseq_t lo = 0; + errseq_t hi = 0; + + errseq_set(&lo, -1); + KUNIT_EXPECT_EQ(test, errseq_check(&lo, 0), -1); + + errseq_set(&hi, -MAX_ERRNO); + KUNIT_EXPECT_EQ(test, errseq_check(&hi, 0), -MAX_ERRNO); +} + +/* + * An error nobody has seen yet samples as zero, so that a check against + * the sample still reports it (see commit b4678df184b3 ("errseq: Always + * report a writeback error once")). + */ +static void errseq_test_sample_of_unseen_error_is_zero(struct kunit *test) +{ + errseq_t eseq = 0; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_sample(&eseq), 0); +} + +static void errseq_test_new_sampler_sees_unseen_error(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since; + + errseq_set(&eseq, -EIO); + since = errseq_sample(&eseq); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, since), -EIO); +} + +/* A given error is reported exactly once per advancing cursor. */ +static void errseq_test_check_and_advance_reports_once(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = errseq_sample(&eseq); + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), 0); +} + +/* + * Once an error has been seen, a fresh sample is non-zero and checking + * against it reports nothing: handled errors do not reach new samplers. + */ +static void errseq_test_sample_after_seen_is_current(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + errseq_t sample; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + + sample = errseq_sample(&eseq); + KUNIT_EXPECT_NE(test, sample, 0); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, sample), 0); +} + +static void errseq_test_new_error_after_advance(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + + errseq_set(&eseq, -ENOSPC); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -ENOSPC); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), 0); +} + +/* + * Recording the same error again after it has been seen must bump the + * sequence, so cursors that consumed the first occurrence see the + * second one too. + */ +static void errseq_test_same_error_reported_again_after_seen(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + errseq_t seen_cursor; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + + seen_cursor = since; + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, since), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + /* The repeat must advance the sequence, not just re-toggle "seen". */ + KUNIT_EXPECT_NE(test, since, seen_cursor); +} + +/* + * A cursor that consumed an error must still observe a repeat of that + * error even when another cursor has already marked the repeat seen: + * recording over a seen value must advance the sequence. + */ +static void errseq_test_repeat_error_visible_to_all_cursors(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t cursor_a = 0; + errseq_t cursor_b = 0; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), -EIO); + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_b), -EIO); + + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, cursor_a), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), 0); +} + +/* An advance with no new error reports nothing and leaves the cursor put. */ +static void errseq_test_advance_stable_when_unchanged(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t since = 0; + errseq_t cursor; + + errseq_set(&eseq, -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), -EIO); + + cursor = since; + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &since), 0); + KUNIT_EXPECT_EQ(test, since, cursor); +} + +/* + * Cursors are independent: one subscriber consuming an error does not + * consume it for another, and each subscriber sees each error once. + */ +static void errseq_test_two_subscribers_independent(struct kunit *test) +{ + errseq_t eseq = 0; + errseq_t cursor_a = errseq_sample(&eseq); + errseq_t cursor_b = errseq_sample(&eseq); + + errseq_set(&eseq, -EIO); + + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check(&eseq, cursor_b), -EIO); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_b), -EIO); + + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_a), 0); + KUNIT_EXPECT_EQ(test, errseq_check_and_advance(&eseq, &cursor_b), 0); +} + +static struct kunit_case errseq_test_cases[] = { + KUNIT_CASE(errseq_test_zero_epoch_reports_no_error), + KUNIT_CASE(errseq_test_set_records_error), + KUNIT_CASE(errseq_test_set_overwrites_error), + KUNIT_CASE(errseq_test_errno_range_extremes), + KUNIT_CASE(errseq_test_sample_of_unseen_error_is_zero), + KUNIT_CASE(errseq_test_new_sampler_sees_unseen_error), + KUNIT_CASE(errseq_test_check_and_advance_reports_once), + KUNIT_CASE(errseq_test_sample_after_seen_is_current), + KUNIT_CASE(errseq_test_new_error_after_advance), + KUNIT_CASE(errseq_test_same_error_reported_again_after_seen), + KUNIT_CASE(errseq_test_repeat_error_visible_to_all_cursors), + KUNIT_CASE(errseq_test_advance_stable_when_unchanged), + KUNIT_CASE(errseq_test_two_subscribers_independent), + {} +}; + +static struct kunit_suite errseq_test_suite = { + .name = "errseq", + .test_cases = errseq_test_cases, +}; + +kunit_test_suite(errseq_test_suite); + +MODULE_DESCRIPTION("KUnit tests for the errseq infrastructure"); +MODULE_LICENSE("GPL"); From 2e6da33c93e1f16a367cda04a0fba99130dd4ad7 Mon Sep 17 00:00:00 2001 From: Eric Biggers Date: Mon, 31 Aug 2026 14:22:48 -0700 Subject: [PATCH 0915/1012] xor: add missing vzeroupper to AVX code Since the AVX optimized XOR code uses YMM registers, execute vzeroupper before returning from it. This is needed to avoid degrading the performance of any later SSE code that may happen to be executed. Link: https://lore.kernel.org/20260831212248.213805-1-ebiggers@kernel.org Fixes: ea4d26ae24e5 ("raid5: add AVX optimized RAID5 checksumming") Signed-off-by: Eric Biggers Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: --- lib/raid/xor/x86/xor-avx.c | 1 + 1 file changed, 1 insertion(+) diff --git a/lib/raid/xor/x86/xor-avx.c b/lib/raid/xor/x86/xor-avx.c index f7777d7aa269bd..95b21e7225e8d7 100644 --- a/lib/raid/xor/x86/xor-avx.c +++ b/lib/raid/xor/x86/xor-avx.c @@ -147,6 +147,7 @@ static void xor_gen_avx(void *dest, void **srcs, unsigned int src_cnt, { kernel_fpu_begin(); xor_gen_avx_inner(dest, srcs, src_cnt, bytes); + asm volatile("vzeroupper"); kernel_fpu_end(); } From 51d7302f21d805a74bd89f26997074fb16ec342e Mon Sep 17 00:00:00 2001 From: Eric Biggers Date: Mon, 31 Aug 2026 14:23:08 -0700 Subject: [PATCH 0916/1012] raid6: add missing vzeroupper to AVX2 code Since the AVX2 optimized RAID6 code uses YMM registers, execute vzeroupper before returning from it. This is needed to avoid degrading the performance of any later SSE code that may happen to be executed. Link: https://lore.kernel.org/20260831212308.213855-1-ebiggers@kernel.org Fixes: 2c935842bdb4 ("lib/raid6: Add AVX2 optimized gen_syndrome functions") Fixes: 7056741fd9fc ("lib/raid6: Add AVX2 optimized recovery functions") Signed-off-by: Eric Biggers Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: --- lib/raid/raid6/x86/avx2.c | 6 ++++++ lib/raid/raid6/x86/recov_avx2.c | 2 ++ 2 files changed, 8 insertions(+) diff --git a/lib/raid/raid6/x86/avx2.c b/lib/raid/raid6/x86/avx2.c index 7d829c669ea795..3cc2fe7ac42c57 100644 --- a/lib/raid/raid6/x86/avx2.c +++ b/lib/raid/raid6/x86/avx2.c @@ -67,6 +67,7 @@ static void raid6_avx21_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -115,6 +116,7 @@ static void raid6_avx21_xor_syndrome(int disks, int start, int stop, } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -175,6 +177,7 @@ static void raid6_avx22_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -243,6 +246,7 @@ static void raid6_avx22_xor_syndrome(int disks, int start, int stop, } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -334,6 +338,7 @@ static void raid6_avx24_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -444,6 +449,7 @@ static void raid6_avx24_xor_syndrome(int disks, int start, int stop, asm volatile("vmovntdq %%ymm14,%0" : "=m" (q[d+96])); } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } diff --git a/lib/raid/raid6/x86/recov_avx2.c b/lib/raid/raid6/x86/recov_avx2.c index a714a780a2d8f6..820871046e3058 100644 --- a/lib/raid/raid6/x86/recov_avx2.c +++ b/lib/raid/raid6/x86/recov_avx2.c @@ -176,6 +176,7 @@ static void raid6_2data_recov_avx2(int disks, size_t bytes, int faila, #endif } + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -293,6 +294,7 @@ static void raid6_datap_recov_avx2(int disks, size_t bytes, int faila, #endif } + asm volatile("vzeroupper"); kernel_fpu_end(); } From 5436671ff5e89e4d9b69184849f7ff7709717409 Mon Sep 17 00:00:00 2001 From: Eric Biggers Date: Mon, 31 Aug 2026 14:23:16 -0700 Subject: [PATCH 0917/1012] raid6: add missing vzeroupper to AVX-512 code Since the AVX-512 optimized RAID6 code uses ZMM registers, execute vzeroupper before returning from it. This is needed to avoid degrading the performance of any later SSE code that may happen to be executed. Link: https://lore.kernel.org/20260831212316.213896-1-ebiggers@kernel.org Fixes: e0a491c12968 ("lib/raid6: Add AVX512 optimized gen_syndrome functions") Fixes: 13c520b2993c ("lib/raid6: Add AVX512 optimized recovery functions") Signed-off-by: Eric Biggers Signed-off-by: Andrew Morton Cc: Christoph Hellwig Cc: --- lib/raid/raid6/x86/avx512.c | 6 ++++++ lib/raid/raid6/x86/recov_avx512.c | 2 ++ 2 files changed, 8 insertions(+) diff --git a/lib/raid/raid6/x86/avx512.c b/lib/raid/raid6/x86/avx512.c index e671eb5bde63e4..772bfc4af6dfd7 100644 --- a/lib/raid/raid6/x86/avx512.c +++ b/lib/raid/raid6/x86/avx512.c @@ -78,6 +78,7 @@ static void raid6_avx5121_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -137,6 +138,7 @@ static void raid6_avx5121_xor_syndrome(int disks, int start, int stop, } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -208,6 +210,7 @@ static void raid6_avx5122_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -292,6 +295,7 @@ static void raid6_avx5122_xor_syndrome(int disks, int start, int stop, } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -396,6 +400,7 @@ static void raid6_avx5124_gen_syndrome(int disks, size_t bytes, void **ptrs) } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -529,6 +534,7 @@ static void raid6_avx5124_xor_syndrome(int disks, int start, int stop, "m" (q[d+128]), "m" (q[d+192])); } asm volatile("sfence" : : : "memory"); + asm volatile("vzeroupper"); kernel_fpu_end(); } const struct raid6_calls raid6_avx512x4 = { diff --git a/lib/raid/raid6/x86/recov_avx512.c b/lib/raid/raid6/x86/recov_avx512.c index ec72d5a30c01ef..299a3f044d6162 100644 --- a/lib/raid/raid6/x86/recov_avx512.c +++ b/lib/raid/raid6/x86/recov_avx512.c @@ -211,6 +211,7 @@ static void raid6_2data_recov_avx512(int disks, size_t bytes, int faila, #endif } + asm volatile("vzeroupper"); kernel_fpu_end(); } @@ -353,6 +354,7 @@ static void raid6_datap_recov_avx512(int disks, size_t bytes, int faila, #endif } + asm volatile("vzeroupper"); kernel_fpu_end(); } From c2dde78d53e03fee2de1978b7ea16e50d6705497 Mon Sep 17 00:00:00 2001 From: Hrushiraj Gandhi Date: Mon, 31 Aug 2026 20:19:53 +0530 Subject: [PATCH 0918/1012] gcov: use strscpy() instead of strcpy() in init_node() node->name is a flexible array member sized to exactly strlen(name) + 1 bytes at allocation time in new_node(), so this copy can never actually overflow. Still, prefer the bounded strscpy() over strcpy() on general principle; pass the same strlen(name) + 1 bound the allocation used, since sizeof() cannot be applied to a flexible array member. No functional change. Link: https://lore.kernel.org/20260831144953.324441-1-hrushirajg23@gmail.com Signed-off-by: Hrushiraj Gandhi Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Cc: Peter Oberparleiter --- kernel/gcov/fs.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/gcov/fs.c b/kernel/gcov/fs.c index 1d19b1be207a7d..764918570de13d 100644 --- a/kernel/gcov/fs.c +++ b/kernel/gcov/fs.c @@ -529,7 +529,7 @@ static void init_node(struct gcov_node *node, struct gcov_info *info, } node->parent = parent; if (name) - strcpy(node->name, name); + strscpy(node->name, name, strlen(name) + 1); } /* From 38a9fb08da922f0ffbe3796de2031bbe952fbbdc Mon Sep 17 00:00:00 2001 From: Mete Durlu Date: Mon, 31 Aug 2026 11:57:15 +0200 Subject: [PATCH 0919/1012] panic: introduce arch_do_panic Patch series "Introduce arch_do_panic", v6. Replace architecture-specific ifdef sections in vpanic() with a clean arch_do_panic() hook. Currently s390 and sparc embed their panic handlers directly in vpanic() using preprocessor conditionals, making the common code path harder to maintain. Introduce arch_do_panic() as an architecture extension point called at the end of vpanic(). Architectures can use this hook to implement their specific panic handling without polluting the generic panic code. Remove s390s ifdef block in vpanic() and move the corresponding code block to s390s own arch_do_panic() implementation in architecture specific code. Move sparc panic handling from ifdef blocks to arch_do_panic(). Remove the preprocessor conditionals from vpanic() and place the Stop-A enablement code in architecture-specific files where it belongs. Stop-A enablement markers are now printed after "end Kernel panic" line. To me, there are no better alternatives other than setup.c to put sparc's arch_do_panic() implementation. The other files under arch/sparc/kernel are either divided to *_32.c and *_64.c variants, which mean code duplication, or unrelated. The cleanup reduces vpanic() complexity and establishes a pattern for other architectures needing custom panic behavior. No functional changes, only minor print order changes. This patch (of 3): Introduce a hook for architectures to put their specific panic handlers. s390 and sparc already have ifdef preprocessor checks to execute architecture specific code. Pave the way for vpanic() cleanup. Link: https://lore.kernel.org/20260831-arch_do_panic-v6-1-a1e170a9e7fd@linux.ibm.com Link: https://lore.kernel.org/all/20260730-arch_do_panic-v3-0-d5401e683cdb@linux.ibm.com/ [1] Signed-off-by: Mete Durlu Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Suggested-by: Sven Schnelle Reviewed-by: Andrew Morton Cc: Alexander Gordeev Cc: Andreas Larsson Cc: Christian Borntraeger Cc: David S. Miller Cc: Heiko Carstens Cc: Petr Mladek Cc: Vasily Gorbik --- include/linux/panic.h | 2 ++ kernel/panic.c | 3 +++ 2 files changed, 5 insertions(+) diff --git a/include/linux/panic.h b/include/linux/panic.h index f1dd417e54b294..98dd7dfd27de7a 100644 --- a/include/linux/panic.h +++ b/include/linux/panic.h @@ -110,4 +110,6 @@ extern void add_taint(unsigned flag, enum lockdep_ok); extern int test_taint(unsigned flag); extern unsigned long get_taint(void); +void arch_do_panic(void); + #endif /* _LINUX_PANIC_H */ diff --git a/kernel/panic.c b/kernel/panic.c index 213725b612aa11..726a978422326f 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -567,6 +567,8 @@ static void panic_other_cpus_shutdown(bool crash_kexec) crash_smp_send_stop(); } +void __weak arch_do_panic(void) {} + /** * vpanic - halt the system * @fmt: The text string to print @@ -756,6 +758,7 @@ void vpanic(const char *fmt, va_list args) #endif pr_emerg("---[ end Kernel panic - not syncing: %s ]---\n", buf); + arch_do_panic(); /* Do not scroll important messages printed above */ suppress_printk = 1; From b3769677aa9c504567d8e5b794ebca0e3deae2ae Mon Sep 17 00:00:00 2001 From: Mete Durlu Date: Mon, 31 Aug 2026 11:57:16 +0200 Subject: [PATCH 0920/1012] s390: implement arch_do_panic Implement s390 specific arch_do_panic() instead of using s390 specific ifdef sections in vpanic() code. disabled_wait() is now called after "end Kernel panic" marker. No functional changes. Link: https://lore.kernel.org/20260831-arch_do_panic-v6-2-a1e170a9e7fd@linux.ibm.com Signed-off-by: Mete Durlu Signed-off-by: Andrew Morton Acked-by: Heiko Carstens Reviewed-by: Bradley Morgan Reviewed-by: Andrew Morton Cc: Alexander Gordeev Cc: Andreas Larsson Cc: Christian Borntraeger Cc: David S. Miller Cc: Petr Mladek Cc: Sven Schnelle Cc: Vasily Gorbik --- arch/s390/kernel/traps.c | 7 +++++++ kernel/panic.c | 3 --- 2 files changed, 7 insertions(+), 3 deletions(-) diff --git a/arch/s390/kernel/traps.c b/arch/s390/kernel/traps.c index b6ba4465f59dea..115cb337324769 100644 --- a/arch/s390/kernel/traps.c +++ b/arch/s390/kernel/traps.c @@ -26,6 +26,7 @@ #include #include #include +#include #include #include #include @@ -33,6 +34,7 @@ #include #include #include +#include #include "entry.h" struct pgm_stat { @@ -283,6 +285,11 @@ static void monitor_event_exception(struct pt_regs *regs) } } +void arch_do_panic(void) +{ + disabled_wait(); +} + void kernel_stack_invalid(struct pt_regs *regs) { /* diff --git a/kernel/panic.c b/kernel/panic.c index 726a978422326f..ee6e3f9e39002e 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -752,9 +752,6 @@ void vpanic(const char *fmt, va_list args) pr_emerg("Press Stop-A (L1-A) from sun keyboard or send break\n" "twice on console to return to the boot prom\n"); } -#endif -#if defined(CONFIG_S390) - disabled_wait(); #endif pr_emerg("---[ end Kernel panic - not syncing: %s ]---\n", buf); From eda7158f73634b7e8c8dd1f87116a73d1a2394b0 Mon Sep 17 00:00:00 2001 From: Mete Durlu Date: Mon, 31 Aug 2026 11:57:17 +0200 Subject: [PATCH 0921/1012] sparc: implement arch_do_panic Implement sparc specific arch_do_panic() instead of using sparc specific ifdef sections in vpanic() code. Reorder arch specific panic handling, sparc's Stop-A messages are now printed after "end Kernel panic" marker. Link: https://lore.kernel.org/20260831-arch_do_panic-v6-3-a1e170a9e7fd@linux.ibm.com Signed-off-by: Mete Durlu Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Reviewed-by: Andrew Morton Cc: Alexander Gordeev Cc: Andreas Larsson Cc: Christian Borntraeger Cc: David S. Miller Cc: Heiko Carstens Cc: Petr Mladek Cc: Sven Schnelle Cc: Vasily Gorbik --- arch/sparc/kernel/setup.c | 9 +++++++++ kernel/panic.c | 9 --------- 2 files changed, 9 insertions(+), 9 deletions(-) diff --git a/arch/sparc/kernel/setup.c b/arch/sparc/kernel/setup.c index 4975867d9001b6..5f43cef8063825 100644 --- a/arch/sparc/kernel/setup.c +++ b/arch/sparc/kernel/setup.c @@ -2,6 +2,8 @@ #include #include +#include +#include static const struct ctl_table sparc_sysctl_table[] = { { @@ -36,6 +38,13 @@ static const struct ctl_table sparc_sysctl_table[] = { #endif }; +void arch_do_panic(void) +{ + /* Make sure the user can actually press Stop-A (L1-A) */ + stop_a_enabled = 1; + pr_emerg("Press Stop-A (L1-A) from sun keyboard or send break\n" + "twice on console to return to the boot prom\n"); +} static int __init init_sparc_sysctls(void) { diff --git a/kernel/panic.c b/kernel/panic.c index ee6e3f9e39002e..7dda841c16f9cc 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -744,15 +744,6 @@ void vpanic(const char *fmt, va_list args) reboot_mode = panic_reboot_mode; emergency_restart(); } -#ifdef __sparc__ - { - extern int stop_a_enabled; - /* Make sure the user can actually press Stop-A (L1-A) */ - stop_a_enabled = 1; - pr_emerg("Press Stop-A (L1-A) from sun keyboard or send break\n" - "twice on console to return to the boot prom\n"); - } -#endif pr_emerg("---[ end Kernel panic - not syncing: %s ]---\n", buf); arch_do_panic(); From d86a3fa64da05a4470a20509a701b7f00e248dcc Mon Sep 17 00:00:00 2001 From: Ivy Lopez Date: Mon, 31 Aug 2026 19:41:38 -0600 Subject: [PATCH 0922/1012] lib: decompress_unxz: make it obvious that there is no memory leak Calling __decompress() or unxz() with fill == NULL && flush == NULL && in == NULL is invalid, thus there were no memory leaks even though it might have looked like that. Move the conditional free() calls so that it's obvious that there are no leaks. Link: https://lore.kernel.org/20260901014138.22699-1-skunkolee@gmail.com Link: https://lore.kernel.org/lkml/20241006072542.66442-2-t.v.s10123@gmail.com/T/ Link: https://lore.kernel.org/lkml/20260825191333.34276-1-skunkolee@gmail.com/T/ Signed-off-by: Ivy Lopez Signed-off-by: Andrew Morton Closes: https://bugzilla.kernel.org/show_bug.cgi?id=207113 Reviewed-by: Lasse Collin --- lib/decompress_unxz.c | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/lib/decompress_unxz.c b/lib/decompress_unxz.c index 05d5cb490a44e7..9ccded9934c667 100644 --- a/lib/decompress_unxz.c +++ b/lib/decompress_unxz.c @@ -342,13 +342,13 @@ STATIC int INIT unxz(unsigned char *in, long in_size, b.out_pos = 0; } } while (ret == XZ_OK); + } - if (must_free_in) - free(in); + if (must_free_in) + free(in); - if (flush != NULL) - free(b.out); - } + if (flush != NULL) + free(b.out); if (in_used != NULL) *in_used += b.in_pos; From 9a082af1491e27321252dfc2d5a4d98c3c8bc04c Mon Sep 17 00:00:00 2001 From: Julian Braha Date: Mon, 31 Aug 2026 00:00:16 +0100 Subject: [PATCH 0923/1012] arch/Kconfig: fix dead conditions by removing dead options These two 'int' options: ARCH_MMAP_RND_BITS_DEFAULT ARCH_MMAP_RND_COMPAT_BITS_DEFAULT are used directly as conditions for defaults. 'int' options should not be used as conditions, because they will always evaluate to false. Let's remove these options, because they are not used anywhere else. This dead code was found by kconfirm, a static analysis tool for Kconfig. Link: https://lore.kernel.org/20260830230016.2730093-1-julianbraha@gmail.com Signed-off-by: Julian Braha Signed-off-by: Andrew Morton Reviewed-by: Arnd Bergmann Reviewed-by: Jinjie Ruan --- arch/Kconfig | 8 -------- 1 file changed, 8 deletions(-) diff --git a/arch/Kconfig b/arch/Kconfig index 45c65777236231..72890200d049cc 100644 --- a/arch/Kconfig +++ b/arch/Kconfig @@ -1238,13 +1238,9 @@ config ARCH_MMAP_RND_BITS_MIN config ARCH_MMAP_RND_BITS_MAX int -config ARCH_MMAP_RND_BITS_DEFAULT - int - config ARCH_MMAP_RND_BITS int "Number of bits to use for ASLR of mmap base address" if EXPERT range ARCH_MMAP_RND_BITS_MIN ARCH_MMAP_RND_BITS_MAX - default ARCH_MMAP_RND_BITS_DEFAULT if ARCH_MMAP_RND_BITS_DEFAULT default ARCH_MMAP_RND_BITS_MIN depends on HAVE_ARCH_MMAP_RND_BITS help @@ -1272,13 +1268,9 @@ config ARCH_MMAP_RND_COMPAT_BITS_MIN config ARCH_MMAP_RND_COMPAT_BITS_MAX int -config ARCH_MMAP_RND_COMPAT_BITS_DEFAULT - int - config ARCH_MMAP_RND_COMPAT_BITS int "Number of bits to use for ASLR of mmap base address for compatible applications" if EXPERT range ARCH_MMAP_RND_COMPAT_BITS_MIN ARCH_MMAP_RND_COMPAT_BITS_MAX - default ARCH_MMAP_RND_COMPAT_BITS_DEFAULT if ARCH_MMAP_RND_COMPAT_BITS_DEFAULT default ARCH_MMAP_RND_COMPAT_BITS_MIN depends on HAVE_ARCH_MMAP_RND_COMPAT_BITS help From f842ec265aebdb94d4f3ed06a107035eb08325c7 Mon Sep 17 00:00:00 2001 From: Yuntao Wang Date: Tue, 11 Aug 2026 20:18:30 +0800 Subject: [PATCH 0924/1012] dyndbg: fix incorrect mod_ct value in dynamic_debug_init() Patch series "dyndbg: fix incorrect mod_ct value in dynamic_debug_init()". Fix and clean up dynamic_debug_init(). This patch (of 2): Suppose all `struct _ddebug` instances belong to the same module, mod_ct should be 1, but it is currently 0. mod_ct is incremented only when iter->modname changes, i.e. when the loop encounters the first _ddebug entry of a new module: if (strcmp(modname, iter->modname)) { mod_ct++; ... } If all _ddebug entries belong to the same module, strcmp() never returns nonzero, so mod_ct remains 0. However, the last (and in this case only) module is added after the loop: di.num_descs = mod_sites; di.descs = iter_mod_start; ret = ddebug_add_module(&di, modname); Thus, mod_ct should be incremented before adding this final module. The bug only affects the diagnostic message printed by vpr_info(): "%d prdebugs in %d modules, ..." It reports one fewer module than the actual number of modules. There is no userspace-visible runtime effect; the dynamic debug tables themselves are initialized correctly. Fix it. Link: https://lore.kernel.org/20260811121831.577848-1-yuntao.wang@linux.dev Link: https://lore.kernel.org/20260811121831.577848-2-yuntao.wang@linux.dev Signed-off-by: Yuntao Wang Signed-off-by: Andrew Morton Cc: Jason Baron Cc: Jim Cromie --- lib/dynamic_debug.c | 2 ++ 1 file changed, 2 insertions(+) diff --git a/lib/dynamic_debug.c b/lib/dynamic_debug.c index 18a71a9108d3e5..16fad5454d6a50 100644 --- a/lib/dynamic_debug.c +++ b/lib/dynamic_debug.c @@ -1456,6 +1456,8 @@ static int __init dynamic_debug_init(void) iter_mod_start = iter; } } + + mod_ct++; di.num_descs = mod_sites; di.descs = iter_mod_start; ret = ddebug_add_module(&di, modname); From eb60c5c74f196915123c17e9e35033188c6e4ed5 Mon Sep 17 00:00:00 2001 From: Yuntao Wang Date: Tue, 11 Aug 2026 20:18:31 +0800 Subject: [PATCH 0925/1012] dyndbg: clean up dynamic_debug_init() to improve readability Keep variable assignments in the same order throughout the function to make the code easier to follow. No functional changes. Link: https://lore.kernel.org/20260811121831.577848-3-yuntao.wang@linux.dev Signed-off-by: Yuntao Wang Signed-off-by: Andrew Morton Cc: Jason Baron Cc: Jim Cromie --- lib/dynamic_debug.c | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/lib/dynamic_debug.c b/lib/dynamic_debug.c index 16fad5454d6a50..49334d1aa4b3af 100644 --- a/lib/dynamic_debug.c +++ b/lib/dynamic_debug.c @@ -1442,28 +1442,29 @@ static int __init dynamic_debug_init(void) i = mod_sites = mod_ct = 0; for (; iter < __stop___dyndbg; iter++, i++, mod_sites++) { - if (strcmp(modname, iter->modname)) { - mod_ct++; - di.num_descs = mod_sites; di.descs = iter_mod_start; + di.num_descs = mod_sites; ret = ddebug_add_module(&di, modname); if (ret) goto out_err; - mod_sites = 0; - modname = iter->modname; + mod_ct++; + iter_mod_start = iter; + modname = iter->modname; + mod_sites = 0; } } - mod_ct++; - di.num_descs = mod_sites; di.descs = iter_mod_start; + di.num_descs = mod_sites; ret = ddebug_add_module(&di, modname); if (ret) goto out_err; + mod_ct++; + ddebug_init_success = 1; vpr_info("%d prdebugs in %d modules, %d KiB in ddebug tables, %d kiB in __dyndbg section\n", i, mod_ct, (int)((mod_ct * sizeof(struct ddebug_table)) >> 10), From 6e420ed44bbc818b3ecd005acb841bee09058cdf Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Fri, 10 Jul 2026 14:39:57 +0200 Subject: [PATCH 0926/1012] fork: honor task_struct's declared alignment Since commit cb7ca40a3882 ("x86/fpu: Make task_struct::thread constant size"), struct task_struct is declared __attribute__((aligned(64))) on all architectures. But fork_init() still sets the task_struct slab cache's alignment to align = max(L1_CACHE_BYTES, ARCH_MIN_TASKALIGN) which is smaller than 64 on architectures whose cache lines are below 64 bytes: e.g. 32 on ARMv5. In practice plain SLUB happens to hand out 64-byte-aligned objects anyway. With CONFIG_SLUB_DEBUG_ON the red-zone padding shifts objects to the requested alignment. With CONFIG_UBSAN_ALIGNMENT=y a boot on QEMU versatilepb (ARM926EJ-S, v7.2-rc2, gcc 13.3) floods the console with reports like: UBSAN: misaligned-access in include/linux/sched.h:2087:9 member access within misaligned address c295d7e0 for type 'struct task_struct' which requires 64 byte alignment CPU: 0 UID: 0 PID: 15 Comm: pr/ttyAMA-1 Not tainted 7.2.0-rc2 #1 VOLUNTARY Set the slab alignment to at least the type's declared alignment. Link: https://lore.kernel.org/20260710123957.31774-1-kmehltretter@gmail.com Fixes: cb7ca40a3882 ("x86/fpu: Make task_struct::thread constant size") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Assisted-by: Claude:claude-fable-5 Cc: Ingo Molnar Cc: Kees Cook Cc: Peter Zijlstra Cc: Vlastimil Babka Cc: --- kernel/fork.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/kernel/fork.c b/kernel/fork.c index 10f2d05d816a5f..4558150fd133a9 100644 --- a/kernel/fork.c +++ b/kernel/fork.c @@ -858,7 +858,8 @@ void __init fork_init(void) #ifndef ARCH_MIN_TASKALIGN #define ARCH_MIN_TASKALIGN 0 #endif - int align = max_t(int, L1_CACHE_BYTES, ARCH_MIN_TASKALIGN); + int align = max3(L1_CACHE_BYTES, ARCH_MIN_TASKALIGN, + __alignof__(struct task_struct)); unsigned long useroffset, usersize; /* create a slab on which task_structs can be allocated */ From e2b3f246a102d6ff69815bc4a286fb9e508ef9a9 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Sat, 1 Aug 2026 22:58:44 +0000 Subject: [PATCH 0927/1012] taskstats: fold the two pid/tgid handlers into one cmd_attr_pid() and cmd_attr_tgid() are copy paste. fold them into one handler that takes the attr type and fill function as parameters, same pattern as the cpumask fold in 59c0bc949c5e. No functional change. Link: https://lore.kernel.org/20260801225845.23855-1-include@grrlz.net Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Reviewed-by: Andrew Morton Cc: Balbir Singh Cc: Balbir Singh --- kernel/taskstats.c | 47 ++++++++++++---------------------------------- 1 file changed, 12 insertions(+), 35 deletions(-) diff --git a/kernel/taskstats.c b/kernel/taskstats.c index 9a48827e22bce3..598d9cd8325018 100644 --- a/kernel/taskstats.c +++ b/kernel/taskstats.c @@ -473,7 +473,8 @@ static size_t taskstats_packet_size(void) return size; } -static int cmd_attr_pid(struct genl_info *info) +static int cmd_attr_pid_tgid(struct genl_info *info, int attr, + int (*fill)(pid_t, struct taskstats *)) { struct taskstats *stats; struct sk_buff *rep_skb; @@ -488,41 +489,15 @@ static int cmd_attr_pid(struct genl_info *info) return rc; rc = -EINVAL; - pid = nla_get_u32(info->attrs[TASKSTATS_CMD_ATTR_PID]); - stats = mk_reply(rep_skb, TASKSTATS_TYPE_PID, pid); + pid = nla_get_u32(info->attrs[attr]); + stats = mk_reply(rep_skb, + attr == TASKSTATS_CMD_ATTR_PID + ? TASKSTATS_TYPE_PID : TASKSTATS_TYPE_TGID, + pid); if (!stats) goto err; - rc = fill_stats_for_pid(pid, stats); - if (rc < 0) - goto err; - return send_reply(rep_skb, info); -err: - nlmsg_free(rep_skb); - return rc; -} - -static int cmd_attr_tgid(struct genl_info *info) -{ - struct taskstats *stats; - struct sk_buff *rep_skb; - size_t size; - u32 tgid; - int rc; - - size = taskstats_packet_size(); - - rc = prepare_reply(info, TASKSTATS_CMD_NEW, &rep_skb, size); - if (rc < 0) - return rc; - - rc = -EINVAL; - tgid = nla_get_u32(info->attrs[TASKSTATS_CMD_ATTR_TGID]); - stats = mk_reply(rep_skb, TASKSTATS_TYPE_TGID, tgid); - if (!stats) - goto err; - - rc = fill_stats_for_tgid(tgid, stats); + rc = fill(pid, stats); if (rc < 0) goto err; return send_reply(rep_skb, info); @@ -542,9 +517,11 @@ static int taskstats_user_cmd(struct sk_buff *skb, struct genl_info *info) TASKSTATS_CMD_ATTR_DEREGISTER_CPUMASK, DEREGISTER); else if (info->attrs[TASKSTATS_CMD_ATTR_PID]) - return cmd_attr_pid(info); + return cmd_attr_pid_tgid(info, TASKSTATS_CMD_ATTR_PID, + fill_stats_for_pid); else if (info->attrs[TASKSTATS_CMD_ATTR_TGID]) - return cmd_attr_tgid(info); + return cmd_attr_pid_tgid(info, TASKSTATS_CMD_ATTR_TGID, + fill_stats_for_tgid); else return -EINVAL; } From 0c513d24e33ad32edab1c95b3e7d6e3cd47f8f56 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 1 Sep 2026 20:52:18 +0800 Subject: [PATCH 0928/1012] ocfs2: restrict OCFS2_INVALID_SLOT suballoc slot to system inodes Patch series "ocfs2: validate suballoc slot and bit of metadata blocks", v3. The ocfs2 metadata validators trust the on-disk suballoc slot and bit without checking them against the slot range of the mounted filesystem and the capacity of the block group bitmap. A corrupted image can carry OCFS2_INVALID_SLOT or another out-of-range slot, or a suballoc bit beyond the bitmap, and once the corresponding inode, extent block, xattr block, dir index root or refcount block gets freed, the bad value goes straight into ocfs2_get_system_file_inode() or _ocfs2_free_suballoc_bits() and hits a BUG_ON() or runs off the end of local_system_inodes[]. This series rejects such values at read time, in the existing validators, so corrupted objects fail with -EROFS (and a read-only remount) instead of crashing: patch 1 restricts OCFS2_INVALID_SLOT dinodes to system inodes, completing fe7a283b3916 ("ocfs2: add suballoc slot check in ocfs2_validate_inode_block()"), and turns the "system file state is ambiguous" BUG_ON() in ocfs2_read_locked_inode() into an ocfs2_error(); patch 2 rejects oversized dinode suballoc bits; patch 3 validates the suballoc slot and bit of xattr and dir index blocks; patch 4 validates the suballoc slot and bit of extent and refcount blocks. The checks only enforce what the kernel and mkfs.ocfs2 already write: a valid slot from meta_ac->ac_alloc_slot and a bit within the block group bitmap, with system inodes carrying OCFS2_INVALID_SLOT plus OCFS2_SYSTEM_FL and extent blocks using slot 0. Nothing changes for healthy filesystems. Each new check was exercised under QEMU by corrupting the field in question with an out-of-range value; with the series applied the access fails with -EROFS and the filesystem remounts read-only instead of hitting the BUG_ON(). This patch (of 4): ocfs2_validate_inode_block() currently permits i_suballoc_slot to be OCFS2_INVALID_SLOT for any dinode. Only system inodes created by mkfs.ocfs2 are allocated from the global allocator and thus legitimately carry this value; regular inodes are always allocated from a per-slot suballocator and hence must have a valid slot. If a corrupted regular inode with OCFS2_INVALID_SLOT is accepted, ocfs2_remove_inode() will pass the slot to ocfs2_get_system_file_inode() and get_local_system_inode() will hit BUG_ON(slot == OCFS2_INVALID_SLOT) when the inode is deleted. This can be triggered by an unprivileged user unlinking such a corrupted file. Reject OCFS2_INVALID_SLOT for non-system dinodes during validation, while still accepting it for system inodes. Note that a crafted dinode carrying OCFS2_SYSTEM_FL passes the check above, yet a plain lookup of it still used to BUG() in ocfs2_read_locked_inode() ("system file state is ambiguous"). Since i_flags comes from disk, handle that mismatch with ocfs2_error() instead of BUG_ON() as well. Link: https://lore.kernel.org/20260901125221.1634686-1-joseph.qi@linux.alibaba.com Link: https://lore.kernel.org/20260901125221.1634686-2-joseph.qi@linux.alibaba.com Fixes: fe7a283b3916 ("ocfs2: add suballoc slot check in ocfs2_validate_inode_block()") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Changwei Ge Cc: Heming Zhao Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh Cc: --- fs/ocfs2/inode.c | 37 ++++++++++++++++++++++++++++--------- 1 file changed, 28 insertions(+), 9 deletions(-) diff --git a/fs/ocfs2/inode.c b/fs/ocfs2/inode.c index 180107a11046c5..9228d6ef23c24d 100644 --- a/fs/ocfs2/inode.c +++ b/fs/ocfs2/inode.c @@ -638,14 +638,18 @@ static int ocfs2_read_locked_inode(struct inode *inode, fe = (struct ocfs2_dinode *) bh->b_data; /* - * This is a code bug. Right now the caller needs to - * understand whether it is asking for a system file inode or - * not so the proper lock names can be built. + * The caller must know whether it is asking for a system file inode + * or not so the proper lock names can be built. Since i_flags comes + * from disk, a mismatch is filesystem corruption instead of a code + * bug, so handle it with ocfs2_error() rather than BUG_ON(). */ - mlog_bug_on_msg(!!(fe->i_flags & cpu_to_le32(OCFS2_SYSTEM_FL)) != - !!(args->fi_flags & OCFS2_FI_FLAG_SYSFILE), - "Inode %llu: system file state is ambiguous\n", - (unsigned long long)args->fi_blkno); + if (!!(fe->i_flags & cpu_to_le32(OCFS2_SYSTEM_FL)) != + !!(args->fi_flags & OCFS2_FI_FLAG_SYSFILE)) { + status = ocfs2_error(osb->sb, + "Inode %llu: system file state is ambiguous\n", + (unsigned long long)args->fi_blkno); + goto bail; + } if (S_ISCHR(le16_to_cpu(fe->i_mode)) || S_ISBLK(le16_to_cpu(fe->i_mode))) @@ -1520,8 +1524,23 @@ int ocfs2_validate_inode_block(struct super_block *sb, goto bail; } - if (le16_to_cpu(di->i_suballoc_slot) != (u16)OCFS2_INVALID_SLOT && - (u32)le16_to_cpu(di->i_suballoc_slot) > OCFS2_SB(sb)->max_slots - 1) { + /* + * Only system inodes created by mkfs.ocfs2 are allocated from the + * global allocator and thus legitimately carry OCFS2_INVALID_SLOT. + * Regular inodes are always allocated from a per-slot suballocator. + * If a regular inode with OCFS2_INVALID_SLOT was accepted here, + * deleting it would pass the slot to get_local_system_inode() via + * ocfs2_remove_inode() and trigger BUG_ON(slot == OCFS2_INVALID_SLOT). + */ + if (le16_to_cpu(di->i_suballoc_slot) == (u16)OCFS2_INVALID_SLOT) { + if (!(le32_to_cpu(di->i_flags) & OCFS2_SYSTEM_FL)) { + rc = ocfs2_error(sb, + "Invalid dinode %llu: suballoc slot %u for non-system inode\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(di->i_suballoc_slot)); + goto bail; + } + } else if ((u32)le16_to_cpu(di->i_suballoc_slot) > OCFS2_SB(sb)->max_slots - 1) { rc = ocfs2_error(sb, "Invalid dinode %llu: suballoc slot %u\n", (unsigned long long)bh->b_blocknr, le16_to_cpu(di->i_suballoc_slot)); From 1c7e8be5a4862d33f656a4020511cb118922498f Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 1 Sep 2026 20:52:19 +0800 Subject: [PATCH 0929/1012] ocfs2: validate suballoc bit during inode read i_suballoc_bit of a dinode is currently not validated at all. A corrupted dinode can carry an abnormally large i_suballoc_bit, which bypasses ocfs2_validate_inode_block(). When the inode is deleted, ocfs2_remove_inode() calls ocfs2_free_dinode(), which passes the unvalidated bit to _ocfs2_free_suballoc_bits() and triggers BUG_ON((count + start_bit) > ocfs2_bits_per_group(cl)). A suballocator block group bitmap is contained in a single block and starts after the group descriptor header, so a valid suballoc bit must be smaller than the number of bits fitting in the remaining space. Reject oversized i_suballoc_bit values during dinode validation. The bound is derived from ocfs2_group_bitmap_size() so it is also tight when discontig_bg caps the suballocator bitmap at OCFS2_MAX_BG_BITMAP_SIZE. Note the above check alone is not sufficient since the freeing path compares the bit against ocfs2_bits_per_group(), which is derived from cl_cpg/cl_bpc of the allocator dinode that is not validated against the actual group capacity and can be artificially smaller on a corrupted image. Convert this BUG_ON in _ocfs2_free_suballoc_bits() to ocfs2_error() as well. Link: https://lore.kernel.org/20260901125221.1634686-3-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Changwei Ge Cc: Heming Zhao Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh --- fs/ocfs2/inode.c | 16 ++++++++++++++++ fs/ocfs2/ocfs2.h | 12 ++++++++++++ fs/ocfs2/suballoc.c | 18 +++++++++++++++--- 3 files changed, 43 insertions(+), 3 deletions(-) diff --git a/fs/ocfs2/inode.c b/fs/ocfs2/inode.c index 9228d6ef23c24d..92f3450010fbb0 100644 --- a/fs/ocfs2/inode.c +++ b/fs/ocfs2/inode.c @@ -1547,6 +1547,22 @@ int ocfs2_validate_inode_block(struct super_block *sb, goto bail; } + /* + * A suballocator block group bitmap is contained in a single block + * and starts after the group descriptor header, so a valid suballoc + * bit can never exceed ocfs2_suballoc_bits_per_block(). Otherwise + * deleting the inode will pass the oversized bit to + * _ocfs2_free_suballoc_bits() via ocfs2_free_dinode() and trigger + * BUG_ON((count + start_bit) > ocfs2_bits_per_group(cl)), since any + * group holds at most ocfs2_suballoc_bits_per_block() bits. + */ + if (le16_to_cpu(di->i_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + rc = ocfs2_error(sb, "Invalid dinode %llu: suballoc bit %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(di->i_suballoc_bit)); + goto bail; + } + if ((le32_to_cpu(di->i_flags) & OCFS2_ORPHANED_FL) && le16_to_cpu(di->i_orphaned_slot) >= OCFS2_SB(sb)->max_slots) { rc = ocfs2_error(sb, "Invalid dinode %llu: orphaned slot %u\n", diff --git a/fs/ocfs2/ocfs2.h b/fs/ocfs2/ocfs2.h index b747cdec178758..e3bb3cc0b25a29 100644 --- a/fs/ocfs2/ocfs2.h +++ b/fs/ocfs2/ocfs2.h @@ -593,6 +593,18 @@ static inline int ocfs2_supports_discontig_bg(struct ocfs2_super *osb) return 0; } +/* + * A suballocator block group bitmap starts right after the group + * descriptor header, so a suballoc bit can never exceed this number + * of bits. Derive it from ocfs2_group_bitmap_size() which also caps + * it at OCFS2_MAX_BG_BITMAP_SIZE when discontig_bg is enabled. + */ +static inline u32 ocfs2_suballoc_bits_per_block(struct super_block *sb) +{ + return ocfs2_group_bitmap_size(sb, 1, + OCFS2_SB(sb)->s_feature_incompat) * 8; +} + static inline unsigned int ocfs2_link_max(struct ocfs2_super *osb) { if (ocfs2_supports_indexed_dirs(osb)) diff --git a/fs/ocfs2/suballoc.c b/fs/ocfs2/suballoc.c index 453b56be9624c6..ce22d0c3d28748 100644 --- a/fs/ocfs2/suballoc.c +++ b/fs/ocfs2/suballoc.c @@ -3040,10 +3040,22 @@ static int _ocfs2_free_suballoc_bits(handle_t *handle, /* The alloc_bh comes from ocfs2_free_dinode() or * ocfs2_free_clusters(). The callers have all locked the * allocator and gotten alloc_bh from the lock call. This - * validates the dinode buffer. Any corruption that has happened - * is a code bug. */ + * validates the dinode buffer. */ BUG_ON(!OCFS2_IS_VALID_DINODE(fe)); - BUG_ON((count + start_bit) > ocfs2_bits_per_group(cl)); + + /* + * ocfs2_bits_per_group() is derived from cl_cpg and cl_bpc of the + * allocator dinode, which are not validated against the volume + * geometry. A corrupted image can carry a suballoc bit beyond it, + * so error out instead of crashing. + */ + if ((count + start_bit) > ocfs2_bits_per_group(cl)) { + return ocfs2_error(alloc_inode->i_sb, + "Allocator #%llu: freeing bits %u+%u exceeds bits per group %u\n", + (unsigned long long)le64_to_cpu(fe->i_blkno), + count, start_bit, + ocfs2_bits_per_group(cl)); + } trace_ocfs2_free_suballoc_bits( (unsigned long long)OCFS2_I(alloc_inode)->ip_blkno, From 3dd5830bbcc6cb3beb1e92a2b65da9b60ae00ad9 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 1 Sep 2026 20:52:20 +0800 Subject: [PATCH 0930/1012] ocfs2: validate suballoc slot and bit of xattr and dir index blocks ocfs2_validate_xattr_block() and ocfs2_validate_dx_root() do not validate xb_suballoc_slot, xb_suballoc_bit, dr_suballoc_slot and dr_suballoc_bit at all. Since xattr blocks and dir index root blocks are allocated from a per-slot suballocator at runtime, their suballoc slots must be within range and their suballoc bits must fit in a block group bitmap. Otherwise a corrupted image can carry an out-of-range slot. When the xattr block or dir index is removed, ocfs2_xattr_block_remove() or ocfs2_dx_dir_remove_index() passes the unvalidated slot to ocfs2_get_system_file_inode() and get_local_system_inode() will either hit BUG_ON(slot == OCFS2_INVALID_SLOT) or compute an out-of-bounds index into the local_system_inodes array. Similarly an oversized suballoc bit will error out the filesystem in _ocfs2_free_suballoc_bits(). Furthermore ocfs2_validate_dx_root() does not verify dr_blkno against the physical block number like the extent and xattr block validators do, so a misplaced dir index root block can pass validation. Reject misplaced dir index root blocks, out-of-range suballoc slots and oversized suballoc bits during validation. Link: https://lore.kernel.org/20260901125221.1634686-4-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Changwei Ge Cc: Heming Zhao Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh --- fs/ocfs2/dir.c | 35 +++++++++++++++++++++++++++++++++++ fs/ocfs2/xattr.c | 25 +++++++++++++++++++++++++ 2 files changed, 60 insertions(+) diff --git a/fs/ocfs2/dir.c b/fs/ocfs2/dir.c index 0075e1624310e8..6bb6aa133f0150 100644 --- a/fs/ocfs2/dir.c +++ b/fs/ocfs2/dir.c @@ -605,6 +605,41 @@ static int ocfs2_validate_dx_root(struct super_block *sb, goto bail; } + if (le64_to_cpu(dx_root->dr_blkno) != bh->b_blocknr) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has an invalid dr_blkno of %llu\n", + (unsigned long long)bh->b_blocknr, + (unsigned long long)le64_to_cpu(dx_root->dr_blkno)); + goto bail; + } + + /* + * Dir index root blocks are allocated from a per-slot suballocator, + * so the slot must be in range. Otherwise removing the index passes + * it to get_local_system_inode(), which hits BUG_ON() for + * OCFS2_INVALID_SLOT or computes an out-of-bounds index otherwise. + */ + if ((u32)le16_to_cpu(dx_root->dr_suballoc_slot) >= OCFS2_SB(sb)->max_slots) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has invalid dr_suballoc_slot %u\n", + (unsigned long long)le64_to_cpu(dx_root->dr_blkno), + le16_to_cpu(dx_root->dr_suballoc_slot)); + goto bail; + } + + /* + * Similarly the suballoc bit must fit in a block group bitmap. + * Otherwise removing the index will pass the oversized bit to + * _ocfs2_free_suballoc_bits() and trigger ocfs2_error() there. + */ + if (le16_to_cpu(dx_root->dr_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has invalid dr_suballoc_bit %u\n", + (unsigned long long)le64_to_cpu(dx_root->dr_blkno), + le16_to_cpu(dx_root->dr_suballoc_bit)); + goto bail; + } + if (!(dx_root->dr_flags & OCFS2_DX_FLAG_INLINE)) { struct ocfs2_extent_list *el = &dx_root->dr_list; diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index 740d4bb3890f9c..e27f5f925df52e 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -532,6 +532,31 @@ static int ocfs2_validate_xattr_block(struct super_block *sb, le32_to_cpu(xb->xb_fs_generation)); } + /* + * Xattr blocks are allocated from a per-slot suballocator, so the + * slot must be in range. Otherwise freeing the block passes it to + * get_local_system_inode(), which hits BUG_ON() for + * OCFS2_INVALID_SLOT or computes an out-of-bounds index otherwise. + */ + if ((u32)le16_to_cpu(xb->xb_suballoc_slot) >= OCFS2_SB(sb)->max_slots) { + return ocfs2_error(sb, + "Extended attribute block #%llu has an invalid xb_suballoc_slot of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(xb->xb_suballoc_slot)); + } + + /* + * Similarly the suballoc bit must fit in a block group bitmap. + * Otherwise freeing the block will pass the oversized bit to + * _ocfs2_free_suballoc_bits() and trigger ocfs2_error() there. + */ + if (le16_to_cpu(xb->xb_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + return ocfs2_error(sb, + "Extended attribute block #%llu has an invalid xb_suballoc_bit of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(xb->xb_suballoc_bit)); + } + if (!(le16_to_cpu(xb->xb_flags) & OCFS2_XATTR_INDEXED)) { size_t region_offset = offsetof(struct ocfs2_xattr_block, xb_attrs.xb_header); From fa2b9a026311a416db111d4e7e1da2a70acdcd61 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 1 Sep 2026 20:52:21 +0800 Subject: [PATCH 0931/1012] ocfs2: validate suballoc slot and bit of extent and refcount blocks ocfs2_validate_extent_block() and ocfs2_validate_refcount_block() do not validate h_suballoc_slot, h_suballoc_bit, rf_suballoc_slot and rf_suballoc_bit at all. Since extent blocks and refcount blocks are allocated from a per-slot suballocator at runtime, their suballoc slots must be within range and their suballoc bits must fit in a block group bitmap. Otherwise a corrupted image can carry an out-of-range slot. When the extent block is freed, ocfs2_cache_extent_block_free() caches it and ocfs2_free_cached_blocks() later passes the unvalidated slot to ocfs2_get_system_file_inode(); when the refcount block is freed, ocfs2_remove_refcount_extent() passes it via ocfs2_cache_block_dealloc(). get_local_system_inode() will then either hit BUG_ON(slot == OCFS2_INVALID_SLOT) or compute an out-of-bounds index into the local_system_inodes array. Similarly an oversized suballoc bit will error out the filesystem in _ocfs2_free_suballoc_bits(). Furthermore group descriptor validation only guarantees bg_bits within the physical bitmap size, so a corrupted image can still carry a suballoc bit beyond bg_bits, which would let ocfs2_block_group_clear_bits() clear bits beyond bg_bitmap. Convert the remaining BUG_ON against group->bg_bits in _ocfs2_free_suballoc_bits() to ocfs2_error() as well. Reject out-of-range suballoc slots and oversized suballoc bits during validation. Link: https://lore.kernel.org/20260901125221.1634686-5-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao --- fs/ocfs2/alloc.c | 27 +++++++++++++++++++++++++++ fs/ocfs2/refcounttree.c | 27 +++++++++++++++++++++++++++ fs/ocfs2/suballoc.c | 15 ++++++++++++++- 3 files changed, 68 insertions(+), 1 deletion(-) diff --git a/fs/ocfs2/alloc.c b/fs/ocfs2/alloc.c index be09e766ac1fc9..2fdc5403b10b04 100644 --- a/fs/ocfs2/alloc.c +++ b/fs/ocfs2/alloc.c @@ -925,6 +925,33 @@ static int ocfs2_validate_extent_block(struct super_block *sb, goto bail; } + /* + * Extent blocks are allocated from a per-slot suballocator, so the + * slot must be in range. Otherwise freeing the block passes it to + * get_local_system_inode(), which hits BUG_ON() for + * OCFS2_INVALID_SLOT or computes an out-of-bounds index otherwise. + */ + if ((u32)le16_to_cpu(eb->h_suballoc_slot) >= OCFS2_SB(sb)->max_slots) { + rc = ocfs2_error(sb, + "Extent block #%llu has an invalid h_suballoc_slot of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(eb->h_suballoc_slot)); + goto bail; + } + + /* + * Similarly the suballoc bit must fit in a block group bitmap. + * Otherwise freeing the block will pass the oversized bit to + * _ocfs2_free_suballoc_bits() and trigger ocfs2_error() there. + */ + if (le16_to_cpu(eb->h_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + rc = ocfs2_error(sb, + "Extent block #%llu has an invalid h_suballoc_bit of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(eb->h_suballoc_bit)); + goto bail; + } + if (le16_to_cpu(eb->h_list.l_count) != ocfs2_extent_recs_per_eb(sb)) { rc = ocfs2_error(sb, "Extent block #%llu has invalid l_count %u (expected %u)\n", diff --git a/fs/ocfs2/refcounttree.c b/fs/ocfs2/refcounttree.c index d9f22b4a265461..3e9cccf06e48cb 100644 --- a/fs/ocfs2/refcounttree.c +++ b/fs/ocfs2/refcounttree.c @@ -117,6 +117,33 @@ static int ocfs2_validate_refcount_block(struct super_block *sb, goto out; } + /* + * Refcount blocks are allocated from a per-slot suballocator, so the + * slot must be in range. Otherwise freeing the block passes it to + * get_local_system_inode(), which hits BUG_ON() for + * OCFS2_INVALID_SLOT or computes an out-of-bounds index otherwise. + */ + if ((u32)le16_to_cpu(rb->rf_suballoc_slot) >= OCFS2_SB(sb)->max_slots) { + rc = ocfs2_error(sb, + "Refcount block #%llu has an invalid rf_suballoc_slot of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(rb->rf_suballoc_slot)); + goto out; + } + + /* + * Similarly the suballoc bit must fit in a block group bitmap. + * Otherwise freeing the block will pass the oversized bit to + * _ocfs2_free_suballoc_bits() and trigger ocfs2_error() there. + */ + if (le16_to_cpu(rb->rf_suballoc_bit) >= ocfs2_suballoc_bits_per_block(sb)) { + rc = ocfs2_error(sb, + "Refcount block #%llu has an invalid rf_suballoc_bit of %u\n", + (unsigned long long)bh->b_blocknr, + le16_to_cpu(rb->rf_suballoc_bit)); + goto out; + } + /* * rf_records (rl_count/rl_used/rl_recs[]) is only meaningful when * this block is not an interior tree block (OCFS2_REFCOUNT_TREE_FL); diff --git a/fs/ocfs2/suballoc.c b/fs/ocfs2/suballoc.c index ce22d0c3d28748..624152e4f7fedf 100644 --- a/fs/ocfs2/suballoc.c +++ b/fs/ocfs2/suballoc.c @@ -3070,7 +3070,20 @@ static int _ocfs2_free_suballoc_bits(handle_t *handle, } group = (struct ocfs2_group_desc *) group_bh->b_data; - BUG_ON((count + start_bit) > le16_to_cpu(group->bg_bits)); + /* + * Group descriptor validation only guarantees bg_bits within the + * physical bitmap size, so double check the freeing range here. + * Otherwise ocfs2_block_group_clear_bits() would clear bits beyond + * bg_bitmap. + */ + if ((count + start_bit) > le16_to_cpu(group->bg_bits)) { + status = ocfs2_error(alloc_inode->i_sb, + "Group descriptor #%llu has %u bits, cannot free bits %u+%u\n", + (unsigned long long)le64_to_cpu(group->bg_blkno), + le16_to_cpu(group->bg_bits), + count, start_bit); + goto bail; + } if (ocfs2_is_cluster_bitmap(alloc_inode)) old_bg_contig_free_bits = group->bg_contig_free_bits; From c9ac50b19814b095e80840d90171bf1077cde427 Mon Sep 17 00:00:00 2001 From: Feng Tang Date: Wed, 2 Sep 2026 19:48:51 +0800 Subject: [PATCH 0932/1012] panic: remove the unneeded panic_print_get() panic_print_get() was introduced in commit 2683df6539cb ("panic: add note that 'panic_print' parameter is deprecated") to print out warning message of the deprecation of 'panic_print' on read access. Since commit 90f3c123247e ("panic: only warn about deprecated panic_print on write access"), panic_print_get() wrapper is not needed anymore for read access, so remove it and use param_get_ulong() instead. Link: https://lore.kernel.org/20260902114851.77062-1-feng.tang@linux.alibaba.com Signed-off-by: Feng Tang Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Reviewed-by: Andrew Morton Reviewed-by: Petr Mladek --- kernel/panic.c | 7 +------ 1 file changed, 1 insertion(+), 6 deletions(-) diff --git a/kernel/panic.c b/kernel/panic.c index 7dda841c16f9cc..50715f14cf04ef 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -1216,14 +1216,9 @@ static int panic_print_set(const char *val, const struct kernel_param *kp) return param_set_ulong(val, kp); } -static int panic_print_get(char *val, const struct kernel_param *kp) -{ - return param_get_ulong(val, kp); -} - static const struct kernel_param_ops panic_print_ops = { .set = panic_print_set, - .get = panic_print_get, + .get = param_get_ulong, }; __core_param_cb(panic_print, &panic_print_ops, &panic_print, 0644); From fda52db7213c30c1839e3ea27b6c922ebf6de9da Mon Sep 17 00:00:00 2001 From: Nick Desaulniers Date: Wed, 2 Sep 2026 14:01:53 -0700 Subject: [PATCH 0933/1012] scripts/checkstack.pl: support llvm-objdump disassembly for x86 When running `make LLVM=1 checkstack`, OBJDUMP is set to llvm-objdump. llvm-objdump outputs disassembly with instruction size suffixes (such as subq/addq/subl/addl), tabs/whitespace differences, spaces after commas, and trailing comments (e.g. `# imm = 0x...`). Because scripts/checkstack.pl used rigid regular expressions specifically tuned to GNU objdump format (e.g. requiring exactly four spaces, no suffix, and no space after comma), checkstack.pl failed to match any stack adjustment instructions and yielded no output when using llvm-objdump on x86. Update the regular expressions for x86 to match optional suffixes, variable whitespace, and trailing comments. Link: https://lore.kernel.org/20260902-checkstack_llvm_objdump-v1-1-edb4eca5f163@google.com Signed-off-by: Nick Desaulniers Signed-off-by: Andrew Morton Assisted-by: LLM Gemini Cc: Bill Wendling Cc: Justin Stitt Cc: Nathan Chancellor --- scripts/checkstack.pl | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/scripts/checkstack.pl b/scripts/checkstack.pl index 14ce31f732ee8a..f8011e789a9b45 100755 --- a/scripts/checkstack.pl +++ b/scripts/checkstack.pl @@ -66,8 +66,10 @@ #c0105234: 81 ec ac 05 00 00 sub $0x5ac,%esp # or # 2f60: 48 81 ec e8 05 00 00 sub $0x5e8,%rsp - $re = qr/^.*[as][du][db] \$(0x$x{1,8}),\%(e|r)sp$/o; - $dre = qr/^.*[as][du][db] (%.*),\%(e|r)sp$/o; + # or + # 0: 48 81 ec e8 05 00 00 subq $0x5e8, %rsp + $re = qr/^.*[as][du][db][ql]?\s+\$(0x$x{1,8}),\s*\%(e|r)sp/o; + $dre = qr/^.*[as][du][db][ql]?\s+(%.*),\s*\%(e|r)sp/o; } elsif ($arch eq 'm68k') { # 2b6c: 4e56 fb70 linkw %fp,#-1168 # 1df770: defc ffe4 addaw #-28,%sp From ee5fba5d4d503dfc68be5c3051d54d0d90fb6939 Mon Sep 17 00:00:00 2001 From: Maximilian Heyne Date: Fri, 19 Jun 2026 11:24:29 +0000 Subject: [PATCH 0934/1012] selftests: uevent filtering: don't shrink the socket buffer The uevent_filtering test shrinks the uevent socket buffer to 4 KB although the default socket buffer size is much higher. This leads to this test being flaky when too many unrelated uevents are fired on the machine. They might fill up the netlink receive buffer leading to ENOBUFS errors when trying to receive the uevents. For example, I could trigger test failures when running triggering a lot of udev events in the background: $ # run multiple of that in the background: $ while :; do sudo udevadm trigger --action=change; done & $ sudo ./uevent_filtering # Starting 1 tests from 1 test cases. # RUN global.uevent_filtering ... add@/devices/virtual/mem/fullACTION=addDEVPATH=/devices/virtual/mem/fullSUBSYSTEM=memSYNTH_UUID=0MAJOR=1MINOR=7DEVNAME=fullDEVMODE=0666SEQNUM=304458 add@/devices/virtual/mem/fullACTION=addDEVPATH=/devices/virtual/mem/fullSUBSYSTEM=memSYNTH_UUID=0MAJOR=1MINOR=7DEVNAME=fullDEVMODE=0666SEQNUM=304471 add@/devices/virtual/mem/fullACTION=addDEVPATH=/devices/virtual/mem/fullSUBSYSTEM=memSYNTH_UUID=0MAJOR=1MINOR=7DEVNAME=fullDEVMODE=0666SEQNUM=304481 add@/devices/virtual/mem/fullACTION=addDEVPATH=/devices/virtual/mem/fullSUBSYSTEM=memSYNTH_UUID=0MAJOR=1MINOR=7DEVNAME=fullDEVMODE=0666SEQNUM=349156 No buffer space available - Failed to receive uevent # uevent_filtering.c:463:uevent_filtering:Expected 0 (0) == ret (-1) # uevent_filtering: Test failed # FAIL global.uevent_filtering not ok 1 global.uevent_filtering The default receive buffer size (SK_RMEM_MAX) is far larger than the requested 4 KB, so keep this to make the test less flaky. Link: https://lore.kernel.org/20260619-get-swam-a1cd4cca@mheyne-amazon Fixes: 9d3df886d17b ("selftests: uevent filtering") Signed-off-by: Maximilian Heyne Signed-off-by: Andrew Morton Cc: Christian Brauner Cc: David S. Miller Cc: Shuah Khan Cc: Wei Yang Cc: --- tools/testing/selftests/uevent/uevent_filtering.c | 8 -------- 1 file changed, 8 deletions(-) diff --git a/tools/testing/selftests/uevent/uevent_filtering.c b/tools/testing/selftests/uevent/uevent_filtering.c index 33a09f66d7e22f..8c623845e3d042 100644 --- a/tools/testing/selftests/uevent/uevent_filtering.c +++ b/tools/testing/selftests/uevent/uevent_filtering.c @@ -78,7 +78,6 @@ static int uevent_listener(unsigned long post_flags, bool expect_uevent, { int sk_fd, ret; socklen_t sk_addr_len; - int rcv_buf_sz = __UEVENT_BUFFER_SIZE; uint64_t sync_add = 1; struct sockaddr_nl sk_addr = { 0 }, rcv_addr = { 0 }; char buf[__UEVENT_BUFFER_SIZE] = { 0 }; @@ -96,13 +95,6 @@ static int uevent_listener(unsigned long post_flags, bool expect_uevent, return -1; } - ret = setsockopt(sk_fd, SOL_SOCKET, SO_RCVBUF, &rcv_buf_sz, - sizeof(rcv_buf_sz)); - if (ret < 0) { - fprintf(stderr, "%s - Failed to set socket options\n", strerror(errno)); - goto on_error; - } - sk_addr.nl_family = AF_NETLINK; sk_addr.nl_groups = __UEVENT_LISTEN_ALL; From f764ae2ff3709d8d38be54b940752546055335df Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Thu, 3 Sep 2026 21:13:12 +0800 Subject: [PATCH 0935/1012] ocfs2: allow xattr bucket entries to span multiple blocks Patch series "ocfs2: xattr bucket validation fixes", v2. This series fixes two problems around xattr bucket validation. Patch 1 fixes a false-corruption failure on blocksize-512 volumes: the bucket validator limited the entry array to the first bucket block while the write path stores entries across the whole 4096-byte bucket region, so a legitimately written, fsck-clean bucket could be rejected and force the filesystem read-only. It also adds an alignment check on the bucket block number, since the entry array is accessed as one contiguous region and a corrupted xattr tree could otherwise point a bucket at blocks straddling a page boundary. Patch 2 converts two mlog_bug_on_msg() checks in the bucket defrag path to ocfs2_error() returns, so that a corrupt bucket holding overlapping entries or an inflated xh_free_start marks the filesystem read-only and fails the setxattr instead of panicking the kernel. Both patches have been tested in QEMU: the blocksize-512 reproducer (40 xattrs with 100-byte values, previously failing with "entry count 32 exceeds maximum 31") now passes with a clean fsck.ocfs2 result, and the ocfs2 testsuite xattr tests pass 48/48 across blocksize combinations. This patch (of 2): ocfs2_validate_xattr_bucket() limits the entry array to the first bucket block, but the write path stores entries across the whole OCFS2_XATTR_BUCKET_SIZE region. With 512-byte blocks a bucket spans eight blocks, and a bucket filled with small xattrs places its last entries past offset 512. Reading such a bucket back errors out: OCFS2: ERROR (device loop0): ocfs2_validate_xattr_bucket: Invalid xattr bucket 86072: entry count 32 exceeds maximum 31 On-disk corruption discovered. Please run fsck.ocfs2 once the filesystem is unmounted. OCFS2: File system is now read-only. This is reproducible by setting ~33 xattrs with 100-byte values on a file on a blocksize-512 volume; fsck.ocfs2 reports the resulting image clean. Check the entry count against the full bucket region instead. The per-block bounds checks for names and values stay as they are, since ocfs2_bucket_align_free_start() keeps each name+value pair within a single block. The entry array is one contiguous region, so a bucket from a corrupted xattr tree whose first block is not aligned to OCFS2_XATTR_BUCKET_SIZE could straddle a page and make the validation loop read out of bounds. Buckets allocated within clusters are always aligned, so reject any other block number while validating. Link: https://lore.kernel.org/20260903131313.2396208-1-joseph.qi@linux.alibaba.com Link: https://lore.kernel.org/20260903131313.2396208-2-joseph.qi@linux.alibaba.com Fixes: 2cf82b46d5e4 ("ocfs2: validate external xattr entries when reading metadata") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao --- fs/ocfs2/xattr.c | 21 ++++++++++++++++++++- 1 file changed, 20 insertions(+), 1 deletion(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index e27f5f925df52e..5f65ae6ebb92bd 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -1150,11 +1150,30 @@ static int ocfs2_validate_xattr_bucket(struct ocfs2_xattr_bucket *bucket, struct ocfs2_xattr_header *xh = bucket_xh(bucket); u16 xattr_count = le16_to_cpu(xh->xh_count); size_t region_size = (size_t)sb->s_blocksize * bucket->bu_blocks; - size_t entries_limit = sb->s_blocksize; + /* + * The entry array grows up from the header across the whole + * bucket region, so it may extend beyond the first bucket block + * when the blocksize is smaller than OCFS2_XATTR_BUCKET_SIZE. + * Name/value pairs, however, always live within a single block. + */ + size_t entries_limit = region_size; size_t nv_limit = sb->s_blocksize; size_t max_entries; int i, ret; + /* + * The entry array is one contiguous region that may span the + * bucket's buffer_heads. Buckets are allocated within clusters, + * so their first block is always aligned to + * OCFS2_XATTR_BUCKET_SIZE and the whole bucket fits in one page. + * A corrupted xattr tree can point a bucket at blocks straddling + * a page, so reject it before touching the entry array. + */ + if (blkno & (bucket->bu_blocks - 1)) + return ocfs2_error(sb, + "Invalid xattr bucket %llu: unaligned block number\n", + (unsigned long long)blkno); + if (region_size < sizeof(*xh)) return ocfs2_error(sb, "Invalid xattr bucket %llu: region size %zu is too small\n", From e55488ca78f6baddabe818587ebe9165a4409663 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Thu, 3 Sep 2026 21:13:13 +0800 Subject: [PATCH 0936/1012] ocfs2: reject inconsistent xattr bucket during defrag ocfs2_defrag_xattr_bucket() has two mlog_bug_on_msg() checks that assume the name/value pairs in a bucket are disjoint and that xh_free_start is not below the compacted region. ocfs2_validate_xattr_bucket() only checks each entry in isolation, so a corrupt bucket holding overlapping entries, or one with an inflated xh_free_start, passes validation and then hits BUG() in defrag when a setxattr triggers it. Defrag works on a linear copy of the bucket and does not touch the real blocks before the copy back, so the checks can return an error instead of calling BUG(). Link: https://lore.kernel.org/20260903131313.2396208-3-joseph.qi@linux.alibaba.com Fixes: 012255961c9e ("ocfs2: Enable xattr set in index btree") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Changwei Ge Cc: Heming Zhao Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh --- fs/ocfs2/xattr.c | 16 +++++++++++----- 1 file changed, 11 insertions(+), 5 deletions(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index 5f65ae6ebb92bd..a2d6c3d0f2e8d8 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -4801,16 +4801,22 @@ static int ocfs2_defrag_xattr_bucket(struct inode *inode, memmove(bucket_buf + end - len, bucket_buf + offset, len); xe->xe_name_offset = cpu_to_le16(end - len); + } else if (end < offset + len) { + ret = ocfs2_error(inode->i_sb, + "Defrag check failed for bucket %llu\n", + (unsigned long long)blkno); + goto out; } - mlog_bug_on_msg(end < offset + len, "Defrag check failed for " - "bucket %llu\n", (unsigned long long)blkno); - end -= len; } - mlog_bug_on_msg(xh_free_start > end, "Defrag check failed for " - "bucket %llu\n", (unsigned long long)blkno); + if (xh_free_start > end) { + ret = ocfs2_error(inode->i_sb, + "Defrag check failed for bucket %llu\n", + (unsigned long long)blkno); + goto out; + } if (xh_free_start == end) goto out; From d4bc6a94ade58d7e887ebb88715452da2410a5b2 Mon Sep 17 00:00:00 2001 From: Hengyu Liang Date: Wed, 2 Sep 2026 13:01:15 -0400 Subject: [PATCH 0937/1012] fat: calculate data area start without overflow On 32-bit architectures, sbi->fat_length, sbi->dir_start and sbi->data_start are unsigned long. The number of FATs is an 8-bit BPB field, while the FAT32 length is a 32-bit BPB field. Therefore, the calculation sbi->fat_start + sbi->fats * sbi->fat_length can wrap before data_start is checked against total_sectors. For example, with fat_start=32, fats=2 and fat_length=0x80000001, the unwrapped data area start is 0x100000022 (4294967330), but the calculation wraps to 34 on i386. With total_sectors=36, the validation then incorrectly passes. The following script creates an image that demonstrates the problem: python3 - <<'PY' import struct S = 512 b = bytearray(36 * S) def p(off, fmt, value): struct.pack_into(fmt, b, off, value) # FAT32 BPB b[0:3] = b'\xeb\x58\x90' b[3:11] = b'MSWIN4.1' p(11, ' Signed-off-by: Andrew Morton Acked-by: OGAWA Hirofumi --- fs/fat/inode.c | 23 +++++++++++++++++------ 1 file changed, 17 insertions(+), 6 deletions(-) diff --git a/fs/fat/inode.c b/fs/fat/inode.c index f775a004cae1e2..0b0bbe777842da 100644 --- a/fs/fat/inode.c +++ b/fs/fat/inode.c @@ -20,6 +20,7 @@ #include #include #include +#include #include #include #include @@ -1577,6 +1578,7 @@ int fat_fill_super(struct super_block *sb, struct fs_context *fc, struct msdos_sb_info *sbi; u16 logical_sector_size; u32 total_sectors, total_clusters, fat_clusters, rootdir_sectors; + u32 dir_start, data_start; long error; char buf[50]; struct timespec64 ts; @@ -1752,7 +1754,6 @@ int fat_fill_super(struct super_block *sb, struct fs_context *fc, sbi->dir_per_block = sb->s_blocksize / sizeof(struct msdos_dir_entry); sbi->dir_per_block_bits = ffs(sbi->dir_per_block) - 1; - sbi->dir_start = sbi->fat_start + sbi->fats * sbi->fat_length; sbi->dir_entries = bpb.fat_dir_entries; if (sbi->dir_entries & (sbi->dir_per_block - 1)) { if (!silent) @@ -1763,20 +1764,30 @@ int fat_fill_super(struct super_block *sb, struct fs_context *fc, rootdir_sectors = sbi->dir_entries * sizeof(struct msdos_dir_entry) / sb->s_blocksize; - sbi->data_start = sbi->dir_start + rootdir_sectors; + if (check_mul_overflow(sbi->fats, sbi->fat_length, &dir_start) || + check_add_overflow(sbi->fat_start, dir_start, &dir_start) || + check_add_overflow(dir_start, rootdir_sectors, &data_start)) { + if (!silent) + fat_msg(sb, KERN_ERR, + "overflow of root dir or data layout"); + goto out_invalid; + } + total_sectors = bpb.fat_sectors; if (total_sectors == 0) total_sectors = bpb.fat_total_sect; - if (total_sectors < sbi->data_start) { + if (total_sectors < data_start) { if (!silent) fat_msg(sb, KERN_ERR, - "data area starts beyond volume (%lu > %u)", - sbi->data_start, total_sectors); + "data area starts beyond volume (%u > %u)", + data_start, total_sectors); goto out_invalid; } - total_clusters = (total_sectors - sbi->data_start) / sbi->sec_per_clus; + sbi->dir_start = dir_start; + sbi->data_start = data_start; + total_clusters = (total_sectors - data_start) / sbi->sec_per_clus; if (!is_fat32(sbi)) sbi->fat_bits = (total_clusters > MAX_FAT12) ? 16 : 12; From 14fb67a87d09261ace8f7780005aa565477d6fb7 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Thu, 3 Sep 2026 18:47:44 +0800 Subject: [PATCH 0938/1012] ocfs2: skip uninitialized lockres in ocfs2_mark_lockres_freeing() A hard readonly mount skips ocfs2_dlm_init(), so the per-osb lock resources are never initialized and osb->cconn stays NULL. Before commit 550842cc60987 ("ocfs2: fix freeing uninitialized resource on ocfs2_dlm_shutdown") ocfs2_dismount_volume() only called ocfs2_dlm_shutdown() when osb->cconn was set. It now calls it unconditionally, so unmounting a hard readonly mount drops the osb locks and takes the never initialized l_lock in ocfs2_mark_lockres_freeing(). With lockdep enabled this triggers: INFO: trying to register non-static key. The code is fine but needs lockdep annotation, or maybe you didn't initialize this object before use? turning off the locking correctness validator. ocfs2_drop_lock() and ocfs2_lock_res_free() already skip lock resources without OCFS2_LOCK_INITIALIZED. Add the same check to ocfs2_mark_lockres_freeing(), which is reachable before them through ocfs2_simple_drop_lockres(), so an uninitialized lockres is never touched. Link: https://lore.kernel.org/20260903104744.2164235-1-joseph.qi@linux.alibaba.com Fixes: 550842cc6098 ("ocfs2: fix freeing uninitialized resource on ocfs2_dlm_shutdown") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reported-by: ZW Tang Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao Cc: --- fs/ocfs2/dlmglue.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/fs/ocfs2/dlmglue.c b/fs/ocfs2/dlmglue.c index a23dd8f86c8950..cf3318b0d3a8c9 100644 --- a/fs/ocfs2/dlmglue.c +++ b/fs/ocfs2/dlmglue.c @@ -3525,6 +3525,10 @@ void ocfs2_mark_lockres_freeing(struct ocfs2_super *osb, struct ocfs2_mask_waiter mw; unsigned long flags, flags2; + /* We didn't get anywhere near actually using this lockres. */ + if (!(lockres->l_flags & OCFS2_LOCK_INITIALIZED)) + return; + ocfs2_init_mask_waiter(&mw); spin_lock_irqsave(&lockres->l_lock, flags); From 9f3814968c9aad06f5fa84917be7bfd32639707c Mon Sep 17 00:00:00 2001 From: Adam Harshbarger Date: Thu, 3 Sep 2026 17:24:56 -0500 Subject: [PATCH 0939/1012] lib/plist: fix plist_requeue() corrupting order in the last bucket plist_requeue() is meant to move a node to the end of its own priority run. When the node heads the *last* priority bucket it is instead placed at the head of the whole list, leaving the plist unsorted: built: A(prio 0) B(prio 1) C(prio 1) requeue(B): B(prio 1) A(prio 0) C(prio 1) expected: A(prio 0) C(prio 1) B(prio 1) prio_list is a *headless* circular ring of the nodes that lead each priority bucket. The shortcut added by commit 95d4b3450ebe ("lib/plist.c: add shortcut for plist_requeue()") takes iter = list_entry(iter->prio_list.next, struct plist_node, prio_list); node_next = &iter->node_list; which from the last bucket wraps round to the *first* bucket, so node_next ends up pointing at the head of the list rather than at its end. The plist_for_each_continue() loop immediately below it computes the correct answer (&head->node_list) for that case. With any bucket after it the shortcut is correct, which is why this went unnoticed: the benchmark in that commit measured elapsed time and never checked the resulting order. Keep the shortcut -- it is a real win -- but exclude the case where iter's bucket is the last one, which is exactly when its ring successor is the first bucket again. Reachable from mm/swapfile.c, which rotates swap_avail_heads[] with plist_requeue(). It takes three or more swap devices: at least two distinct priorities, so that a later bucket exists for the ring to wrap round from, and two or more devices sharing the lowest priority, so that plist_requeue() does not return early. One device per priority returns early at the node->prio != iter->prio test. A single priority is also safe, but for a different reason worth stating: with one bucket no node is ever linked onto prio_list at all -- plist_add() skips it for the first node and for every node whose predecessor shares its priority -- so list_empty(&iter->prio_list) holds and the shortcut is never entered. Tested by driving three implementations -- the pre-95d4b3450ebe code, current mainline, and this patch -- through 1,084,492 identical random add/del/requeue operations over 24 nodes and 1..5 distinct priorities, comparing the resulting node_list node for node after every operation: variant differs from pre-95d4b3450ebe left list unsorted pre-95d4b3450ebe -- (reference) 0 mainline 289,297 276,660 this patch 0 0 Link: https://lore.kernel.org/20260903222456.1881786-1-handyhandyman.adam@gmail.com Fixes: 95d4b3450ebe ("lib/plist.c: add shortcut for plist_requeue()") Signed-off-by: Adam Harshbarger Signed-off-by: Andrew Morton Assisted-by: Claude:claude-opus-5 Cc: I Hsin Cheng Cc: # v6.15+ --- lib/plist.c | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/lib/plist.c b/lib/plist.c index a5bef38add431d..b0273533c04d2a 100644 --- a/lib/plist.c +++ b/lib/plist.c @@ -174,8 +174,15 @@ void plist_requeue(struct plist_node *node, struct plist_head *head) /* * After plist_del(), iter is the replacement of the node. If the node * was on prio_list, take shortcut to find node_next instead of looping. + * + * prio_list is a headless ring, so from the LAST bucket ->next wraps + * round to the first one; in that case node_next is the list head. */ if (!list_empty(&iter->prio_list)) { + struct plist_node *first = plist_first(head); + + if (iter->prio_list.next == &first->prio_list) + goto queue; iter = list_entry(iter->prio_list.next, struct plist_node, prio_list); node_next = &iter->node_list; From fb185ac7933b143f6b51ee1bd9271a9474df00ec Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ali=20Ahmet=20Memi=C5=9F?= Date: Sat, 5 Sep 2026 22:49:39 +0300 Subject: [PATCH 0940/1012] =?UTF-8?q?mailmap:=20update=20email=20address?= =?UTF-8?q?=20for=20Ali=20Ahmet=20Memi=C5=9F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit I'm switching to my disroot address for kernel contributions. Map the address my earlier patches were sent from to it, so that get_maintainer.pl stops offering the old one as a recipient. Also switch to the proper spelling of my name with diacritics, which is what I use with the new address. Link: https://lore.kernel.org/20260905195000.556185-1-aliamemis@disroot.org Signed-off-by: Ali Ahmet Memiş Signed-off-by: Andrew Morton --- .mailmap | 1 + 1 file changed, 1 insertion(+) diff --git a/.mailmap b/.mailmap index 3940b0a12c2820..b399a10744fc08 100644 --- a/.mailmap +++ b/.mailmap @@ -67,6 +67,7 @@ Alex Hung Alex Shi Alex Shi Alex Shi +Ali Ahmet Memiş Alice Mikityanska Alice Mikityanska Alice Mikityanska From 0e2fb64bf0fed83165836f39c5052cd39feea0f9 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Sat, 5 Sep 2026 22:21:43 +0800 Subject: [PATCH 0941/1012] ocfs2: validate dr_fs_generation of dir index root blocks ocfs2_validate_dx_root() does not verify dr_fs_generation against the superblock generation, unlike the extent and xattr block validators which check h_fs_generation and xb_fs_generation respectively. The field is documented as "Must match super block". Without the check, a stale dir index root block left on the device from a previously formatted filesystem at the same physical block number can pass validation as long as its signature, dr_blkno and checksum match. Its index entries and suballocator information would then be used in the new filesystem context. Reject dir index root blocks whose dr_fs_generation does not match the mounted filesystem, like the extent and xattr block validators do. Link: https://lore.kernel.org/20260905142144.2869105-1-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reviewed-by: Heming Zhao Cc: Changwei Ge Cc: Joel Becker Cc: Jun Piao Cc: Junxiao Bi Cc: Mark Fasheh --- fs/ocfs2/dir.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/fs/ocfs2/dir.c b/fs/ocfs2/dir.c index 6bb6aa133f0150..329680b4622739 100644 --- a/fs/ocfs2/dir.c +++ b/fs/ocfs2/dir.c @@ -613,6 +613,14 @@ static int ocfs2_validate_dx_root(struct super_block *sb, goto bail; } + if (le32_to_cpu(dx_root->dr_fs_generation) != OCFS2_SB(sb)->fs_generation) { + ret = ocfs2_error(sb, + "Dir Index Root # %llu has an invalid dr_fs_generation of #%u\n", + (unsigned long long)bh->b_blocknr, + le32_to_cpu(dx_root->dr_fs_generation)); + goto bail; + } + /* * Dir index root blocks are allocated from a per-slot suballocator, * so the slot must be in range. Otherwise removing the index passes From 97432f2c24b5fd33389115d616b6d8e452ee27ec Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Sat, 5 Sep 2026 22:21:44 +0800 Subject: [PATCH 0942/1012] ocfs2: validate dl_blkno and dl_fs_generation of dir index leaf blocks ocfs2_validate_dx_leaf() checks the checksum, the signature and the entry list counts, but it never checks dl_blkno or dl_fs_generation. The inode, extent block, xattr block, refcount block and dir index root validators all check the on-disk block number against bh->b_blocknr and the generation against the superblock, and both dir index leaf fields are documented as "Must match super block". Without the checks, a stale dir index leaf block left on the device from a previously formatted filesystem at the same physical block number can pass validation as long as its signature, entry counts and checksum match. Its index entries would then be used in the new filesystem context. Both fields are written unconditionally when a leaf block is formatted in ocfs2_dx_dir_format_cluster(), from the live superblock generation and the real block number, so a correctly formatted filesystem cannot trip the new checks. The leaf block number read back here comes from on-disk dir index root extent records. Reject dir index leaf blocks whose dl_blkno or dl_fs_generation does not match, like the dir index root validator does. Link: https://lore.kernel.org/20260905142144.2869105-2-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reviewed-by: Heming Zhao Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao --- fs/ocfs2/dir.c | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/fs/ocfs2/dir.c b/fs/ocfs2/dir.c index 329680b4622739..55c4a305a2823e 100644 --- a/fs/ocfs2/dir.c +++ b/fs/ocfs2/dir.c @@ -733,6 +733,18 @@ static int ocfs2_validate_dx_leaf(struct super_block *sb, return ocfs2_error(sb, "Dir Index Leaf has bad signature %.*s\n", 7, dx_leaf->dl_signature); + if (le64_to_cpu(dx_leaf->dl_blkno) != bh->b_blocknr) + return ocfs2_error(sb, + "Dir Index Leaf # %llu has an invalid dl_blkno of %llu\n", + (unsigned long long)bh->b_blocknr, + (unsigned long long)le64_to_cpu(dx_leaf->dl_blkno)); + + if (le32_to_cpu(dx_leaf->dl_fs_generation) != OCFS2_SB(sb)->fs_generation) + return ocfs2_error(sb, + "Dir Index Leaf # %llu has an invalid dl_fs_generation of #%u\n", + (unsigned long long)bh->b_blocknr, + le32_to_cpu(dx_leaf->dl_fs_generation)); + if (le16_to_cpu(dx_leaf->dl_list.de_count) != ocfs2_dx_entries_per_leaf(sb)) return ocfs2_error(sb, From 4f066318f5444054cda15e0bb22bbff300452496 Mon Sep 17 00:00:00 2001 From: Petr Vorel Date: Fri, 4 Sep 2026 13:02:27 +0200 Subject: [PATCH 0943/1012] checkpatch: add more userspace directories to is_userspace() Patch series "checkpatch: userspace improvements", v6. Few improvements for user space code + --userspace option for projects which vendored checkpatch.pl. There could be probably more checks which are kernel space only. This patch (of 4): arch/ directory contains subdirectories with userspace tools (at least arch/*/tools/ and arch/*/boot/tools/). Add check to consider any arch/.*/tools/ subdirectory as userspace tools directory. This helps not only to strscpy() checks but also to CamelCase checks in the next commit to be more precise. This is a follow-up to 99b70ece33d8 ("checkpatch: suppress strscpy warnings for userspace tools"). Link: https://lore.kernel.org/20260904110230.1219037-1-pvorel@suse.cz Link: https://lore.kernel.org/20260904110230.1219037-2-pvorel@suse.cz Signed-off-by: Petr Vorel Signed-off-by: Andrew Morton Cc: Andy Whitcroft Cc: Dwaipayan Ray Cc: Joe Perches Cc: Lukas Bulwahn --- scripts/checkpatch.pl | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index f424dafce5bce7..b8702f6bc9b572 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -2667,7 +2667,9 @@ sub exclude_global_initialisers { sub is_userspace { my ($realfile) = @_; - return ($realfile =~ m@^tools/@ || $realfile =~ m@^scripts/@); + return ($realfile =~ m@^tools/@ || + $realfile =~ m@^scripts/@ || + $realfile =~ m@^arch/.*/tools/@); } sub process { From def86cc01ad9262433117280cee26213f392dc1e Mon Sep 17 00:00:00 2001 From: Petr Vorel Date: Fri, 4 Sep 2026 13:02:28 +0200 Subject: [PATCH 0944/1012] checkpatch: ignore format macros for userspace tools Constants from are used only in userspace tools, they are from ISO C99, let's don't report it: arch/mips/boot/tools/relocs.c:572: CHECK: Avoid CamelCase: arch/s390/tools/relocs.c:52: CHECK: Avoid CamelCase: tools/testing/selftests/mm/vm_util.c:244: CHECK: Avoid CamelCase: Link: https://lore.kernel.org/20260904110230.1219037-3-pvorel@suse.cz Signed-off-by: Petr Vorel Signed-off-by: Andrew Morton Cc: Andy Whitcroft Cc: Dwaipayan Ray Cc: Joe Perches Cc: Lukas Bulwahn --- scripts/checkpatch.pl | 2 ++ 1 file changed, 2 insertions(+) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index b8702f6bc9b572..b458c7f2268484 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -5950,6 +5950,8 @@ sub process { #Ignore SI style variants like nS, mV and dB #(ie: max_uV, regulator_min_uA_show, RANGE_mA_VALUE) $var !~ /^(?:[a-z0-9_]*|[A-Z0-9_]*)?_?[a-z][A-Z](?:_[a-z0-9_]+|_[A-Z0-9_]+)?$/ && +#Ignore format macros (e.g. PRIu64, SCNu64) + (is_userspace($realfile) ? $var !~ /^(?:PRI|SCN)[dioux][A-Z0-9]+$/ : 1) && #Ignore some three character SI units explicitly, like MiB and KHz $var !~ /^(?:[a-z_]*?)_?(?:[KMGT]iB|[KMGT]?Hz)(?:_[a-z_]+)?$/) { while ($var =~ m{\b($Ident)}g) { From 44c9d58ec491487e902946aaeae7a5d7cbbf1b3d Mon Sep 17 00:00:00 2001 From: Petr Vorel Date: Fri, 4 Sep 2026 13:02:29 +0200 Subject: [PATCH 0945/1012] checkpatch: add --userspace to force userspace rules Also allow to use --no-userspace for userspace projects which vendored checkpatch.pl and use --userspace globally to be able switch it off for files with kernel code. Link: https://lore.kernel.org/20260904110230.1219037-4-pvorel@suse.cz Signed-off-by: Petr Vorel Signed-off-by: Andrew Morton Cc: Andy Whitcroft Cc: Dwaipayan Ray Cc: Joe Perches Cc: Lukas Bulwahn --- scripts/checkpatch.pl | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index b458c7f2268484..ead35e6abba7ba 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -63,6 +63,7 @@ my $max_line_length = 100; my $ignore_perl_version = 0; my $spdx_cxx_comments = 0; +my $userspace; my $minimum_perl_version = 5.10.0; my $min_conf_desc_length = 4; my $spelling_file = "$D/spelling.txt"; @@ -143,6 +144,7 @@ sub help { (required by old toolchains), allow also C++ comments (//). NOTE: it should *not* be used for Linux mainline. + --userspace Force rules specific for userspace. --codespell Use the codespell dictionary for spelling/typos (default:$codespellfile) --codespellfile Use this codespell dictionary @@ -358,6 +360,7 @@ sub load_docs { 'codespell!' => \$codespell, 'codespellfile=s' => \$user_codespellfile, 'typedefsfile=s' => \$typedefsfile, + 'userspace!' => \$userspace, 'color=s' => \$color, 'no-color' => \$color, #keep old behaviors of -nocolor 'nocolor' => \$color, #keep old behaviors of -nocolor @@ -2667,6 +2670,9 @@ sub exclude_global_initialisers { sub is_userspace { my ($realfile) = @_; + + return $userspace if (defined $userspace); + return ($realfile =~ m@^tools/@ || $realfile =~ m@^scripts/@ || $realfile =~ m@^arch/.*/tools/@); From fe6fe78861e348dcb1b3aac7ca813622e4077913 Mon Sep 17 00:00:00 2001 From: Petr Vorel Date: Fri, 4 Sep 2026 13:02:30 +0200 Subject: [PATCH 0946/1012] checkpatch: skip kernel specific checks for userspace These check are kernel specific, do not warn about it when testing userspace code: * BIT_MACRO * LONG_UDELAY * MSLEEP * PREFER_KERNEL_TYPES * USLEEP_RANGE This is a follow-up to 99b70ece33d8 ("checkpatch: suppress strscpy warnings for userspace tools"). Link: https://lore.kernel.org/20260904110230.1219037-5-pvorel@suse.cz Signed-off-by: Petr Vorel Signed-off-by: Andrew Morton Cc: Andy Whitcroft Cc: Dwaipayan Ray Cc: Joe Perches Cc: Lukas Bulwahn --- scripts/checkpatch.pl | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index ead35e6abba7ba..3614cfe4dcbb45 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -6696,7 +6696,8 @@ sub process { } # prefer usleep_range over udelay - if ($line =~ /\budelay\s*\(\s*(\d+)\s*\)/) { + if (!is_userspace($realfile) && + $line =~ /\budelay\s*\(\s*(\d+)\s*\)/) { my $delay = $1; # ignore udelay's < 10, however if (! ($delay < 10) ) { @@ -6710,7 +6711,8 @@ sub process { } # warn about unexpectedly long msleep's - if ($line =~ /\bmsleep\s*\((\d+)\);/) { + if (!is_userspace($realfile) && + $line =~ /\bmsleep\s*\((\d+)\);/) { if ($1 < 20) { WARN("MSLEEP", "msleep < 20ms can sleep for up to 20ms; see function description of msleep().\n" . $herecurr); @@ -6932,7 +6934,7 @@ sub process { # check for c99 types like uint8_t used outside of uapi/ and tools/ if ($realfile !~ m@\binclude/uapi/@ && - $realfile !~ m@\btools/@ && + !is_userspace($realfile) && $line =~ /\b($Declare)\s*$Ident\s*[=;,\[]/) { my $type = $1; if ($type =~ /\b($typeC99Typedefs)\b/) { @@ -7182,6 +7184,7 @@ sub process { # check usleep_range arguments if ($perl_version_ok && defined $stat && + !is_userspace($realfile) && $stat =~ /^\+(?:.*?)\busleep_range\s*\(\s*($FuncArg)\s*,\s*($FuncArg)\s*\)/) { my $min = $1; my $max = $7; @@ -7425,6 +7428,7 @@ sub process { # check for #defines like: 1 << that could be BIT(digit), it is not exported to uapi if ($realfile !~ m@^include/uapi/@ && + !is_userspace($realfile) && $line =~ /#\s*define\s+\w+\s+\(?\s*1\s*([ulUL]*)\s*\<\<\s*(?:\d+|$Ident)\s*\)?/) { my $ull = ""; $ull = "_ULL" if (defined($1) && $1 =~ /ll/i); From 9d68cf02f187c90d532a91c9b32decb49d40baa0 Mon Sep 17 00:00:00 2001 From: Kazuki Hanai Date: Fri, 28 Aug 2026 00:25:16 +0900 Subject: [PATCH 0947/1012] tmpfs: fix unicode_map leaks in casefold option handling MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit shmem_parse_opt_casefold() stores the unicode_map returned by utf8_load() in ctx->encoding. The casefold parameter can be supplied more than once for the same filesystem context, but replacing the stored map does not release the previous reference. The final reference is also leaked when an unmounted filesystem context is freed. Release the previous map before replacing it, clear ctx->encoding after transferring ownership to the superblock, and release any remaining reference from shmem_free_fc(). An unprivileged user can repeatedly set the casefold parameter on a tmpfs filesystem context from a user namespace. This causes unbounded kernel memory consumption and can result in a local denial of service. Link: https://lore.kernel.org/20260827152516.805622-1-hnkz.64@gmail.com Fixes: 58e55efd6c72 ("tmpfs: Add casefold lookup support") Signed-off-by: Kazuki Hanai Signed-off-by: Andrew Morton Reviewed-by: Andrew Morton Cc: Baolin Wang Cc: Hugh Dickins Cc: André Almeida Cc: Christian Brauner Cc: --- mm/shmem.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/mm/shmem.c b/mm/shmem.c index 848316eaa7f4fb..c6be5961956256 100644 --- a/mm/shmem.c +++ b/mm/shmem.c @@ -4530,6 +4530,7 @@ static int shmem_parse_opt_casefold(struct fs_context *fc, struct fs_parameter * pr_info("tmpfs: Using encoding : utf8-%u.%u.%u\n", unicode_major(version), unicode_minor(version), unicode_rev(version)); + utf8_unload(ctx->encoding); ctx->encoding = encoding; return 0; @@ -4998,6 +4999,7 @@ static int shmem_fill_super(struct super_block *sb, struct fs_context *fc) if (ctx->encoding) { sb->s_encoding = ctx->encoding; + ctx->encoding = NULL; set_default_d_op(sb, &shmem_ci_dentry_ops); if (ctx->strict_encoding) sb->s_encoding_flags = SB_ENC_STRICT_MODE_FL; @@ -5095,6 +5097,9 @@ static void shmem_free_fc(struct fs_context *fc) struct shmem_options *ctx = fc->fs_private; if (ctx) { +#if IS_ENABLED(CONFIG_UNICODE) + utf8_unload(ctx->encoding); +#endif mpol_put(ctx->mpol); kfree(ctx); } From 58df99fb5e9e395c63e596abc5e02bc73d58769a Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Fri, 11 Sep 2026 01:30:48 +0900 Subject: [PATCH 0948/1012] bootconfig: remove redundant assignment in xbc_parse_array() After the loop, node is the last array value that xbc_add_child() added, and xbc_parse_array() sets node->child to 0. xbc_init_node() already set node->child to 0 when the value was added, and node->child is not modified after that, so the assignment is redundant. So remove the assignment. No functional change. Link: https://lore.kernel.org/20260910163051.1973775-1-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton Acked-by: Masami Hiramatsu --- lib/bootconfig.c | 1 - 1 file changed, 1 deletion(-) diff --git a/lib/bootconfig.c b/lib/bootconfig.c index 89c88e359179f0..f248b5770291cb 100644 --- a/lib/bootconfig.c +++ b/lib/bootconfig.c @@ -836,7 +836,6 @@ static int __init xbc_parse_array(char **__v) return -ENOMEM; *__v = next; } while (c == ','); - node->child = 0; return c; } From 85e8e0fd6042059aa0c95e6bdd0d8b13e04f5726 Mon Sep 17 00:00:00 2001 From: "Masami Hiramatsu (Google)" Date: Fri, 11 Sep 2026 23:13:06 +0900 Subject: [PATCH 0949/1012] bootconfig: reject unexpected data after null character Patch series "bootconfig: Reject unexpected data after null character and cleanups", v2. Make bootconfig reject unexpected config data after null character and implement other cleanups including tools/bootconfig to consolidate bootconfig initialization with errors, and to skip internal tree sanity checks in kernel. This patch (of 4): If a bootconfig buffer contains an intermediate null character in the middle of the configuration, xbc_parse_tree() stops at the null character because string delimiter searches (e.g. strpbrk()) stop at '\0', and cleanly breaks out of the loop without error. As a result, any configuration data following the intermediate null character is silently ignored, allowing unparsed or potentially malicious data to be hidden after an early termination. Fix this in xbc_parse_tree() by checking that no non-null data remains between the parser termination point and the end of the input buffer. Trailing null characters (such as alignment padding in initrd) continue to be accepted as valid. Also update apply_xbc() in tools/bootconfig/main.c to calculate the buffer size based on the loaded file size rather than strlen(), so that files with intermediate null characters are not truncated before validation. Link: https://lore.kernel.org/178913597653.248794.1237187523153227751.stgit@devnote2 Link: https://lore.kernel.org/178913598628.248794.1048773774471250986.stgit@devnote2 Signed-off-by: Masami Hiramatsu (Google) Signed-off-by: Andrew Morton Reviewed-by: Sang-Heon Jeon Assisted-by: Antigravity:gemini-3.8-flash --- lib/bootconfig.c | 7 +++++++ tools/bootconfig/main.c | 4 +++- tools/bootconfig/test-bootconfig.sh | 12 ++++++++++++ 3 files changed, 22 insertions(+), 1 deletion(-) diff --git a/lib/bootconfig.c b/lib/bootconfig.c index f248b5770291cb..b6adf8b274746c 100644 --- a/lib/bootconfig.c +++ b/lib/bootconfig.c @@ -1118,6 +1118,13 @@ static int __init xbc_parse_tree(void) } } while (!ret); + if (!ret) { + while (p < xbc_data + xbc_data_size - 1 && *p == '\0') + p++; + if (p < xbc_data + xbc_data_size - 1) + ret = xbc_parse_error("Unexpected data after null character", p); + } + return ret; } diff --git a/tools/bootconfig/main.c b/tools/bootconfig/main.c index 17d971d47f8791..aff169ba75b828 100644 --- a/tools/bootconfig/main.c +++ b/tools/bootconfig/main.c @@ -433,7 +433,9 @@ static int apply_xbc(const char *path, const char *xbc_path) pr_err("Failed to load %s : %d\n", xbc_path, ret); return ret; } - size = strlen(buf) + 1; + size = ret; + if (size == 0 || buf[size - 1] != '\0') + size++; csum = xbc_calc_checksum(buf, size); /* Backup the bootconfig data */ diff --git a/tools/bootconfig/test-bootconfig.sh b/tools/bootconfig/test-bootconfig.sh index fc69f815ce4af0..530ce7e28d634e 100755 --- a/tools/bootconfig/test-bootconfig.sh +++ b/tools/bootconfig/test-bootconfig.sh @@ -180,6 +180,18 @@ EOF $BOOTCONF -a $TEMPCONF $INITRD 2> $OUTFILE xpass grep -q "1:1" $OUTFILE +echo "Intermediate null character test" +printf "key = value\n\0extra = data\n" > $TEMPCONF +xfail $BOOTCONF -a $TEMPCONF $INITRD +$BOOTCONF -a $TEMPCONF $INITRD 2> $OUTFILE +xpass grep -q "Unexpected" $OUTFILE + +echo "Trailing null character test" +printf "key = value\n\0" > $TEMPCONF +xpass $BOOTCONF -a $TEMPCONF $INITRD +$BOOTCONF $INITRD > $OUTFILE +xpass grep -q "value" $OUTFILE + echo "=== expected failure cases ===" for i in samples/bad-* ; do xfail $BOOTCONF -a $i $INITRD From 0f322f241a62012c1f6a7c0ddfc18bb1fa715350 Mon Sep 17 00:00:00 2001 From: "Masami Hiramatsu (Google)" Date: Fri, 11 Sep 2026 23:13:15 +0900 Subject: [PATCH 0950/1012] tools/bootconfig: consolidate xbc_init() to error message wrapper Use init_xbc_with_error() for all bootconfig initialization in the bootconfig tool instead of showing errors in different way. This simplifies the code logic and make it easy to maintain. Link: https://lore.kernel.org/178913599508.248794.10388592925402087434.stgit@devnote2 Signed-off-by: Masami Hiramatsu (Google) Signed-off-by: Andrew Morton Reviewed-by: Sang-Heon Jeon --- tools/bootconfig/main.c | 100 +++++++++++++++++----------------------- 1 file changed, 43 insertions(+), 57 deletions(-) diff --git a/tools/bootconfig/main.c b/tools/bootconfig/main.c index aff169ba75b828..652e491b9c338f 100644 --- a/tools/bootconfig/main.c +++ b/tools/bootconfig/main.c @@ -21,6 +21,39 @@ #define BOOTCONFIG_FOOTER_SIZE \ (sizeof(uint32_t) * 2 + BOOTCONFIG_MAGIC_LEN) +static void show_xbc_error(const char *data, const char *msg, int pos) +{ + int lin = 1, col, i; + + if (pos < 0) { + pr_err("Error: %s.\n", msg); + return; + } + + /* Note that pos starts from 0 but lin and col should start from 1. */ + col = pos + 1; + for (i = 0; i < pos; i++) { + if (data[i] == '\n') { + lin++; + col = pos - i; + } + } + pr_err("Parse Error: %s at %d:%d\n", msg, lin, col); + +} + +static int init_xbc_with_error(char *buf, int len) +{ + const char *msg; + int ret, pos; + + ret = xbc_init(buf, len, &msg, &pos); + if (ret < 0) + show_xbc_error(buf, msg, pos); + + return ret; +} + static int xbc_show_value(struct xbc_node *node, bool semicolon) { const char *val, *eol; @@ -197,7 +230,6 @@ static int load_xbc_from_initrd(int fd, char **buf) int ret; uint32_t size = 0, csum = 0, rcsum; char magic[BOOTCONFIG_MAGIC_LEN]; - const char *msg; ret = fstat(fd, &stat); if (ret < 0) @@ -249,52 +281,9 @@ static int load_xbc_from_initrd(int fd, char **buf) return -EINVAL; } - ret = xbc_init(*buf, size, &msg, NULL); - /* Wrong data */ - if (ret < 0) { - pr_err("parse error: %s.\n", msg); - return ret; - } - - return size; -} - -static void show_xbc_error(const char *data, const char *msg, int pos) -{ - int lin = 1, col, i; - - if (pos < 0) { - pr_err("Error: %s.\n", msg); - return; - } - - /* Note that pos starts from 0 but lin and col should start from 1. */ - col = pos + 1; - for (i = 0; i < pos; i++) { - if (data[i] == '\n') { - lin++; - col = pos - i; - } - } - pr_err("Parse Error: %s at %d:%d\n", msg, lin, col); + ret = init_xbc_with_error(*buf, size); -} - -static int init_xbc_with_error(char *buf, int len) -{ - char *copy = strdup(buf); - const char *msg; - int ret, pos; - - if (!copy) - return -ENOMEM; - - ret = xbc_init(buf, len, &msg, &pos); - if (ret < 0) - show_xbc_error(copy, msg, pos); - free(copy); - - return ret; + return ret < 0 ? ret : size; } static int show_xbc_kernel_cmdline(void) @@ -423,9 +412,8 @@ static int apply_xbc(const char *path, const char *xbc_path) char *buf, *data; size_t total_size; struct stat stat; - const char *msg; uint32_t size, csum; - int pos, pad; + int pad; int ret, fd; ret = load_xbc_file(xbc_path, &buf); @@ -438,6 +426,13 @@ static int apply_xbc(const char *path, const char *xbc_path) size++; csum = xbc_calc_checksum(buf, size); + /* Verify the data format */ + ret = init_xbc_with_error(buf, size); + if (ret < 0) { + free(buf); + return ret; + } + /* Backup the bootconfig data */ data = calloc(size + BOOTCONFIG_ALIGN + BOOTCONFIG_FOOTER_SIZE, 1); if (!data) { @@ -446,15 +441,6 @@ static int apply_xbc(const char *path, const char *xbc_path) } memcpy(data, buf, size); - /* Check the data format */ - ret = xbc_init(buf, size, &msg, &pos); - if (ret < 0) { - show_xbc_error(data, msg, pos); - free(data); - free(buf); - - return ret; - } printf("Apply %s to %s\n", xbc_path, path); xbc_get_info(&ret, NULL); printf("\tNumber of nodes: %d\n", ret); From e132f2bb64e6e27c4038a3916d564868a9b1c672 Mon Sep 17 00:00:00 2001 From: "Masami Hiramatsu (Google)" Date: Fri, 11 Sep 2026 23:13:24 +0900 Subject: [PATCH 0951/1012] bootconfig: skip internal tree sanity checks in kernel In xbc_verify_tree(), the loop iterating through all nodes to check that xbc_nodes[i].next < xbc_node_num and xbc_nodes[i].child < xbc_node_num is a defensive sanity check against implementation regressions (such an out-of-bounds index cannot be produced by malformed input). Running this check in the kernel adds unnecessary boot-time overhead. Split this check out into xbc_sanity_check_tree() for userspace, so that it continues to run during userspace bootconfig validation (e.g. when applying or testing bootconfig with tools/bootconfig), but is omitted in the kernel to speed up initialization. Link: https://lore.kernel.org/178913600409.248794.12941798680046327623.stgit@devnote2 Signed-off-by: Masami Hiramatsu (Google) Signed-off-by: Andrew Morton Reported-by: Sang-Heon Jeon Closes: https://lore.kernel.org/all/20260905141637.1547429-1-ekffu200098@gmail.com/ Reviewed-by: Sang-Heon Jeon --- lib/bootconfig.c | 38 ++++++++++++++++++++++++++------------ 1 file changed, 26 insertions(+), 12 deletions(-) diff --git a/lib/bootconfig.c b/lib/bootconfig.c index b6adf8b274746c..192a60a9f833cb 100644 --- a/lib/bootconfig.c +++ b/lib/bootconfig.c @@ -1002,9 +1002,30 @@ static int __init xbc_close_brace(char **k, char *n) return __xbc_close_brace(n - 1); } +#ifndef __KERNEL__ +/* Sanity check for regression: node indices must be within bounds */ +static int __init xbc_sanity_check_tree(void) +{ + int i; + + for (i = 0; i < xbc_node_num; i++) { + if (xbc_nodes[i].next >= xbc_node_num) { + return xbc_parse_error("No closing brace", + xbc_node_get_data(xbc_nodes + i)); + } + if (xbc_nodes[i].child >= xbc_node_num) { + return xbc_parse_error("Broken child node", + xbc_node_get_data(xbc_nodes + i)); + } + } + + return 0; +} +#endif + static int __init xbc_verify_tree(void) { - int i, depth; + int depth; size_t len, wlen; struct xbc_node *n, *m; @@ -1021,17 +1042,6 @@ static int __init xbc_verify_tree(void) return -ENOENT; } - for (i = 0; i < xbc_node_num; i++) { - if (xbc_nodes[i].next >= xbc_node_num) { - return xbc_parse_error("No closing brace", - xbc_node_get_data(xbc_nodes + i)); - } - if (xbc_nodes[i].child >= xbc_node_num) { - return xbc_parse_error("Broken child node", - xbc_node_get_data(xbc_nodes + i)); - } - } - /* Key tree limitation check */ n = &xbc_nodes[0]; depth = 1; @@ -1202,6 +1212,10 @@ int __init xbc_init(const char *data, size_t size, const char **emsg, int *epos) ret = xbc_parse_tree(); if (!ret) ret = xbc_verify_tree(); +#ifndef __KERNEL__ + if (!ret) + ret = xbc_sanity_check_tree(); +#endif if (ret < 0) { if (epos) From 0793e3131266ffd7cb4037719610d90152851843 Mon Sep 17 00:00:00 2001 From: Hemanth Selam Date: Mon, 7 Sep 2026 14:43:24 +0530 Subject: [PATCH 0952/1012] scripts/spelling.txt: keep British spelling for "initalise" These are the only two entries in the file that map a British spelling to an American one: initalised||initialized initalise||initialize doc-guide/checkpatch.rst asks that British and American spellings be left alone, so correcting the missing "i" should not also change the variety of English. Following checkpatch here silently rewrites "initalised" to "initialized" in files that otherwise use British forms; arch/arm64 alone has 58 uses of "initialised" against 132 of "initialized", so both are clearly in use. Map them to the British forms instead, so the misspelling is still caught while the spelling variety is left to the author. Link: https://lore.kernel.org/20260907091324.32211-1-hemanth.selam@gmail.com Signed-off-by: Hemanth Selam Signed-off-by: Andrew Morton Suggested-by: Randy Dunlap Acked-by: Randy Dunlap Assisted-by: Cursor:claude-opus-5 Cc: Colin Ian King Cc: Joe Perches Cc: Jonathan Corbet --- scripts/spelling.txt | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scripts/spelling.txt b/scripts/spelling.txt index 3372873cd7bb7d..9ec2529bc58472 100644 --- a/scripts/spelling.txt +++ b/scripts/spelling.txt @@ -872,8 +872,8 @@ infromation||information ingore||ignore inheritence||inheritance inital||initial -initalised||initialized -initalise||initialize +initalised||initialised +initalise||initialise initalized||initialized initalize||initialize initation||initiation From a48bbcbfd1c70d08530b2268ea7a8f5d303a3d35 Mon Sep 17 00:00:00 2001 From: Matt Turner Date: Sat, 12 Sep 2026 14:17:35 -0400 Subject: [PATCH 0953/1012] lib/decompress_bunzip2: fix off-by-one in run-length bounds check The run-length path rejects a block when dbufCount+t equals dbufSize, but the loop that follows writes exactly t bytes starting at dbufCount, so a block that fills the buffer exactly is legal. bzip2 allows it too: its decompressor bounds a block at 100000 * blockSize100k and checks that limit per byte appended. Use > instead of >=. bzip2's encoder stops filling a block 19 bytes early, so nothing it produces ever reaches the limit and the bug stays hidden. Compressors that use the full block size do reach it: an lbzip2 -9 image whose block ends on a run fails to decode, and a self-extracting kernel built that way does not boot. This code came from busybox, which fixed the same line in 2013 in commit 932e233a491b ("bunzip2: fix off-by-one check"). I ran into this because my system uses lbzip2 as /bin/bzip2 -- a common thing on Gentoo I believe. As far as I can tell, anyone using lbzip2 as their system bzip2 would run into this and it's only because it's very uncommon these days to compress a kernel with bzip2 that no one has noticed. I only noticed because I was adding support for various compression formats on alpha. Link: https://lore.kernel.org/20260912-b4-bunzip2-blocksize-fix-v1-1-c7384bbfc954@gmail.com Fixes: bc22c17e12c1 ("bzip2/lzma: library support for gzip, bzip2 and lzma decompression") Signed-off-by: Matt Turner Signed-off-by: Andrew Morton Cc: Alain Knaff Cc: "H. Peter Anvin" Cc: --- lib/decompress_bunzip2.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/lib/decompress_bunzip2.c b/lib/decompress_bunzip2.c index 1288f146661f1c..aaa75404250e2d 100644 --- a/lib/decompress_bunzip2.c +++ b/lib/decompress_bunzip2.c @@ -439,7 +439,7 @@ static int INIT get_next_block(struct bunzip_data *bd) array.) */ if (runPos) { runPos = 0; - if (dbufCount+t >= dbufSize) + if (dbufCount+t > dbufSize) return RETVAL_DATA_ERROR; uc = symToByte[mtfSymbol[0]]; From 27df13e78f0cd49b3e9e62b0fee744ed777774c8 Mon Sep 17 00:00:00 2001 From: Andrey Golovko Date: Sat, 12 Sep 2026 19:26:30 +0300 Subject: [PATCH 0954/1012] mailmap: update email address for Andrey Golovko Patches sent from andrey.golovko@gmail.com carry the authorship of my commits, but some of the tags I gave on other people's patches use andrey@golovko.me. Map the second address onto the first so the two do not look like two different contributors. Link: https://lore.kernel.org/20260912162630.7731-1-andrey.golovko@gmail.com Signed-off-by: Andrey Golovko Signed-off-by: Andrew Morton --- .mailmap | 1 + 1 file changed, 1 insertion(+) diff --git a/.mailmap b/.mailmap index b399a10744fc08..1a288972949802 100644 --- a/.mailmap +++ b/.mailmap @@ -92,6 +92,7 @@ Andrew Morton Andrew Murray Andrew Murray Andrew Vasquez +Andrey Golovko Andrey Konovalov Andrey Ryabinin Andrey Ryabinin From 0bc3a48508478e9121b2c72f17363fbe095bb39f Mon Sep 17 00:00:00 2001 From: James Kim Date: Mon, 14 Sep 2026 13:23:27 +0900 Subject: [PATCH 0955/1012] rapidio: mport_cdev: fix use-after-free in mport_mm_close() A use-after-free vulnerability was identified in mport_mm_close() in drivers/rapidio/devices/rio_mport_cdev.c, identical to the pattern previously addressed in dma_req_free(). This is observable from userspace when an application creates an mmap mapping via the RapidIO character device and subsequently unmaps it (or terminates, triggering exit_mmap()). During munmap, mport_mm_close() is invoked and drops the mapping reference via kref_put(). If kref_put() drops the last reference, mport_release_mapping() is called, which frees the underlying rio_mport_mapping structure. The subsequent mutex_unlock() then dereferences map->md to unlock buf_mutex, leading to a use-after-free: mport_mm_close() -> mutex_lock(&map->md->buf_mutex); ... -> kref_put(&map->ref, mport_release_mapping); /* map is freed */ -> mutex_unlock(&map->md->buf_mutex); /* UAF: map used */ Fix this by caching map->md before kref_put() and using the cached pointer for mutex unlocking, ensuring that freed memory is not accessed. Link: https://lore.kernel.org/20260914042327.49798-1-james010kim@gmail.com Fixes: e8de370188d0 ("rapidio: add mport char device driver") Signed-off-by: James Kim Signed-off-by: Andrew Morton Cc: Alexandre Bounine Cc: Dan Carpenter Cc: Greg Kroah-Hartman Cc: Matt Porter Cc: --- drivers/rapidio/devices/rio_mport_cdev.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/drivers/rapidio/devices/rio_mport_cdev.c b/drivers/rapidio/devices/rio_mport_cdev.c index ad82c2108a567d..47ba34b4afb296 100644 --- a/drivers/rapidio/devices/rio_mport_cdev.c +++ b/drivers/rapidio/devices/rio_mport_cdev.c @@ -2167,11 +2167,12 @@ static void mport_mm_open(struct vm_area_struct *vma) static void mport_mm_close(struct vm_area_struct *vma) { struct rio_mport_mapping *map = vma->vm_private_data; + struct mport_dev *md = map->md; rmcd_debug(MMAP, "%pad", &map->phys_addr); - mutex_lock(&map->md->buf_mutex); + mutex_lock(&md->buf_mutex); kref_put(&map->ref, mport_release_mapping); - mutex_unlock(&map->md->buf_mutex); + mutex_unlock(&md->buf_mutex); } static const struct vm_operations_struct vm_ops = { From 565b0941f61963fced4e35523e5b1b9849bf6dc8 Mon Sep 17 00:00:00 2001 From: Zhiling Zou Date: Sat, 12 Sep 2026 21:32:17 +0800 Subject: [PATCH 0956/1012] lib: validate in-memory LZ4 chunk length We found and validated an issue in lib/decompress_unlz4.c. The bug is reachable by a root user through kexec_file_load() with a crafted external initrd. unlz4() reads the compressed chunk length from an in-memory initrd and passes it to LZ4_decompress_safe(). It only checks the chunk length against the allocation size when the input is filled by a callback. Reject an in-memory chunk that extends past the remaining input before calling the LZ4 decoder. This prevents malformed initrds from making the decoder read past the mapped archive. Link: https://lore.kernel.org/59c6555c27aa7ba18ee227f9f02c78d15a37605f.1789219453.git.zhilinz@nebusec.ai Fixes: e76e1fdfa8f8 ("lib: add support for LZ4-compressed kernel") Signed-off-by: Zhiling Zou Signed-off-by: Andrew Morton Reported-by: VEGA Assisted-by: LLM Cc: Kyungsik Lee --- lib/decompress_unlz4.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/lib/decompress_unlz4.c b/lib/decompress_unlz4.c index c0dbb3cea915eb..86e9aaec04f6d3 100644 --- a/lib/decompress_unlz4.c +++ b/lib/decompress_unlz4.c @@ -139,6 +139,10 @@ STATIC inline int INIT unlz4(u8 *input, long in_len, if (!fill) { inp += 4; size -= 4; + if (chunksize > size) { + error("data corrupted"); + goto exit_2; + } } else { if (chunksize > LZ4_compressBound(uncomp_chunksize)) { error("chunk length is longer than allocated"); From 389cb2a8766dbd1d168a3cdd634d5c0131d0f0d9 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Tue, 15 Sep 2026 18:06:23 +0000 Subject: [PATCH 0957/1012] taskstats: drop the unused CPU_DONT_CARE enum member CPU_DONT_CARE was added by the original listener cpumask code in 2006 (f9fd8914c1ac) and has never had a single user, not even in the commit that introduced it. It survived every refactor of the listener path since, including the recent handler folds. Kill it, twenty years is enough. Link: https://lore.kernel.org/20260915180623.22058-1-brads@mainlining.org Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Reviewed-by: Balbir Singh --- kernel/taskstats.c | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/kernel/taskstats.c b/kernel/taskstats.c index 598d9cd8325018..3997dbb5aeadc1 100644 --- a/kernel/taskstats.c +++ b/kernel/taskstats.c @@ -59,8 +59,7 @@ static DEFINE_PER_CPU(struct listener_list, listener_array); enum actions { REGISTER, - DEREGISTER, - CPU_DONT_CARE + DEREGISTER }; static int prepare_reply(struct genl_info *info, u8 cmd, struct sk_buff **skbp, From d6f448544d82e1ceb5ca6653db4bdf1ad12a80ac Mon Sep 17 00:00:00 2001 From: Jiaming Zhang Date: Mon, 14 Sep 2026 11:49:40 +0800 Subject: [PATCH 0958/1012] ocfs2: fix chunk number of the first chunk in a local quota file Mounting a crafted OCFS2 image can corrupt kernel memory. When the local quota file header of such an image claims zero chunks, the first quota entry allocated during the mount gets a chunk number taken from a kernel pointer, and releasing that entry clears a bit far outside the chunk bitmap. The write will corrupt the data stored at that address, and the corruption may lead to a system crash or damage unrelated data. Mounting requires CAP_SYS_ADMIN, so it takes an untrusted image, such as removable media or a loop mount of a file from elsewhere. Since the chunk number comes from a kernel pointer, the address of the bit that gets cleared differs from boot to boot, and the issue has been reported under several titles: - BUG: unable to handle kernel paging request in ocfs2_local_release_dquot - KASAN: use-after-free Write in ocfs2_local_release_dquot - KASAN: slab-out-of-bounds Write in ocfs2_local_release_dquot - KASAN: slab-use-after-free Write in ocfs2_local_release_dquot - KFENCE: use-after-free write in ocfs2_local_release_dquot Note that none of these is a use-after-free in the quota code: the chunk and its buffer head are alive. The bit that gets cleared lies far outside the bitmap. KASAN and KFENCE name each report based on the object occupying that address, which explains why the same issue is reported under so many different titles. In the report I sent, the address fell in a free page, which KASAN labels use-after-free. There are known reproducers. Both a syzkaller and a C reproducer are available from the Google Drive link [1] in my report thread. The issue was found by our modified syzkaller, not a theoretical thing found by an LLM. I have not seen it outside fuzzing, so it is not affecting me in the real world. The local quota file in OCFS2 is divided into chunks, and each chunk begins with a header block holding a bitmap of the quota entries that chunk has handed out. Chunks are numbered from zero, and that number is used to convert the file offset of an entry back into a bit position in the bitmap. ocfs2_local_quota_add_chunk() appends a new chunk to the in-memory list and numbers it one past the chunk that was last: list_add_tail(&chunk->qc_chunk, &oinfo->dqi_chunk); chunk->qc_num = list_entry(chunk->qc_chunk.prev, struct ocfs2_quota_chunk, qc_chunk)->qc_num + 1; The predecessor is looked up after the new chunk is added to the list, so if the list was empty, the prev pointer is the list head itself. The head is the dqi_chunk member of struct ocfs2_mem_dqinfo and is not a chunk, so reading qc_num through it lands 16 bytes past the start of the head, on the dqi_gqinode pointer that follows it. The first chunk of the file is then numbered with the lower half of a kernel pointer instead of 0. The list is empty when the local quota file header claims the file has no chunks. ocfs2_local_read_info() takes dqi_chunks from that header without validating it, so an image with dqi_chunks == 0 takes this path when the first quota entry is allocated. ocfs2_create_local_dquot() turns the bad number into a file offset with ol_dqblk_off(), which shifts a 32-bit block number left by the block size bits, so the top bits of such a large block number are lost. ocfs2_local_release_dquot() turns the offset back into a bit index with ol_dqblk_chunk_off(), using the full chunk number, so the lost bits push that index far outside the bitmap, and clearing it corrupts unrelated memory. Compute the chunk number before putting the chunk on the list, and use 0 when the list is empty. Link: https://lore.kernel.org/20260914034940.4070970-1-r772577952@gmail.com Link: https://drive.google.com/drive/folders/1-LzPTgOALEc3eOjmgM6oOfRKkYXJ-bnO?usp=drive_link [1] Fixes: 9e33d69f553a ("ocfs2: Implementation of local and global quota file handling") Signed-off-by: Jiaming Zhang Signed-off-by: Andrew Morton Closes: https://lore.kernel.org/lkml/CANypQFZ05tpth0Xc33gmP6jgPnkY4VuezyHcsajV-SqCsmN_gg@mail.gmail.com/ Reviewed-by: Joseph Qi Assisted-by: Claude Code:claude-opus-5 Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: Heming Zhao Cc: --- fs/ocfs2/quota_local.c | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/fs/ocfs2/quota_local.c b/fs/ocfs2/quota_local.c index f55810c59b1b11..d351cda9211f85 100644 --- a/fs/ocfs2/quota_local.c +++ b/fs/ocfs2/quota_local.c @@ -1071,10 +1071,13 @@ static struct ocfs2_quota_chunk *ocfs2_local_quota_add_chunk( goto out; } + if (list_empty(&oinfo->dqi_chunk)) + chunk->qc_num = 0; + else + chunk->qc_num = list_entry(oinfo->dqi_chunk.prev, + struct ocfs2_quota_chunk, + qc_chunk)->qc_num + 1; list_add_tail(&chunk->qc_chunk, &oinfo->dqi_chunk); - chunk->qc_num = list_entry(chunk->qc_chunk.prev, - struct ocfs2_quota_chunk, - qc_chunk)->qc_num + 1; chunk->qc_headerbh = bh; *offset = 0; return chunk; From d27faa9fab1e9afbe7e8260582143f833fd7aa4e Mon Sep 17 00:00:00 2001 From: Andy Shevchenko Date: Tue, 15 Sep 2026 10:53:53 +0200 Subject: [PATCH 0959/1012] resource: replace open coded resource_overlaps() In iomem_map_sanity_check() a piece of code resembles the content of the resource_overlaps(). Replace open coded piece with the call to the existing helper. Link: https://lore.kernel.org/20260915085353.3518153-1-andriy.shevchenko@linux.intel.com Signed-off-by: Andy Shevchenko Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Cc: Bjorn Helgaas --- kernel/resource.c | 14 ++++++-------- 1 file changed, 6 insertions(+), 8 deletions(-) diff --git a/kernel/resource.c b/kernel/resource.c index 54d7199695fbe9..cfc1a00e86aa03 100644 --- a/kernel/resource.c +++ b/kernel/resource.c @@ -1833,7 +1833,7 @@ __setup("reserve=", reserve_setup); */ int iomem_map_sanity_check(resource_size_t addr, unsigned long size) { - resource_size_t end = addr + size - 1; + struct resource mem = DEFINE_RES_MEM(addr, size); struct resource *p; int err = 0; @@ -1843,12 +1843,10 @@ int iomem_map_sanity_check(resource_size_t addr, unsigned long size) * We can probably skip the resources without * IORESOURCE_IO attribute? */ - if (p->start > end) + if (!resource_overlaps(p, &mem)) continue; - if (p->end < addr) - continue; - if (PFN_DOWN(p->start) <= PFN_DOWN(addr) && - PFN_DOWN(p->end) >= PFN_DOWN(end)) + if (PFN_DOWN(p->start) <= PFN_DOWN(mem.start) && + PFN_DOWN(p->end) >= PFN_DOWN(mem.end)) continue; /* * if a resource is "BUSY", it's not a hardware resource @@ -1859,8 +1857,8 @@ int iomem_map_sanity_check(resource_size_t addr, unsigned long size) if (p->flags & IORESOURCE_BUSY) continue; - pr_debug("resource sanity check: requesting [mem %pa-%pa], which spans more than %s %pR\n", - &addr, &end, p->name, p); + pr_debug("resource sanity check: requesting %pR, which spans more than %s %pR\n", + &mem, p->name, p); err = -1; break; } From 5489fd71059877ae162bf9400805dbe8f609c064 Mon Sep 17 00:00:00 2001 From: Andy Shevchenko Date: Tue, 15 Sep 2026 10:42:45 +0200 Subject: [PATCH 0960/1012] CREDITS: fix the ordering and other issues I assume that the ordering should be done in default (C) locale as other gives more lines shuffled. Fix the ordering accordingly. With that being said, clearly make at the top header how it should be sorted. Note, I took some empirical assumptions that people usually put their names in accordance with the First name(s) Last Name(s) split based on the records around (before and after the move). I hope I made no or little amount of mistakes. In any case it's just a dozen of names to update. Link: https://lore.kernel.org/20260915084248.3460260-1-andriy.shevchenko@linux.intel.com Signed-off-by: Andy Shevchenko Signed-off-by: Andrew Morton Cc: AngeloGiaocchino Del Regno Cc: Mathias Brugger --- CREDITS | 181 ++++++++++++++++++++++++++++---------------------------- 1 file changed, 91 insertions(+), 90 deletions(-) diff --git a/CREDITS b/CREDITS index a0bd941ca5dd26..3ae05ff1a0921b 100644 --- a/CREDITS +++ b/CREDITS @@ -1,6 +1,7 @@ This is at least a partial credits-file of people that have - contributed to the Linux project. It is sorted by name and - formatted to allow easy grepping and beautification by + contributed to the Linux project. It is sorted by family name, + assuming the used locale is default, e.g. by setting LC_ALL=C, + and formatted to allow easy grepping and beautification by scripts. The fields are: name (N), email (E), web-address (W), PGP key ID and fingerprint (P), description (D), and snail-mail address (S). @@ -344,6 +345,14 @@ D: Hardware spinlock (hwspinlock) subsystem D: OMAP hwspinlock driver D: OMAP remoteproc driver +N: Muli Ben-Yehuda +E: mulix@mulix.org +E: muli@il.ibm.com +W: http://www.mulix.org +D: trident OSS sound driver, x86-64 dma-ops and Calgary IOMMU, +D: KVM and Xen bits and other misc. hackery. +S: Haifa, Israel + N: Krzysztof Benedyczak E: golbi@mat.uni.torun.pl W: http://www.mat.uni.torun.pl/~golbi @@ -362,14 +371,6 @@ S: 2322 37th Ave SW S: Seattle, Washington 98126-2010 S: USA -N: Muli Ben-Yehuda -E: mulix@mulix.org -E: muli@il.ibm.com -W: http://www.mulix.org -D: trident OSS sound driver, x86-64 dma-ops and Calgary IOMMU, -D: KVM and Xen bits and other misc. hackery. -S: Haifa, Israel - N: Johannes Berg E: johannes@sipsolutions.net W: https://johannes.sipsolutions.net/ @@ -393,14 +394,14 @@ S: 1549 Hiironen Rd. S: Brimson, MN 55602 S: USA -N: Arnd Bergmann -D: Maintainer of Cell Broadband Engine Architecture - N: Hennus Bergman P: 1024/77D50909 76 99 FD 31 91 E1 96 1C 90 BB 22 80 62 F6 BD 63 D: Author and maintainer of the QIC-02 tape driver S: The Netherlands +N: Arnd Bergmann +D: Maintainer of Cell Broadband Engine Architecture + N: Tomas Berndtsson E: tomas@nocrew.org W: http://tomas.nocrew.org/ @@ -492,10 +493,6 @@ D: Various fixes (mostly networking) S: Montreal, Quebec S: Canada -N: Zoltán Böszörményi -E: zboszor@mail.externet.hu -D: MTRR emulation with Cyrix style ARR registers, Athlon MTRR support - N: John Boyd E: boyd@cis.ohio-state.edu D: Co-author of wd7000 SCSI driver @@ -589,9 +586,6 @@ N: Zach Brown E: zab@zabbo.net D: maestro pci sound -N: Zefan Li -D: Contribution to control group stuff - N: David Brownell D: Kernel engineer, mentor, and friend. Maintained USB EHCI and D: gadget layers, SPI subsystem, GPIO subsystem, and more than a few @@ -632,6 +626,10 @@ S: Ravenhorst 58 S: 2317 AK Leiden S: The Netherlands +N: Zoltán Böszörményi +E: zboszor@mail.externet.hu +D: MTRR emulation with Cyrix style ARR registers, Athlon MTRR support + N: Michael Callahan E: callahan@maths.ox.ac.uk D: PPP for Linux @@ -703,6 +701,10 @@ S: Tamsui town, Taipei county, S: Taiwan 251 S: Republic of China +N: Landen Chao +E: Landen.Chao@mediatek.com +D: MT7531 Ethernet switch support + N: Michael Elizabeth Chastain E: mec@shout.net D: Configure, Menuconfig, xconfig @@ -715,10 +717,6 @@ D: Media subsystem (V4L/DVB) drivers and core D: EDAC drivers and EDAC 3.0 core rework S: Brazil -N: Landen Chao -E: Landen.Chao@mediatek.com -D: MT7531 Ethernet switch support - N: Raymond Chen E: raymondc@microsoft.com D: Author of Configure script @@ -1447,6 +1445,10 @@ D: Made support for modules, ramdisk, generic-serial, etc. optional. D: Transformed old user space bdflush into 1st kernel thread - kflushd. D: Many other patches, documentation files, mini kernels, utilities, ... +N: Andy Gospodarek +E: andy@greyhouse.net +D: Maintenance and contributions to the network interface bonding driver. + N: Masanori GOTO E: gotom@debian.or.jp D: Workbit NinjaSCSI-32Bi/UDE driver @@ -1459,10 +1461,6 @@ S: 8124 Constitution Apt. 7 S: Sterling Heights, Michigan 48313 S: USA -N: Andy Gospodarek -E: andy@greyhouse.net -D: Maintenance and contributions to the network interface bonding driver. - N: Vivek Goyal E: vgoyal@redhat.com D: KDUMP, KEXEC, and VIRTIO FILE SYSTEM @@ -1535,6 +1533,14 @@ S: 44 St. Joseph Street, Suite 506 S: Toronto, Ontario, M4Y 2W4 S: Canada +N: Nitin Gupta +E: ngupta@vflare.org +D: zsmalloc memory allocator and zram block device driver + +N: Justin Guyett +E: jguyett@andrew.cmu.edu +D: via-rhine net driver hacking + N: Richard Günther E: rguenth@tat.physik.uni-tuebingen.de W: http://www.tat.physik.uni-tuebingen.de/~rguenth @@ -1543,14 +1549,6 @@ D: binfmt_misc S: 72074 Tübingen S: Germany -N: Justin Guyett -E: jguyett@andrew.cmu.edu -D: via-rhine net driver hacking - -N: Nitin Gupta -E: ngupta@vflare.org -D: zsmalloc memory allocator and zram block device driver - N: Danny ter Haar E: dth@cistron.nl D: /proc/cpuinfo, reboot on panic , kernel pre-patch tester ;) @@ -1945,16 +1943,6 @@ E: sjenning@redhat.com D: Creation and maintenance of zswap D: Creation and maintenace of the zbud allocator -N: Jeremy Kerr -D: Maintainer of SPU File System - -N: Michael Kerrisk -E: mtk.manpages@gmail.com -W: https://man7.org/ -P: 4096R/3A35CE5E E522 595B 52ED A4E6 BFCC CB5E 8561 9911 3A35 CE5E -D: Maintainer of the Linux man-pages project -D: Linux man pages online, at - N: Niels Kristian Bech Jensen E: nkbj1970@hotmail.com D: Miscellaneous kernel updates and fixes. @@ -2099,6 +2087,16 @@ S: Keplerstr. 6 S: 4050 Traun S: Austria +N: Jeremy Kerr +D: Maintainer of SPU File System + +N: Michael Kerrisk +E: mtk.manpages@gmail.com +W: https://man7.org/ +P: 4096R/3A35CE5E E522 595B 52ED A4E6 BFCC CB5E 8561 9911 3A35 CE5E +D: Maintainer of the Linux man-pages project +D: Linux man pages online, at + N: Karl Keyte E: karl@koft.com D: Disk usage statistics and modifications to line printer driver @@ -2483,6 +2481,9 @@ D: and much more. He was the maintainer of MD from 2016 to 2018. Shaohua D: passed away late 2018, he will be greatly missed. W: https://www.spinics.net/lists/raid/msg61993.html +N: Zefan Li +D: Contribution to control group stuff + N: Stephan Linz E: linz@mazet.de E: Stephan.Linz@gmx.de @@ -2542,11 +2543,6 @@ D: misc. kernel hacking and debugging S: Cambridge, MA 02139 S: USA -N: Martin von Löwis -E: loewis@informatik.hu-berlin.de -D: script binary format -D: NTFS driver - N: H.J. Lu E: hjl@gnu.ai.mit.edu D: GCC + libraries hacker @@ -2575,6 +2571,11 @@ S: Puistokaari 1 E 18 S: 00200 Helsinki S: Finland +N: Martin von Löwis +E: loewis@informatik.hu-berlin.de +D: script binary format +D: NTFS driver + N: Daniel J. Maas E: dmaas@dcine.com W: https://www.maasdigital.com @@ -2625,10 +2626,6 @@ S: PO BOX 220, HFX. CENTRAL S: Halifax, Nova Scotia S: Canada B3J 3C8 -N: Kai Mäkisara -E: Kai.Makisara@kolumbus.fi -D: SCSI Tape Driver - N: Asit Mallick E: asit.k.mallick@intel.com D: Linux/IA-64 @@ -2912,13 +2909,6 @@ S: 12725 SW Millikan Way, Suite 400 S: Beaverton, Oregon 97005 S: USA -N: Eberhard Mönkeberg -E: emoenke@gwdg.de -D: CDROM driver "sbpcd" (Matsushita/Panasonic/Soundblaster) -S: Ruhstrathöhe 2 b. -S: D-37085 Göttingen -S: Germany - N: Thomas Molina E: tmolina@cablespeed.com D: bug fixes, documentation, minor hackery @@ -3005,6 +2995,17 @@ S: Dragonvagen 1 A 13 S: FIN-00330 Helsingfors S: Finland +N: Kai Mäkisara +E: Kai.Makisara@kolumbus.fi +D: SCSI Tape Driver + +N: Eberhard Mönkeberg +E: emoenke@gwdg.de +D: CDROM driver "sbpcd" (Matsushita/Panasonic/Soundblaster) +S: Ruhstrathöhe 2 b. +S: D-37085 Göttingen +S: Germany + N: Matija Nalis E: mnalis@jagor.srce.hr E: mnalis@voyager.hr @@ -3200,13 +3201,6 @@ S: RR #5, 497 Pole Line Road S: Thunder Bay, Ontario S: CANADA P7C 5M9 -N: Inaky Perez-Gonzalez -E: inaky.perez-gonzalez@intel.com -E: linux-wimax@intel.com -E: inakypg@yahoo.com -D: WiMAX stack -D: Intel Wireless WiMAX Connection 2400 driver - N: Yuri Per E: yuri@pts.mipt.ru D: Some smbfs fixes @@ -3214,6 +3208,13 @@ S: Demonstratsii 8-382 S: Tula 300000 S: Russia +N: Inaky Perez-Gonzalez +E: inaky.perez-gonzalez@intel.com +E: linux-wimax@intel.com +E: inakypg@yahoo.com +D: WiMAX stack +D: Intel Wireless WiMAX Connection 2400 driver + N: Thomas Petazzoni E: thomas.petazzoni@bootlin.com D: Driver for the Marvell Armada 370/XP network unit. @@ -3328,14 +3329,14 @@ N: Frederic Potter E: fpotter@cirpack.com D: Some PCI kernel support -N: Rui Prior -E: rprior@inescn.pt -D: ATM device driver for NICStAR based cards - N: Roopa Prabhu E: roopa@nvidia.com D: Bridge co-maintainer, vxlan and networking contributor +N: Rui Prior +E: rprior@inescn.pt +D: ATM device driver for NICStAR based cards + N: Stefan Probst E: sp@caldera.de D: The Linux Support Team Erlangen, 1993-97 @@ -4023,12 +4024,6 @@ S: C/ Federico Garcia Lorca 1 10-A S: Sevilla 41005 S: Spain -N: Björn Töpel -E: bjorn@kernel.org -D: AF_XDP -S: Gothenburg -S: Sweden - N: Linus Torvalds E: torvalds@linux-foundation.org D: Original kernel hacker @@ -4082,17 +4077,6 @@ D: Linux-Workshop Köln (aka LUG Cologne, Germany), Installfests S: Tacitusstr. 6 S: D-50968 Köln -N: Tsu-Sheng Tsao -E: tsusheng@scf.usc.edu -D: IGMP (Internet Group Management Protocol) version 2 -S: 2F 14 ALY 31 LN 166 SEC 1 SHIH-PEI RD -S: Taipei -S: Taiwan 112 -S: Republic of China -S: 24335 Delta Drive -S: Diamond Bar, California 91765 -S: USA - N: Theodore Ts'o E: tytso@mit.edu D: Random Linux hacker @@ -4109,6 +4093,17 @@ S: 1 Amherst Street S: Cambridge, Massachusetts 02139 S: USA +N: Tsu-Sheng Tsao +E: tsusheng@scf.usc.edu +D: IGMP (Internet Group Management Protocol) version 2 +S: 2F 14 ALY 31 LN 166 SEC 1 SHIH-PEI RD +S: Taipei +S: Taiwan 112 +S: Republic of China +S: 24335 Delta Drive +S: Diamond Bar, California 91765 +S: USA + N: Luben Tuikov E: Luben Tuikov D: Maintainer of the DRM GPU Scheduler @@ -4135,6 +4130,12 @@ S: 44 Campbell Park Crescent S: Edinburgh EH13 0HT S: United Kingdom +N: Björn Töpel +E: bjorn@kernel.org +D: AF_XDP +S: Gothenburg +S: Sweden + N: Thomas Uhl E: uhl@sun1.rz.fh-heilbronn.de D: Application programmer From b9c1e480b8b567f6420e2bc35765a2c71e6014c4 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 16 Sep 2026 18:29:52 +0000 Subject: [PATCH 0961/1012] panic: fix redirect CPU race in panic_try_force_cpu() Patch series "panic: fix panic_force_cpu= redirect races and NMI bypass", v7. The panic_force_cpu= parameter redirects a panic to a specific CPU so the crash kernel runs there. The redirect code in panic_try_force_cpu() had two races and an NMI bypass, all found by Sashiko. This series closes them and kills one more hack that the rework turned up. Panic from NMI still goes through the redirect path first, so the crash kernel ends up on the CPU the admin asked for. The redirect buffer is a static 1KB now, no initcall, no fallback message. This patch (of 6): The cmpxchg() in panic_try_force_cpu() makes sure that only one CPU tries to redirect panic() to the requested CPU. It is similar to the cmpxchg() in panic_try_start() which makes sure that only one CPU does the panic(). In both situations, only the winner of cmpxchg() should proceed further. Other CPUs should go offline. There is a bug because the cmpxchg loser returns false and falls through into vpanic(). Two non-target CPUs A and B panic, the requested CPU is C: cpu A cpu B ---------- ---------- panic() panic() vpanic() vpanic() panic_try_force_cpu() panic_try_force_cpu() cmpxchg wins cmpxchg fails redirect = A old_cpu = A IPI -> C return false <- BUG return true panic_try_start() wins panic_smp_self_stop() __crash_kexec() on B (A stops) (target C bypassed) The loser must stop, not fall through. It cannot just return true, though. A CPU that already won the redirect cmpxchg can reenter panic_try_force_cpu() on the same CPU, for example a nested NMI during the message formatting, before the IPI is sent: cpu A (1st) cpu A (nested) ---------- ---------- panic() vpanic() panic_try_force_cpu() cmpxchg wins (redirect = A) vsnprintf(msg) ... <-- NMI, nested panic --> panic() vpanic() panic_try_force_cpu() cmpxchg fails old_cpu == A (this CPU) return true <- would halt panic_smp_self_stop() (IPI never sent, panic abandoned) Check old_cpu against this_cpu so a second call from the same CPU returns false and falls through to panic_try_start() instead. Also fix the panic_in_progress() check. We must not redirect when panic_cpu is already assigned. Return true to stop when the panic is on another CPU, false to proceed when it is this one. Update the panic_try_force_cpu() doc comment for the new return value semantics. Link: https://lore.kernel.org/20260916182957.7788-1-brads@mainlining.org Link: https://lore.kernel.org/20260916182957.7788-2-brads@mainlining.org Fixes: 2e171ab29f91 ("panic: add panic_force_cpu= parameter to redirect panic to a specific CPU") Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Reported-by: Sashiko Closes: https://sashiko.dev/#/patchset/20260705164123.18746-1-include@grrlz.net Closes: https://sashiko.dev/#/patchset/20260707172252.4842-1-include@grrlz.net Reviewed-by: Petr Mladek Cc: Guenetr Roeck Cc: Wang Jinchao Cc: Wim Van Sebroeck Cc: --- kernel/panic.c | 19 ++++++++++++------- 1 file changed, 12 insertions(+), 7 deletions(-) diff --git a/kernel/panic.c b/kernel/panic.c index 50715f14cf04ef..08072bfae42219 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -371,8 +371,9 @@ int __weak panic_smp_redirect_cpu(int target_cpu, void *msg) * for the crash kernel to function correctly. This function redirects * panic handling to the CPU specified via the panic_force_cpu= boot parameter. * - * Returns false if panic should proceed on current CPU. - * Returns true if panic was redirected. + * Returns true when this CPU must stop: the panic was redirected or is + * already running on another CPU. + * Returns false when panic() should proceed on this CPU. */ __printf(1, 0) static bool panic_try_force_cpu(const char *fmt, va_list args) @@ -396,16 +397,20 @@ static bool panic_try_force_cpu(const char *fmt, va_list args) return false; } - /* Another panic already in progress */ + /* + * Don't redirect when a panic is already in progress. Stop this + * CPU when it's another one, proceed when it's this one. + */ if (panic_in_progress()) - return false; + return panic_on_other_cpu(); /* - * Only one CPU can do the redirect. Use atomic cmpxchg to ensure - * we don't race with another CPU also trying to redirect. + * Only one CPU can do the redirection. Others should go offline. + * Continue with panic() when we already tried the redirection + * from this CPU before, for example via nmi_panic(). */ if (!atomic_try_cmpxchg(&panic_redirect_cpu, &old_cpu, this_cpu)) - return false; + return old_cpu != this_cpu; /* * Use dynamically allocated buffer if available, otherwise From 7d19f9258234393bf71139e674ad30bfe9466b56 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 16 Sep 2026 18:29:53 +0000 Subject: [PATCH 0962/1012] panic: flatten nmi_panic control flow panic() is __noreturn, so the else after panic_try_start() is dead. Drop it so the force_cpu path can be added cleanly on top. Link: https://lore.kernel.org/20260916182957.7788-3-brads@mainlining.org Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Reviewed-by: Petr Mladek Cc: Guenetr Roeck Cc: Sashiko Cc: Wang Jinchao Cc: Wim Van Sebroeck Cc: --- kernel/panic.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/kernel/panic.c b/kernel/panic.c index 08072bfae42219..5646d4fb82b7d4 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -517,7 +517,8 @@ void nmi_panic(struct pt_regs *regs, const char *msg) { if (panic_try_start()) panic("%s", msg); - else if (panic_on_other_cpu()) + + if (panic_on_other_cpu()) nmi_panic_self_stop(regs); } EXPORT_SYMBOL(nmi_panic); From 1694060052d87593cbef3ff0775762a57c13a197 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 16 Sep 2026 18:29:54 +0000 Subject: [PATCH 0963/1012] panic: fix va_list reuse in panic_try_force_cpu() vsnprintf() consumes the caller's va_list. When the redirect fails, vpanic() reuses it for the panic message, which is undefined behavior. Use va_copy(). Link: https://lore.kernel.org/20260916182957.7788-4-brads@mainlining.org Fixes: 2e171ab29f91 ("panic: add panic_force_cpu= parameter to redirect panic to a specific CPU") Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Reviewed-by: Petr Mladek Cc: Guenetr Roeck Cc: Sashiko Cc: Wang Jinchao Cc: Wim Van Sebroeck Cc: --- kernel/panic.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/kernel/panic.c b/kernel/panic.c index 5646d4fb82b7d4..7388eb81a1c471 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -417,7 +417,12 @@ static bool panic_try_force_cpu(const char *fmt, va_list args) * fall back to static message for early boot panics or allocation failure. */ if (panic_force_buf) { - vsnprintf(panic_force_buf, PANIC_MSG_BUFSZ, fmt, args); + va_list ap; + + /* Do not consume args, the caller reuses it if we fail */ + va_copy(ap, args); + vsnprintf(panic_force_buf, PANIC_MSG_BUFSZ, fmt, ap); + va_end(ap); msg = panic_force_buf; } else { msg = "Redirected panic (buffer unavailable)"; From 2373d90be5e007b02bf0e481db6ff20f706598b6 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 16 Sep 2026 18:29:55 +0000 Subject: [PATCH 0964/1012] panic: restore variable arguments to nmi_panic() nmi_panic() used to accept variable arguments until commit ebc41f20d77f ("panic: change nmi_panic from macro to function") flattened it to a final message string. vpanic() did not exist back then, so the function had to format through panic("%s", msg). Bring the variable arguments back and format with vpanic() directly. The next patch makes nmi_panic() try the panic_force_cpu= redirect before claiming panic_cpu. panic_try_force_cpu() needs to print the message but it is used also by panic() which accepts variable argument list. The string could be formatted only when a CPU gets assigned to process the redirection. Otherwise, there might be a race when writing to the `panic_force_buf`. No current caller passes a string with format specifiers. The closest one is hpwdt_pretimeout(), which builds panic_msg with hex_byte_pack() and has only two variants, both plain strings. But the new __printf() annotation on nmi_panic() would warn with -Wformat-security there because the buffer is passed directly as the format argument, so switch it to nmi_panic(regs, "%s", panic_msg). [pmladek@suse.com: changelog fix] Link: https://lore.kernel.org/arJxFybmtD7OBIcL@pathway.suse.cz Link: https://lore.kernel.org/20260916182957.7788-5-brads@mainlining.org Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Suggested-by: Petr Mladek Reviewed-by: Petr Mladek Cc: Guenetr Roeck Cc: Sashiko Cc: Wang Jinchao Cc: Wim Van Sebroeck Cc: --- drivers/watchdog/hpwdt.c | 2 +- include/linux/panic.h | 3 ++- kernel/panic.c | 10 ++++++++-- 3 files changed, 11 insertions(+), 4 deletions(-) diff --git a/drivers/watchdog/hpwdt.c b/drivers/watchdog/hpwdt.c index 8af1fad2de0bd3..78227d200afe4b 100644 --- a/drivers/watchdog/hpwdt.c +++ b/drivers/watchdog/hpwdt.c @@ -199,7 +199,7 @@ static int hpwdt_pretimeout(unsigned int ulReason, struct pt_regs *regs) } hex_byte_pack(panic_msg, nmistat); - nmi_panic(regs, panic_msg); + nmi_panic(regs, "%s", panic_msg); return NMI_HANDLED; } diff --git a/include/linux/panic.h b/include/linux/panic.h index 98dd7dfd27de7a..17e61b61c45f8a 100644 --- a/include/linux/panic.h +++ b/include/linux/panic.h @@ -13,7 +13,8 @@ __printf(1, 2) void panic(const char *fmt, ...) __noreturn __cold; __printf(1, 0) void vpanic(const char *fmt, va_list args) __noreturn __cold; -void nmi_panic(struct pt_regs *regs, const char *msg); +__printf(2, 3) +void nmi_panic(struct pt_regs *regs, const char *fmt, ...); void check_panic_on_warn(const char *origin); extern void oops_enter(void); extern void oops_exit(void); diff --git a/kernel/panic.c b/kernel/panic.c index 7388eb81a1c471..e240ca06faab21 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -518,13 +518,19 @@ EXPORT_SYMBOL(panic_on_other_cpu); * nmi_panic_self_stop() which can provide architecture dependent code such * as saving register state for crash dump. */ -void nmi_panic(struct pt_regs *regs, const char *msg) +void nmi_panic(struct pt_regs *regs, const char *fmt, ...) { + va_list args; + + va_start(args, fmt); + if (panic_try_start()) - panic("%s", msg); + vpanic(fmt, args); if (panic_on_other_cpu()) nmi_panic_self_stop(regs); + + va_end(args); } EXPORT_SYMBOL(nmi_panic); From f911e23ebb9d31710dfce927faae7990e5cc6818 Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 16 Sep 2026 18:29:56 +0000 Subject: [PATCH 0965/1012] panic: allow force_cpu redirect from an NMI nmi_panic() claims panic_cpu via panic_try_start() before calling panic(). When the panic later reaches panic_try_force_cpu(), the panic_in_progress() check sees panic_cpu set and refuses to redirect. The crash kernel runs on the CPU that took the NMI instead of the CPU requested with panic_force_cpu=: nmi_panic() panic_try_start() wins, panic_cpu = X panic("%s", msg) vpanic() panic_try_force_cpu() panic_in_progress() true, panic_cpu is X return false redirect bypassed panic_try_start() already won __crash_kexec() on X, not the requested CPU Try the redirect before claiming panic_cpu instead, as suggested by Petr Mladek. nmi_panic() now calls panic_try_force_cpu() first and claims panic_cpu only when no redirect happened. The requested CPU claims panic_cpu itself when it runs panic(), so panic_cpu does not need to be handed off. panic_try_force_cpu() copies the arguments before formatting (patch 3), so nmi_panic() can pass them to vpanic() again when no redirect happens. The redirect IPI is sent with smp_call_function_single_async(), which is not guaranteed to work from NMI context. Treat it as best effort. It is worth the risk because the redirection is only used when the crash kernel would not work on the panicking CPU anyway. Keep returning when the panic is already running on this CPU. A nested NMI, for example with unknown_nmi_panic while this CPU is inside panic(), must return and let the interrupted panic() continue instead of parking the CPU in nmi_panic_self_stop(). Mark the redirecting CPU offline before stopping it, like vpanic() does, so that panic_other_cpus_shutdown() on the target CPU does not wait for it. Link: https://lore.kernel.org/20260916182957.7788-6-brads@mainlining.org Fixes: 2e171ab29f91 ("panic: add panic_force_cpu= parameter to redirect panic to a specific CPU") Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Reported-by: Sashiko Closes: https://sashiko.dev/#/patchset/20260708164312.19044-1-include@grrlz.net Reviewed-by: Petr Mladek Cc: Guenetr Roeck Cc: Wang Jinchao Cc: Wim Van Sebroeck Cc: --- kernel/panic.c | 19 +++++++++++++++---- 1 file changed, 15 insertions(+), 4 deletions(-) diff --git a/kernel/panic.c b/kernel/panic.c index e240ca06faab21..29c981926f5f34 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -513,10 +513,11 @@ bool panic_on_other_cpu(void) EXPORT_SYMBOL(panic_on_other_cpu); /* - * A variant of panic() called from NMI context. We return if we've already - * panicked on this CPU. If another CPU already panicked, loop in - * nmi_panic_self_stop() which can provide architecture dependent code such - * as saving register state for crash dump. + * A variant of panic() called from NMI context. The panic is first + * redirected to the CPU requested via panic_force_cpu=, when configured. + * We return if we've already panicked on this CPU. If another CPU already + * panicked, loop in nmi_panic_self_stop() which can provide architecture + * dependent code for saving register state for crash dump. */ void nmi_panic(struct pt_regs *regs, const char *fmt, ...) { @@ -524,6 +525,16 @@ void nmi_panic(struct pt_regs *regs, const char *fmt, ...) va_start(args, fmt); + /* Try to redirect to the requested CPU before claiming panic_cpu. */ + if (panic_try_force_cpu(fmt, args)) { + /* + * Mark ourselves offline so panic_other_cpus_shutdown() won't + * wait for us on architectures that check num_online_cpus(). + */ + set_cpu_online(raw_smp_processor_id(), false); + nmi_panic_self_stop(regs); + } + if (panic_try_start()) vpanic(fmt, args); From 4ae2113b94a3669faa26ed3535d185fed956a98e Mon Sep 17 00:00:00 2001 From: Bradley Morgan Date: Wed, 16 Sep 2026 18:29:57 +0000 Subject: [PATCH 0966/1012] panic: kill the "buffer unavailable" redirect fallback The redirect buffer is a disgusting terrible hack. panic_force_buf is kmalloc'ed in a late_initcall, and until then the redirect delivers this as the panic message: Redirected panic (buffer unavailable) The whole point of the redirect is to hand the panic message to the target CPU, so the crash kernel boots knowing it panicked and not why. And that window is the entire boot, from the early_param to the late_initcall, which is exactly when you most want the message. Make it a static 1KB buffer and kill the initcall. The cost is 1KB of .bss in SMP crash dump builds, and it is only ever touched when panic_force_cpu= is set anyway. The local msg variable is gone too, the buffer goes directly to panic_smp_redirect_cpu(). Link: https://lore.kernel.org/20260916182957.7788-7-brads@mainlining.org Signed-off-by: Bradley Morgan Signed-off-by: Andrew Morton Suggested-by: Petr Mladek Reviewed-by: Petr Mladek Cc: Guenetr Roeck Cc: Sashiko Cc: Wang Jinchao Cc: Wim Van Sebroeck Cc: --- kernel/panic.c | 34 +++++++--------------------------- 1 file changed, 7 insertions(+), 27 deletions(-) diff --git a/kernel/panic.c b/kernel/panic.c index 29c981926f5f34..170744163fc2a8 100644 --- a/kernel/panic.c +++ b/kernel/panic.c @@ -307,7 +307,7 @@ atomic_t panic_cpu = ATOMIC_INIT(PANIC_CPU_INVALID); atomic_t panic_redirect_cpu = ATOMIC_INIT(PANIC_CPU_INVALID); #if defined(CONFIG_SMP) && defined(CONFIG_CRASH_DUMP) -static char *panic_force_buf; +static char panic_force_buf[PANIC_MSG_BUFSZ]; static int __init panic_force_cpu_setup(char *str) { @@ -326,17 +326,6 @@ static int __init panic_force_cpu_setup(char *str) } early_param("panic_force_cpu", panic_force_cpu_setup); -static int __init panic_force_cpu_late_init(void) -{ - if (panic_force_cpu < 0) - return 0; - - panic_force_buf = kmalloc(PANIC_MSG_BUFSZ, GFP_KERNEL); - - return 0; -} -late_initcall(panic_force_cpu_late_init); - static void do_panic_on_target_cpu(void *info) { panic("%s", (char *)info); @@ -380,7 +369,7 @@ static bool panic_try_force_cpu(const char *fmt, va_list args) { int this_cpu = raw_smp_processor_id(); int old_cpu = PANIC_CPU_INVALID; - const char *msg; + va_list ap; /* Feature not enabled via boot parameter */ if (panic_force_cpu < 0) @@ -413,20 +402,11 @@ static bool panic_try_force_cpu(const char *fmt, va_list args) return old_cpu != this_cpu; /* - * Use dynamically allocated buffer if available, otherwise - * fall back to static message for early boot panics or allocation failure. + * Do not consume args, the caller reuses them if we fail. */ - if (panic_force_buf) { - va_list ap; - - /* Do not consume args, the caller reuses it if we fail */ - va_copy(ap, args); - vsnprintf(panic_force_buf, PANIC_MSG_BUFSZ, fmt, ap); - va_end(ap); - msg = panic_force_buf; - } else { - msg = "Redirected panic (buffer unavailable)"; - } + va_copy(ap, args); + vsnprintf(panic_force_buf, PANIC_MSG_BUFSZ, fmt, ap); + va_end(ap); console_verbose(); bust_spinlocks(1); @@ -441,7 +421,7 @@ static bool panic_try_force_cpu(const char *fmt, va_list args) dump_stack(); } - if (panic_smp_redirect_cpu(panic_force_cpu, (void *)msg) != 0) { + if (panic_smp_redirect_cpu(panic_force_cpu, panic_force_buf) != 0) { atomic_set(&panic_redirect_cpu, PANIC_CPU_INVALID); pr_warn("panic: failed to redirect to CPU %d, continuing on CPU %d\n", panic_force_cpu, this_cpu); From 6ff3222b40e152a36335b505b95b73a911bd4095 Mon Sep 17 00:00:00 2001 From: Jonas Rebmann Date: Wed, 16 Sep 2026 19:38:06 +0200 Subject: [PATCH 0967/1012] lib/tests: string_helpers: check null terminator too Patch series "lib/string_helpers: fixes and test cases for string_unescape()". This series fixes two bugs in string_unescape() regarding the destination buffer length. Both fixes are accompanied with kunit tests which would fail without the fixes. To make this possible, preparatory patches 1 and 2 improve and clean up testing helpers and 3 introduces test_unescape_one() which allows for targeted testing of the string_unescape() function. This patch (of 5): string_unescape() returns the number of character written to dst, not counting the null terminator which is always written. Ensure string_unescape has included the null terminator by adding a separate check. While at it, improve output for failed tests by showing the memory dump even if the length differs. Link: https://lore.kernel.org/20260916-string_unescape-v1-0-7f8bd986fa33@pengutronix.de Link: https://lore.kernel.org/20260916-string_unescape-v1-1-7f8bd986fa33@pengutronix.de Signed-off-by: Jonas Rebmann Signed-off-by: Andrew Morton Cc: Andy Shevchenko Cc: Kees Cook Cc: Sascha Hauer --- lib/tests/string_helpers_kunit.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/lib/tests/string_helpers_kunit.c b/lib/tests/string_helpers_kunit.c index 9fbe91079c7ed0..9bc3acffaf2f10 100644 --- a/lib/tests/string_helpers_kunit.c +++ b/lib/tests/string_helpers_kunit.c @@ -22,7 +22,7 @@ static void test_string_check_buf(struct kunit *test, char *out_real, size_t q_real, char *out_test, size_t q_test) { - KUNIT_ASSERT_EQ_MSG(test, q_real, q_test, "name:%s", name); + KUNIT_EXPECT_EQ_MSG(test, q_real, q_test, "name:%s", name); KUNIT_EXPECT_MEMEQ_MSG(test, out_test, out_real, q_test, "name:%s", name); } @@ -103,6 +103,7 @@ static void test_string_unescape(struct kunit *test, test_string_check_buf(test, name, flags, in, p - 1, out_real, q_real, out_test, q_test); + KUNIT_EXPECT_EQ_MSG(test, out_real[q_real], '\0', "name:%s", name); } struct test_string_1 { From 0d4a436031d3c8fb14bfad6785798327748a710a Mon Sep 17 00:00:00 2001 From: Jonas Rebmann Date: Wed, 16 Sep 2026 19:38:07 +0200 Subject: [PATCH 0968/1012] lib/tests: string_helpers: drop unused parameters Drop the unused parameters from test_string_check_buf() to improve readability. Link: https://lore.kernel.org/20260916-string_unescape-v1-2-7f8bd986fa33@pengutronix.de Signed-off-by: Jonas Rebmann Signed-off-by: Andrew Morton Cc: Andy Shevchenko Cc: Kees Cook Cc: Sascha Hauer --- lib/tests/string_helpers_kunit.c | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/lib/tests/string_helpers_kunit.c b/lib/tests/string_helpers_kunit.c index 9bc3acffaf2f10..1ed652f762d160 100644 --- a/lib/tests/string_helpers_kunit.c +++ b/lib/tests/string_helpers_kunit.c @@ -18,7 +18,6 @@ static void test_string_check_buf(struct kunit *test, const char *name, unsigned int flags, - char *in, size_t p, char *out_real, size_t q_real, char *out_test, size_t q_test) { @@ -101,8 +100,7 @@ static void test_string_unescape(struct kunit *test, q_real = string_unescape(in, out_real, q_real, flags); } - test_string_check_buf(test, name, flags, in, p - 1, out_real, q_real, - out_test, q_test); + test_string_check_buf(test, name, flags, out_real, q_real, out_test, q_test); KUNIT_EXPECT_EQ_MSG(test, out_real[q_real], '\0', "name:%s", name); } @@ -457,8 +455,7 @@ static void test_string_escape(struct kunit *test, const char *name, q_real = string_escape_mem(in, p, out_real, out_size, flags, esc); - test_string_check_buf(test, name, flags, in, p, out_real, q_real, out_test, - q_test); + test_string_check_buf(test, name, flags, out_real, q_real, out_test, q_test); test_string_escape_overflow(test, in, p, flags, esc, q_test, name); } From e5de51fa862569b5ba6c2b36abc3636ca7087367 Mon Sep 17 00:00:00 2001 From: Jonas Rebmann Date: Wed, 16 Sep 2026 19:38:08 +0200 Subject: [PATCH 0969/1012] lib/tests: string_helpers: introduce test_string_unescape_one The existing test_string_unescape() function follows a complex procedure where it, given a set of UNESCAPE flags, appends multiple test fragments and predicts their unescape result for the chosen set of flags. Rename test_string_unescape() to a more descriptive test_string_unescape_combined In preparation to add simple regression tests, introduce test_string_unescape_one() which asserts on exactly one call to string_unescape. Add some tests for corner cases which already pass. Link: https://lore.kernel.org/20260916-string_unescape-v1-3-7f8bd986fa33@pengutronix.de Signed-off-by: Jonas Rebmann Signed-off-by: Andrew Morton Cc: Andy Shevchenko Cc: Kees Cook Cc: Sascha Hauer --- lib/tests/string_helpers_kunit.c | 31 +++++++++++++++++++++++++------ 1 file changed, 25 insertions(+), 6 deletions(-) diff --git a/lib/tests/string_helpers_kunit.c b/lib/tests/string_helpers_kunit.c index 1ed652f762d160..3c6fa7324965ce 100644 --- a/lib/tests/string_helpers_kunit.c +++ b/lib/tests/string_helpers_kunit.c @@ -55,9 +55,9 @@ static const struct test_string strings[] = { }, }; -static void test_string_unescape(struct kunit *test, - const char *name, unsigned int flags, - bool inplace) +static void test_string_unescape_combined(struct kunit *test, + const char *name, unsigned int flags, + bool inplace) { int q_real = 256; char *in = kunit_kzalloc(test, q_real, GFP_KERNEL); @@ -596,14 +596,33 @@ static void test_upper_lower(struct kunit *test) } } +static void test_string_unescape_one(struct kunit *test, + const char *name, unsigned int flags, + char *src, size_t len, + char *out_test, size_t q_test) +{ + char *out_real = kunit_kzalloc(test, len, GFP_KERNEL); + int q_real; + + q_real = string_unescape(src, out_real, len, flags); + test_string_check_buf(test, name, flags, out_real, q_real, out_test, q_test); +} + static void test_unescape(struct kunit *test) { unsigned int i; for (i = 0; i < UNESCAPE_ALL_MASK + 1; i++) - test_string_unescape(test, "unescape", i, false); - test_string_unescape(test, "unescape inplace", - get_random_u32_below(UNESCAPE_ALL_MASK + 1), true); + test_string_unescape_combined(test, "unescape", i, false); + test_string_unescape_combined(test, "unescape inplace", + get_random_u32_below(UNESCAPE_ALL_MASK + 1), true); + + test_string_unescape_one(test, "simple case", UNESCAPE_HEX | UNESCAPE_SPECIAL, "ABC", 6, "ABC", 3); + test_string_unescape_one(test, "single escape", UNESCAPE_HEX | UNESCAPE_SPECIAL, "A\\x42C", 6, "ABC", 3); + test_string_unescape_one(test, "escape before end", UNESCAPE_HEX, "B\\qX", 4, "B\\q", 3); + test_string_unescape_one(test, "escape at end", UNESCAPE_HEX, "a\\qX", 3, "a\\", 2); + test_string_unescape_one(test, "backslash before escape", UNESCAPE_HEX, "\\\\x41B", 12, "\\\\x41B", 6); + test_string_unescape_one(test, "backslash escape", UNESCAPE_HEX | UNESCAPE_SPECIAL, "\\\\x41B", 16, "\\x41B", 5); } static void test_escape(struct kunit *test) From 0a07900326a92bb0b4df77898d394198e73f88d8 Mon Sep 17 00:00:00 2001 From: Jonas Rebmann Date: Wed, 16 Sep 2026 19:38:09 +0200 Subject: [PATCH 0970/1012] lib/string_helpers: use full destination buffer in string_unescape() Although all of the available sequences expand to exactly one byte, the current implementation decrements the remaining bytes in the destination buffer twice, effectively shortening it by one byte per each unescaped character. The extra decrement is only needed in the one case where a single loop iteration produces two output bytes: when the sequence turns out not to be a valid escape sequence, the previously skipped backslash has to be emitted before the character is copied verbatim. Add a kunit regression-test that unescapes into a barely long enough 3 buffer. Link: https://lore.kernel.org/20260916-string_unescape-v1-4-7f8bd986fa33@pengutronix.de Fixes: 16c7fa05829e ("lib/string_helpers: introduce generic string_unescape") Signed-off-by: Jonas Rebmann Signed-off-by: Andrew Morton Cc: Andy Shevchenko Cc: Kees Cook Cc: Sascha Hauer --- lib/string_helpers.c | 2 +- lib/tests/string_helpers_kunit.c | 3 +++ 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/lib/string_helpers.c b/lib/string_helpers.c index 98d6ed0eaab7e9..cb41ef9d8c5bb9 100644 --- a/lib/string_helpers.c +++ b/lib/string_helpers.c @@ -331,7 +331,6 @@ int string_unescape(char *src, char *dst, size_t size, unsigned int flags) while (*src && --size) { if (src[0] == '\\' && src[1] != '\0' && size > 1) { src++; - size--; if (flags & UNESCAPE_SPACE && unescape_space(&src, &out)) @@ -350,6 +349,7 @@ int string_unescape(char *src, char *dst, size_t size, unsigned int flags) continue; *out++ = '\\'; + size--; } *out++ = *src++; } diff --git a/lib/tests/string_helpers_kunit.c b/lib/tests/string_helpers_kunit.c index 3c6fa7324965ce..2e02c680cbb21f 100644 --- a/lib/tests/string_helpers_kunit.c +++ b/lib/tests/string_helpers_kunit.c @@ -623,6 +623,9 @@ static void test_unescape(struct kunit *test) test_string_unescape_one(test, "escape at end", UNESCAPE_HEX, "a\\qX", 3, "a\\", 2); test_string_unescape_one(test, "backslash before escape", UNESCAPE_HEX, "\\\\x41B", 12, "\\\\x41B", 6); test_string_unescape_one(test, "backslash escape", UNESCAPE_HEX | UNESCAPE_SPECIAL, "\\\\x41B", 16, "\\x41B", 5); + + test_string_unescape_one(test, "short buffer", UNESCAPE_HEX, "\\x41\\x41B", 4, "AAB", 3); + test_string_unescape_one(test, "unrecognized escape at end", UNESCAPE_HEX, "B\\qX", 4, "B\\q", 3); } static void test_escape(struct kunit *test) From 0a1169120605a42140ac1693da76177a053091dc Mon Sep 17 00:00:00 2001 From: Jonas Rebmann Date: Wed, 16 Sep 2026 19:38:10 +0200 Subject: [PATCH 0971/1012] lib/string_helpers: fix counting of remaining bytes in string_unescape() All of the available sequences expand to exactly one byte, the size check in the loop condition is sufficient for the case of an escaped character too. Otherwise, an escape sequence that should be unescaped to the last character before terminating with null in the destination buffer will be output as backslash instead of the escaped character. The only exception is when encountering a backslash that turns out to not start a valid escape sequence and both the backslash and the character following are handled in one iteration. Move the check there. Add a kunit regression-test that unescapes a character to right in front of the null terminator of the destination buffer. Link: https://lore.kernel.org/20260916-string_unescape-v1-5-7f8bd986fa33@pengutronix.de Fixes: 16c7fa05829e ("lib/string_helpers: introduce generic string_unescape") Signed-off-by: Jonas Rebmann Signed-off-by: Andrew Morton Cc: Andy Shevchenko Cc: Kees Cook Cc: Sascha Hauer --- lib/string_helpers.c | 5 +++-- lib/tests/string_helpers_kunit.c | 2 ++ 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/lib/string_helpers.c b/lib/string_helpers.c index cb41ef9d8c5bb9..4a621f884bde3f 100644 --- a/lib/string_helpers.c +++ b/lib/string_helpers.c @@ -329,7 +329,7 @@ int string_unescape(char *src, char *dst, size_t size, unsigned int flags) size = SIZE_MAX; while (*src && --size) { - if (src[0] == '\\' && src[1] != '\0' && size > 1) { + if (src[0] == '\\' && src[1] != '\0') { src++; if (flags & UNESCAPE_SPACE && @@ -349,7 +349,8 @@ int string_unescape(char *src, char *dst, size_t size, unsigned int flags) continue; *out++ = '\\'; - size--; + if (!--size) + break; } *out++ = *src++; } diff --git a/lib/tests/string_helpers_kunit.c b/lib/tests/string_helpers_kunit.c index 2e02c680cbb21f..10763a01be83c2 100644 --- a/lib/tests/string_helpers_kunit.c +++ b/lib/tests/string_helpers_kunit.c @@ -626,6 +626,8 @@ static void test_unescape(struct kunit *test) test_string_unescape_one(test, "short buffer", UNESCAPE_HEX, "\\x41\\x41B", 4, "AAB", 3); test_string_unescape_one(test, "unrecognized escape at end", UNESCAPE_HEX, "B\\qX", 4, "B\\q", 3); + + test_string_unescape_one(test, "end of buffer", UNESCAPE_HEX, "B\\x41", 3, "BA", 2); } static void test_escape(struct kunit *test) From 2e062d9919b9da9f738ce9a26b795edeb0b3b7d9 Mon Sep 17 00:00:00 2001 From: Aamir Ahmed Date: Wed, 16 Sep 2026 09:53:51 +0100 Subject: [PATCH 0972/1012] lib: decompress_bunzip2: fix integer overflow in run-length decoding The RUNA/RUNB decoder in get_next_block() accumulates a run length into the signed int t with no bound. runPos doubles per symbol, so t grows as at least 2^n-1 and reaches INT_MAX after 31 RUNA symbols. The guard that follows, dbufCount+t >= dbufSize, is itself a signed addition, and as the kernel builds with -fno-strict-overflow it wraps negative once dbufCount is nonzero, so the guard is skipped. while (t--) dbuf[dbufCount++] = uc then writes INT_MAX entries into a buffer holding at most 900000. One literal symbol ahead of the run is enough to make dbufCount nonzero. Bound t to the block size as it is accumulated. t only grows within a run, so the value tested never exceeds the final run length; any stream that decodes today satisfies dbufCount+t < dbufSize at the flush, hence t < dbufSize throughout, and no such stream is rejected. Because t grows as at least 2^n-1 the bound trips by the 20th symbol, leaving runPos at most 2^19, so the following runPos <<= 1 cannot overflow either. Link: https://lore.kernel.org/AS8P251MB00010FE5E38D253CF98A652AC8B92@AS8P251MB0001.EURP251.PROD.OUTLOOK.COM Fixes: bc22c17e12c1 ("bzip2/lzma: library support for gzip, bzip2 and lzma decompression") Signed-off-by: Aamir Ahmed Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Alain Knaff Cc: "H. Peter Anvin" Cc: --- lib/decompress_bunzip2.c | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/lib/decompress_bunzip2.c b/lib/decompress_bunzip2.c index aaa75404250e2d..a9236d3e9678de 100644 --- a/lib/decompress_bunzip2.c +++ b/lib/decompress_bunzip2.c @@ -428,6 +428,10 @@ static int INIT get_next_block(struct bunzip_data *bd) t += (runPos << nextSym); /* +runPos if RUNA; +2*runPos if RUNB */ + /* Bound the run so t and runPos cannot overflow. */ + if (t >= dbufSize) + return RETVAL_DATA_ERROR; + runPos <<= 1; continue; } From 8a503cd723401d6afad490f3aec5f514f1d19733 Mon Sep 17 00:00:00 2001 From: Su Yue Date: Wed, 16 Sep 2026 10:47:50 +0800 Subject: [PATCH 0973/1012] ocfs2: update xattr count before moving bucket entries Since commit 2f26f58df041 ("ocfs2: annotate flexible array members with __counted_by_le()"), the xh_entries array is annotated with __counted_by_le(xh_count), so FORTIFY uses xh_count to determine its bounds. When inserting an entry into a bucket, ocfs2_xa_bucket_add_entry() shifts existing entries before incrementing xh_count. The destination therefore extends one entry past the bounds described by the old count. With CONFIG_CC_HAS_COUNTED_BY and CONFIG_FORTIFY_SOURCE enabled, ocfs2-test: single_run-WIP.sh -t reflink triggers the following failure while adding an extended attribute: [ 150.156484] memmove: detected buffer overflow: 352 byte write of buffer size 336 [ 150.156487] WARNING: lib/string_helpers.c:1036 at __fortify_report+0x3d/0x50, CPU#15: reflink_test/2336 [ 150.160496] RIP: 0010:__fortify_report+0x40/0x50 [ 150.164031] Call Trace: [ 150.164141] [ 150.164232] __fortify_panic+0x9/0xb [ 150.164383] ocfs2_xa_bucket_add_entry.cold+0x17/0x28 [ocfs2] [ 150.164661] ocfs2_xa_set+0x8fb/0xf40 [ocfs2] Increment xh_count before memmove() so the destination bounds include the new entry. Keep the local count unchanged to calculate the insertion position and move length from the original number of entries. The caller has already checked that there is enough space for the new entry. Link: https://lore.kernel.org/20260916024750.9450-1-glass.su@suse.com Fixes: 2f26f58df041 ("ocfs2: annotate flexible array members with __counted_by_le()") Signed-off-by: Su Yue Signed-off-by: Andrew Morton Reviewed-by: Heming Zhao Reviewed-by: Joseph Qi Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Changwei Ge Cc: Jun Piao Cc: --- fs/ocfs2/xattr.c | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index a2d6c3d0f2e8d8..9292cf26e2769b 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -2146,12 +2146,17 @@ static void ocfs2_xa_bucket_add_entry(struct ocfs2_xa_loc *loc, u32 name_hash) } } + /* + * Increment xh_count before memmove() so __counted_by_le(xh_count) + * includes the new entry in the destination bounds. + */ + le16_add_cpu(&xh->xh_count, 1); + if (low != count) memmove(&xh->xh_entries[low + 1], &xh->xh_entries[low], ((count - low) * sizeof(struct ocfs2_xattr_entry))); - le16_add_cpu(&xh->xh_count, 1); loc->xl_entry = &xh->xh_entries[low]; memset(loc->xl_entry, 0, sizeof(struct ocfs2_xattr_entry)); } From 490f4f41d78fbe7f1de9650e50d81ff0ba555306 Mon Sep 17 00:00:00 2001 From: Fang Xieyan Date: Thu, 17 Sep 2026 18:43:06 +0800 Subject: [PATCH 0974/1012] kcov: ignore an out-of-range comparison count in write_comp_data() write_comp_data() reads the comparison record count from area[0] and uses it to index the coverage buffer: area = (u64 *)t->kcov_area; max_pos = t->kcov_size * sizeof(unsigned long); count = READ_ONCE(area[0]); /* Every record is KCOV_WORDS_PER_CMP 64-bit words. */ start_index = 1 + count * KCOV_WORDS_PER_CMP; end_pos = (start_index + KCOV_WORDS_PER_CMP) * sizeof(u64); if (likely(end_pos <= max_pos)) { The buffer is mmap'd writable into the collecting process, so count is under its control and end_pos <= max_pos is its only bound. A count that wraps the u64 multiply leaves end_pos below max_pos, so the check passes while the record store lands 24 bytes before the buffer, in the unmapped vmalloc guard page, and faults: BUG: unable to handle page fault for address: ffa0000000b60fe8 #PF: supervisor write access in kernel mode #PF: error_code(0x0002) - not-present page Oops: 0002 [#1] SMP KASAN NOPTI RIP: 0010:write_comp_data+0x7e/0xa0 ... Kernel panic - not syncing: Fatal exception Bound count first: only max_pos / (sizeof(u64) * KCOV_WORDS_PER_CMP) records fit, so a larger count is not a valid index and is dropped. kcov_move_area() bounds the same untrusted count this way, and no count the end_pos <= max_pos check accepts reaches that limit, so no valid record is lost. kcov is a root-only debugfs file (debugfs_create_file_unsafe("kcov", 0600, ...)) and write_comp_data() exists only under CONFIG_KCOV_ENABLE_COMPARISONS, so this is a local, debug-kernel robustness fix: the process corrupts its own buffer and the kernel oopses. It crosses no privilege boundary. Additional details at [1] Link: https://lore.kernel.org/20260917104306.22145-1-fangxy@xiaopeng.com [1] Fixes: ded97d2c2b2c ("kcov: support comparison operands collection") Signed-off-by: Fang Xieyan Signed-off-by: Andrew Morton Reviewed-by: Alexander Potapenko Assisted-by: Hawkeye:GLM-5.3-flash Assisted-by: Qoder:Qwen3.8-Max Cc: Andrey Konovalov Cc: Dmitry Vyukov Cc: Marco Elver Cc: Victor Chibotaru Cc: --- kernel/kcov.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/kernel/kcov.c b/kernel/kcov.c index 35420f0ac524d4..54eaae97bfd67d 100644 --- a/kernel/kcov.c +++ b/kernel/kcov.c @@ -253,6 +253,14 @@ static void notrace write_comp_data(u64 type, u64 arg1, u64 arg2, u64 ip) count = READ_ONCE(area[0]); + /* + * area[0] is writable by the collecting process, so count cannot be + * trusted. Bound it to the records that fit, as kcov_move_area() + * does, so the end_pos multiply below cannot wrap past its check. + */ + if (count >= max_pos / (sizeof(u64) * KCOV_WORDS_PER_CMP)) + return; + /* Every record is KCOV_WORDS_PER_CMP 64-bit words. */ start_index = 1 + count * KCOV_WORDS_PER_CMP; end_pos = (start_index + KCOV_WORDS_PER_CMP) * sizeof(u64); From e76a7622a4216da5ab0eb9b21d12bd86a1bcb46e Mon Sep 17 00:00:00 2001 From: Frank Li Date: Thu, 17 Sep 2026 16:50:12 -0400 Subject: [PATCH 0975/1012] rapidio: use dmaengine_get_dma_device() instead of chan->device->dev Replace direct dma_chan::device::dev access with the proper dmaengine_get_dma_device() for consumer API. chan->device->dev is not always the device used for DMA mapping. Some DMA engines support per-channel IOMMU mappings, so different channels may use different DMA devices. dmaengine_get_dma_device() returns the correct device for each channel. This also prepares for making the DMA engine provider data structures private. DMA consumers should not access DMA engine internals directly. Link: https://lore.kernel.org/20260917205016.1293317-1-Frank.Li@oss.nxp.com Signed-off-by: Frank Li Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Alexandre Bounine Cc: Dan Carpenter Cc: Kees Cook Cc: Matt Porter Cc: Vinod Koul --- drivers/rapidio/devices/rio_mport_cdev.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/drivers/rapidio/devices/rio_mport_cdev.c b/drivers/rapidio/devices/rio_mport_cdev.c index 47ba34b4afb296..40d2953a9a7f9e 100644 --- a/drivers/rapidio/devices/rio_mport_cdev.c +++ b/drivers/rapidio/devices/rio_mport_cdev.c @@ -555,7 +555,7 @@ static void dma_req_free(struct kref *ref) refcount); struct mport_cdev_priv *priv = req->priv; - dma_unmap_sg(req->dmach->device->dev, + dma_unmap_sg(dmaengine_get_dma_device(req->dmach), req->sgt.sgl, req->sgt.nents, req->dir); sg_free_table(&req->sgt); if (req->page_list) { @@ -916,7 +916,7 @@ rio_dma_transfer(struct file *filp, u32 transfer_mode, xfer->offset, xfer->length); } - nents = dma_map_sg(chan->device->dev, + nents = dma_map_sg(dmaengine_get_dma_device(chan), req->sgt.sgl, req->sgt.nents, dir); if (nents == 0) { rmcd_error("Failed to map SG list"); From 2aa08fce251833e4d0e7f546a2992a6181f7b39b Mon Sep 17 00:00:00 2001 From: "Jose A. Perez de Azpillaga" Date: Mon, 21 Sep 2026 23:55:04 +0200 Subject: [PATCH 0976/1012] percpu_counter: annotate lockless read in _limited_add() syzbot reports a data race on fbc->count between the lockless read in the fast path of __percpu_counter_limited_add() and the locked update in percpu_counter_add_batch(), reached via shmem's used_blocks counter: the alloc path reads it through percpu_counter_limited_add() while the free path updates it through percpu_counter_sub(). Annotate the read rather than change the logic: the lockless read is deliberate, only the annotation is missing. It is an approximation, not a conservative bound. A concurrent flush moves value between fbc->count and a per-cpu counter, and other CPUs' locked slow paths add to fbc->count too, so a stale low value can let the fast path proceed where the slow path would refuse. The error is bounded by the per-cpu slack (unknown = batch * num_online_cpus()). This is existing behavior, no functional change intended. Read it with data_race(READ_ONCE(fbc->count)): data_race() tells KCSAN the race is intended (READ_ONCE() alone still triggers the report, because KCSAN reports against watchpoints set up by plain accesses), while READ_ONCE() stops the compiler from refetching the value, as in percpu_counter_read_positive(). The lock-protected accesses stay plain, so future buggy lockless writes are still caught. Link: https://lore.kernel.org/20260921215508.141641-1-azpijr@gmail.com Signed-off-by: Jose A. Perez de Azpillaga Signed-off-by: Andrew Morton Reported-by: syzbot+a3c71b9db9c11c270f59@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=a3c71b9db9c11c270f59 Reviewed-by: Andrew Morton Cc: Dennis Zhou Cc: Tejun Heo Cc: Christoph Lameter Cc: Hugh Dickins Cc: Baolin Wang --- lib/percpu_counter.c | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/lib/percpu_counter.c b/lib/percpu_counter.c index 2891f94a11c654..87b261b87b0bc8 100644 --- a/lib/percpu_counter.c +++ b/lib/percpu_counter.c @@ -328,6 +328,7 @@ bool __percpu_counter_limited_add(struct percpu_counter *fbc, s64 limit, s64 amount, s32 batch) { s64 count; + s64 gcount; s64 unknown; unsigned long flags; bool good = false; @@ -338,11 +339,16 @@ bool __percpu_counter_limited_add(struct percpu_counter *fbc, local_irq_save(flags); unknown = batch * num_online_cpus(); count = __this_cpu_read(*fbc->counters); + /* + * Lockless on purpose: gcount may be stale, so this is only an + * approximation, bounded by the per-cpu slack ("unknown"). + */ + gcount = data_race(READ_ONCE(fbc->count)); /* Skip taking the lock when safe */ if (abs(count + amount) <= batch && - ((amount > 0 && fbc->count + unknown <= limit) || - (amount < 0 && fbc->count - unknown >= limit))) { + ((amount > 0 && gcount + unknown <= limit) || + (amount < 0 && gcount - unknown >= limit))) { this_cpu_add(*fbc->counters, amount); local_irq_restore(flags); return true; From 8caf3f7a081bee9cbbe10dd828d3619a23598e8e Mon Sep 17 00:00:00 2001 From: Sang-Heon Jeon Date: Tue, 22 Sep 2026 22:26:27 +0900 Subject: [PATCH 0977/1012] init/main.c: remove unreachable check in obsolete_checksetup() Since commit 1ecfea06386c ("init.h: remove long-dead __setup_null_param() macro"), .init.setup entries cannot have a NULL setup_func. So remove the NULL check. No functional change. Link: https://lore.kernel.org/20260922132630.177941-1-ekffu200098@gmail.com Signed-off-by: Sang-Heon Jeon Signed-off-by: Andrew Morton --- init/main.c | 4 ---- 1 file changed, 4 deletions(-) diff --git a/init/main.c b/init/main.c index d1cd8efb823860..008a0a137f511d 100644 --- a/init/main.c +++ b/init/main.c @@ -215,10 +215,6 @@ static bool __init obsolete_checksetup(char *line) * params and __setups of same names 8( */ if (line[n] == '\0' || line[n] == '=') had_early_param = true; - } else if (!p->setup_func) { - pr_warn("Parameter %s is obsolete, ignored\n", - p->str); - return true; } else if (p->setup_func(line + n)) return true; } From 5f1d3c5d05633c9648570595f224127c2f2df0aa Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 22 Sep 2026 17:42:52 +0800 Subject: [PATCH 0978/1012] ocfs2: restore -ERANGE check in ocfs2_xattr_tree_list_index_block() ocfs2_xattr_list_entry() returns -ERANGE when the buffer handed in by the caller is too small to hold the whole xattr name list. That is the normal listxattr(2) protocol, userspace simply retries with a larger buffer, so it is not an ocfs2 error and must not be logged as one. Commit a46fa684fcb700 ("ocfs2: Don't printk the error when listing too many xattrs.") already knew this and silenced two sites, quoting (27738,0):ocfs2_iterate_xattr_buckets:3158 ERROR: status = -34 (27738,0):ocfs2_xattr_tree_list_index_block:3264 ERROR: status = -34 The refactoring in commit 47bca4950bc40f ("ocfs2: Abstract ocfs2 xattr tree extend rec iteration process.") then rewrote ocfs2_xattr_tree_list_index_block() around the new ocfs2_iterate_xattr_index_block() helper and dropped that guard, so the second message has been emitted ever since: ocfs2_xattr_tree_list_index_block:4513 ERROR: status = -34 The exemption in ocfs2_iterate_xattr_buckets() survived because that function was left alone, and the new helper does check for -ERANGE itself, which is why only the outermost caller still logs it. Restore the missing check. Only the log line changes: -ERANGE is still returned to the VFS untouched, so listxattr(2) behaves exactly as before. Link: https://lore.kernel.org/20260922094252.971631-1-joseph.qi@linux.alibaba.com Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Cc: Heming Zhao Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi --- fs/ocfs2/xattr.c | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/fs/ocfs2/xattr.c b/fs/ocfs2/xattr.c index 9292cf26e2769b..9f620f6c600518 100644 --- a/fs/ocfs2/xattr.c +++ b/fs/ocfs2/xattr.c @@ -4502,7 +4502,8 @@ static int ocfs2_xattr_tree_list_index_block(struct inode *inode, ret = ocfs2_iterate_xattr_index_block(inode, blk_bh, ocfs2_list_xattr_tree_rec, &xl); if (ret) { - mlog_errno(ret); + if (ret != -ERANGE) + mlog_errno(ret); goto out; } From e006c831243196606e8a24387ae552e96402fbd7 Mon Sep 17 00:00:00 2001 From: Joseph Qi Date: Tue, 22 Sep 2026 17:34:37 +0800 Subject: [PATCH 0979/1012] ocfs2: fix possible deadlock between nfs_sync_rwlock and fs_reclaim syzbot detected a circular locking dependency on &osb->nfs_sync_rwlock: CPU0 CPU1 ---- ---- lock(fs_reclaim); lock(&ocfs2_sysfile_lock_key[INODE_ALLOC_SYSTEM_INODE]); lock(fs_reclaim); rlock(&osb->nfs_sync_rwlock); *** DEADLOCK *** Chain exists of: &osb->nfs_sync_rwlock --> &ocfs2_sysfile_lock_key[INODE_ALLOC_SYSTEM_INODE] --> fs_reclaim CPU0 is kswapd. The dentry shrinker runs under fs_reclaim and reaches ocfs2_delete_inode() through ->evict_inode(): kswapd balance_pgdat shrink_node shrink_slab super_cache_scan prune_dcache_sb shrink_dentry_list dentry_kill ocfs2_dentry_iput evict ocfs2_evict_inode ocfs2_delete_inode ocfs2_nfs_sync_lock down_read(&osb->nfs_sync_rwlock) //C0: grabbing CPU1 is a task deleting an inode. It takes nfs_sync_rwlock first, then the orphan dir and inode alloc system inode i_rwsems, and allocates the metadata reservation with GFP_KERNEL while holding them: evict ocfs2_evict_inode ocfs2_delete_inode + ocfs2_nfs_sync_lock | down_read(&osb->nfs_sync_rwlock) //C1: hold + ocfs2_wipe_inode + inode_lock(orphan_dir_inode) //C1: hold + ocfs2_truncate_for_delete | ocfs2_commit_truncate | ocfs2_remove_btree_range | ocfs2_reserve_blocks_for_rec_trunc | ocfs2_reserve_new_metadata_blocks | kzalloc_obj() //C1: grabbing | // GFP_KERNEL -> might_alloc -> fs_reclaim_acquire + ocfs2_remove_inode inode_lock(inode_alloc_inode) jbd2 pins allocations to GFP_NOFS for the lifetime of a transaction handle, but these reservations are made before ocfs2_start_trans(), so fs_reclaim is acquired with ocfs2 locks still held and the cycle closes. The same cycle is reachable from ocfs2_get_dentry() and ocfs2_get_parent(), which hold nfs_sync_rwlock for write across ocfs2_test_inode_bit() and ocfs2_iget(). This is more than a lockdep artifact. A GFP_KERNEL allocation under the orphan dir i_rwsem enters direct reclaim, which runs the shrinkers, which can evict another ocfs2 inode and re-enter ocfs2_wipe_inode() on the same task; the second inode_lock() on the singleton orphan dir inode then self-deadlocks, since i_rwsem is not recursive. Establish a GFP_NOFS allocation context for the whole nfs_sync_rwlock critical section so that nothing holding it can recurse into filesystem reclaim. Do it in ocfs2_nfs_sync_lock()/ocfs2_nfs_sync_unlock() rather than at the call sites, so that future callers cannot forget it. Link: https://lore.kernel.org/20260922093437.954127-1-joseph.qi@linux.alibaba.com Fixes: 6ca497a83e59 ("ocfs2: fix rare stale inode errors when exporting via nfs") Signed-off-by: Joseph Qi Signed-off-by: Andrew Morton Reported-by: syzbot+a68ce48df87b8e36e915@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=a68ce48df87b8e36e915 Cc: Heming Zhao Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi --- fs/ocfs2/dlmglue.c | 29 +++++++++++++++++++++++++---- fs/ocfs2/dlmglue.h | 5 +++-- fs/ocfs2/export.c | 10 ++++++---- fs/ocfs2/inode.c | 5 +++-- 4 files changed, 37 insertions(+), 12 deletions(-) diff --git a/fs/ocfs2/dlmglue.c b/fs/ocfs2/dlmglue.c index cf3318b0d3a8c9..6bc4a21cf5f02f 100644 --- a/fs/ocfs2/dlmglue.c +++ b/fs/ocfs2/dlmglue.c @@ -18,6 +18,7 @@ #include #include #include +#include #include #include @@ -2861,21 +2862,34 @@ void ocfs2_rename_unlock(struct ocfs2_super *osb) ocfs2_cluster_unlock(osb, lockres, DLM_LOCK_EX); } -int ocfs2_nfs_sync_lock(struct ocfs2_super *osb, int ex) +int ocfs2_nfs_sync_lock(struct ocfs2_super *osb, int ex, unsigned int *nofs_flag) { int status; + unsigned int flags; struct ocfs2_lock_res *lockres = &osb->osb_nfs_sync_lockres; if (ocfs2_is_hard_readonly(osb)) return -EROFS; + /* + * ocfs2_delete_inode() takes this lock from ->evict_inode(), which the + * dentry shrinker reaches while holding fs_reclaim. Anything allocated + * under the lock must therefore stay out of filesystem reclaim, or + * reclaim recurses back into the shrinker and tries to take this lock + * again. Cover the whole critical section, including the cluster lock + * and the sysfile inode locks the callers take below it. + */ + flags = memalloc_nofs_save(); + if (ex) down_write(&osb->nfs_sync_rwlock); else down_read(&osb->nfs_sync_rwlock); - if (ocfs2_mount_local(osb)) + if (ocfs2_mount_local(osb)) { + *nofs_flag = flags; return 0; + } status = ocfs2_cluster_lock(osb, lockres, ex ? LKM_EXMODE : LKM_PRMODE, 0, 0); @@ -2886,12 +2900,17 @@ int ocfs2_nfs_sync_lock(struct ocfs2_super *osb, int ex) up_write(&osb->nfs_sync_rwlock); else up_read(&osb->nfs_sync_rwlock); + memalloc_nofs_restore(flags); + return status; } - return status; + *nofs_flag = flags; + + return 0; } -void ocfs2_nfs_sync_unlock(struct ocfs2_super *osb, int ex) +void ocfs2_nfs_sync_unlock(struct ocfs2_super *osb, int ex, + unsigned int nofs_flag) { struct ocfs2_lock_res *lockres = &osb->osb_nfs_sync_lockres; @@ -2902,6 +2921,8 @@ void ocfs2_nfs_sync_unlock(struct ocfs2_super *osb, int ex) up_write(&osb->nfs_sync_rwlock); else up_read(&osb->nfs_sync_rwlock); + + memalloc_nofs_restore(nofs_flag); } int ocfs2_trim_fs_lock(struct ocfs2_super *osb, diff --git a/fs/ocfs2/dlmglue.h b/fs/ocfs2/dlmglue.h index a3ebd7303ea20b..faa58ed8699b1c 100644 --- a/fs/ocfs2/dlmglue.h +++ b/fs/ocfs2/dlmglue.h @@ -161,8 +161,9 @@ void ocfs2_orphan_scan_unlock(struct ocfs2_super *osb, u32 seqno); int ocfs2_rename_lock(struct ocfs2_super *osb); void ocfs2_rename_unlock(struct ocfs2_super *osb); -int ocfs2_nfs_sync_lock(struct ocfs2_super *osb, int ex); -void ocfs2_nfs_sync_unlock(struct ocfs2_super *osb, int ex); +int ocfs2_nfs_sync_lock(struct ocfs2_super *osb, int ex, unsigned int *nofs_flag); +void ocfs2_nfs_sync_unlock(struct ocfs2_super *osb, int ex, + unsigned int nofs_flag); void ocfs2_trim_fs_lock_res_init(struct ocfs2_super *osb); void ocfs2_trim_fs_lock_res_uninit(struct ocfs2_super *osb); int ocfs2_trim_fs_lock(struct ocfs2_super *osb, diff --git a/fs/ocfs2/export.c b/fs/ocfs2/export.c index 9c2665dd24e218..90b9e510e700d5 100644 --- a/fs/ocfs2/export.c +++ b/fs/ocfs2/export.c @@ -37,6 +37,7 @@ static struct dentry *ocfs2_get_dentry(struct super_block *sb, struct inode *inode; struct ocfs2_super *osb = OCFS2_SB(sb); u64 blkno = handle->ih_blkno; + unsigned int nofs_flag = 0; int status, set; struct dentry *result; @@ -59,7 +60,7 @@ static struct dentry *ocfs2_get_dentry(struct super_block *sb, * This will synchronize us against ocfs2_delete_inode() on * all nodes */ - status = ocfs2_nfs_sync_lock(osb, 1); + status = ocfs2_nfs_sync_lock(osb, 1, &nofs_flag); if (status < 0) { mlog(ML_ERROR, "getting nfs sync lock(EX) failed %d\n", status); goto check_err; @@ -90,7 +91,7 @@ static struct dentry *ocfs2_get_dentry(struct super_block *sb, inode = ocfs2_iget(osb, blkno, 0, 0); unlock_nfs_sync: - ocfs2_nfs_sync_unlock(osb, 1); + ocfs2_nfs_sync_unlock(osb, 1, nofs_flag); check_err: if (status < 0) { @@ -133,12 +134,13 @@ static struct dentry *ocfs2_get_parent(struct dentry *child) u64 blkno; struct dentry *parent; struct inode *dir = d_inode(child); + unsigned int nofs_flag = 0; int set; trace_ocfs2_get_parent(child, child->d_name.len, child->d_name.name, (unsigned long long)OCFS2_I(dir)->ip_blkno); - status = ocfs2_nfs_sync_lock(OCFS2_SB(dir->i_sb), 1); + status = ocfs2_nfs_sync_lock(OCFS2_SB(dir->i_sb), 1, &nofs_flag); if (status < 0) { mlog(ML_ERROR, "getting nfs sync lock(EX) failed %d\n", status); parent = ERR_PTR(status); @@ -183,7 +185,7 @@ static struct dentry *ocfs2_get_parent(struct dentry *child) ocfs2_inode_unlock(dir, 0); unlock_nfs_sync: - ocfs2_nfs_sync_unlock(OCFS2_SB(dir->i_sb), 1); + ocfs2_nfs_sync_unlock(OCFS2_SB(dir->i_sb), 1, nofs_flag); bail: trace_ocfs2_get_parent_end(parent); diff --git a/fs/ocfs2/inode.c b/fs/ocfs2/inode.c index 92f3450010fbb0..80c36c60a8cef9 100644 --- a/fs/ocfs2/inode.c +++ b/fs/ocfs2/inode.c @@ -1109,6 +1109,7 @@ static void ocfs2_delete_inode(struct inode *inode) { int wipe, status; sigset_t oldset; + unsigned int nofs_flag = 0; struct buffer_head *di_bh = NULL; struct ocfs2_dinode *di = NULL; @@ -1143,7 +1144,7 @@ static void ocfs2_delete_inode(struct inode *inode) * shared mode so that all nodes can still concurrently * process deletes. */ - status = ocfs2_nfs_sync_lock(OCFS2_SB(inode->i_sb), 0); + status = ocfs2_nfs_sync_lock(OCFS2_SB(inode->i_sb), 0, &nofs_flag); if (status < 0) { mlog(ML_ERROR, "getting nfs sync lock(PR) failed %d\n", status); ocfs2_cleanup_delete_inode(inode, 0); @@ -1215,7 +1216,7 @@ static void ocfs2_delete_inode(struct inode *inode) brelse(di_bh); bail_unlock_nfs_sync: - ocfs2_nfs_sync_unlock(OCFS2_SB(inode->i_sb), 0); + ocfs2_nfs_sync_unlock(OCFS2_SB(inode->i_sb), 0, nofs_flag); bail_unblock: ocfs2_unblock_signals(&oldset); From 0f4befd0caddd0ac783a4058a1b53057f3ff4d74 Mon Sep 17 00:00:00 2001 From: Nguyen Ngoc Thang Date: Sun, 20 Sep 2026 21:07:10 +0700 Subject: [PATCH 0980/1012] ocfs2: validate chunk and block counts of the local quota file ocfs2_local_read_info() trusts dqi_chunks and dqi_blocks from the local quota file header. If dqi_blocks is too small, ocfs2_extend_local_quota_file() computes a bogus chunk length, extends the file, and maps logical block dqi_blocks, which is already in the inode's metadata cache (the header or a chunk header read at mount). sb_getblk() returns that cached buffer and ocfs2_set_new_buffer_uptodate() hits BUG_ON(ocfs2_buffer_cached()): kernel BUG at fs/ocfs2/uptodate.c:509! ocfs2_extend_local_quota_file+0x45c/0x1100 ocfs2_create_local_dquot+0x8ac/0xb50 ocfs2_acquire_dquot+0x614/0xae0 ocfs2_get_init_inode+0xe9/0x1b0 ocfs2_mkdir+0x174/0x430 ocfs2_local_quota_add_chunk() has the same exposure. Reject a header whose counts don't fit the chunk layout or exceed i_size. Reproduced with a crafted image (dqi_blocks = 1, chunk 0 dqc_free = 0); syzbot has no reproducer for this report. Link: https://lore.kernel.org/20260920140710.43938-1-ngocthang2710.1999@gmail.com Fixes: 9e33d69f553a ("ocfs2: Implementation of local and global quota file handling") Signed-off-by: Nguyen Ngoc Thang Signed-off-by: Andrew Morton Reported-by: syzbot+03aaa576f1daa1c1f2f0@syzkaller.appspotmail.com Closes: https://syzkaller.appspot.com/bug?extid=03aaa576f1daa1c1f2f0 Reviewed-by: Joseph Qi Cc: Mark Fasheh Cc: Joel Becker Cc: Junxiao Bi Cc: Heming Zhao --- fs/ocfs2/quota_local.c | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/fs/ocfs2/quota_local.c b/fs/ocfs2/quota_local.c index d351cda9211f85..76e7dd5aecc86b 100644 --- a/fs/ocfs2/quota_local.c +++ b/fs/ocfs2/quota_local.c @@ -245,6 +245,25 @@ static void ocfs2_release_local_quota_bitmaps(struct list_head *head) } } +/* Check that the on-disk chunk and block counts match the file layout */ +static int ocfs2_check_local_quota_info(struct inode *inode, + unsigned int chunks, + unsigned int blocks) +{ + u64 chunk_len = ol_chunk_blocks(inode->i_sb) + 1; + u64 min_blocks = chunks ? 1 + chunk_len * (chunks - 1) + 1 : 1; + u64 max_blocks = 1 + chunk_len * chunks; + + if (blocks >= min_blocks && blocks <= max_blocks && + blocks <= i_size_read(inode) >> inode->i_sb->s_blocksize_bits) + return 0; + + return ocfs2_error(inode->i_sb, + "Quota file %llu has bad info: %u chunks, %u blocks\n", + (unsigned long long)OCFS2_I(inode)->ip_blkno, + chunks, blocks); +} + /* Load quota bitmaps into memory */ static int ocfs2_load_local_quota_bitmaps(struct inode *inode, struct ocfs2_local_disk_dqinfo *ldinfo, @@ -733,6 +752,11 @@ static int ocfs2_local_read_info(struct super_block *sb, int type) oinfo->dqi_blocks = le32_to_cpu(ldinfo->dqi_blocks); oinfo->dqi_libh = bh; + status = ocfs2_check_local_quota_info(lqinode, oinfo->dqi_chunks, + oinfo->dqi_blocks); + if (status < 0) + goto out_err; + /* We crashed when using local quota file? */ if (!(oinfo->dqi_flags & OLQF_CLEAN)) { rec = OCFS2_SB(sb)->quota_rec; From 36aa67192daba09aee750ee6054913730b4b6df3 Mon Sep 17 00:00:00 2001 From: Giorgi Kobakhia Date: Tue, 22 Sep 2026 13:21:59 -0700 Subject: [PATCH 0981/1012] ocfs2: fix array out of bound access in __ocfs2_find_path() __ocfs2_find_path() walks down the extent tree and records path by calling find_path_ins(), which appends entry to path->p_node[]. It only has 5 spots. A corrupted ocfs2 image whose extent block is pointing to itself causes __ocfs2_find_path() descent endlessly, writing past the end of path->p_node[] array. UBSAN: array-index-out-of-bounds in fs/ocfs2/alloc.c:677:14 index 5 is out of range for type 'ocfs2_path_item [5]' Call Trace: find_path_ins (fs/ocfs2/alloc.c:677 fs/ocfs2/alloc.c:1914) __ocfs2_find_path.constprop.0 (fs/ocfs2/alloc.c:1882) ocfs2_commit_truncate (fs/ocfs2/alloc.c:1924 fs/ocfs2/alloc.c:7286) ocfs2_truncate_file (fs/ocfs2/file.c:515) ocfs2_setattr (fs/ocfs2/file.c:1224) notify_change (fs/attr.c:556) do_truncate (fs/open.c:68) do_ftruncate (fs/open.c:194 (discriminator 1)) ksys_ftruncate (fs/open.c:206) __x64_sys_ftruncate (fs/open.c:211 fs/open.c:209 fs/open.c:209) Commit a406aff8c051 ("ocfs2: validate l_tree_depth to avoid out-of-bounds access") already restricts el->l_tree_depth to be less than OCFS2_MAX_PATH_DEPTH, which is equal to 5. However, does not handle the infinite descent case. Check if the el->l_tree_depth decreases on each descent. Maximum descents are restricted to 4 and the path->p_node[] array does not overflow. Link: https://lore.kernel.org/20260922202159.2421642-1-gkobakhi@asu.edu Fixes: dcd0538ff4e8 ("ocfs2: sparse b-tree support") Signed-off-by: Giorgi Kobakhia Signed-off-by: Andrew Morton Tested-by: Xiang Mei Reviewed-by: Joseph Qi Assisted-by: LLM Cc: Mark Fasheh Cc: Joel Becker Cc: --- fs/ocfs2/alloc.c | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/fs/ocfs2/alloc.c b/fs/ocfs2/alloc.c index 2fdc5403b10b04..fc949e59462699 100644 --- a/fs/ocfs2/alloc.c +++ b/fs/ocfs2/alloc.c @@ -1844,6 +1844,7 @@ static int __ocfs2_find_path(struct ocfs2_caching_info *ci, int i, ret = 0; u32 range; u64 blkno; + u32 prev_depth = OCFS2_MAX_PATH_DEPTH; struct buffer_head *bh = NULL; struct ocfs2_extent_block *eb; struct ocfs2_extent_list *el; @@ -1851,14 +1852,16 @@ static int __ocfs2_find_path(struct ocfs2_caching_info *ci, el = root_el; while (el->l_tree_depth) { - if (unlikely(le16_to_cpu(el->l_tree_depth) >= OCFS2_MAX_PATH_DEPTH)) { + if (unlikely(le16_to_cpu(el->l_tree_depth) >= prev_depth)) { ocfs2_error(ocfs2_metadata_cache_get_super(ci), - "Owner %llu has invalid tree depth %u in extent list\n", + "Owner %llu has invalid tree depth %u in extent list (max %u)\n", (unsigned long long)ocfs2_metadata_cache_owner(ci), - le16_to_cpu(el->l_tree_depth)); + le16_to_cpu(el->l_tree_depth), prev_depth - 1); ret = -EROFS; goto out; } + prev_depth = le16_to_cpu(el->l_tree_depth); + if (!el->l_next_free_rec || !el->l_count) { ocfs2_error(ocfs2_metadata_cache_get_super(ci), "Owner %llu has empty extent list at depth %u\n" From d1884019dc6d96335bb2a8817fe3e9e2c62f69b9 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Wed, 23 Sep 2026 03:01:49 +0200 Subject: [PATCH 0982/1012] ocfs2/cluster: hold a reference on the heartbeat thread Since commit 688bc88e2046 ("ocfs2/cluster: keep heartbeat local node stable"), o2hb_thread() leaves its loop and returns when the local node changes, for example after "echo 0 > node//local". The thread was started with kthread_run() and nothing holds a reference to its task_struct, so the task is freed once it exits, while reg->hr_task still points to it. Reading the region's pid attribute then reads the freed task, and removing the region calls kthread_stop() on it: BUG: KASAN: slab-use-after-free in o2hb_region_pid_show+0xb3/0xc0 refcount_t: addition on 0; use-after-free. Oops: Oops: 0000 [#1] SMP KASAN NOPTI RIP: 0010:kthread_stop+0xb1/0x390 The thread could already return by itself before, when heartbeat start was aborted or on an unclean stop, but the local node change makes it reachable from userspace at any time. Create the thread, take a reference on it and only then wake it, so the reference cannot race with the thread exiting. Drop it with kthread_stop_put(). Link: https://lore.kernel.org/20260923010149.14391-1-kmehltretter@gmail.com Fixes: 688bc88e2046 ("ocfs2/cluster: keep heartbeat local node stable") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Reported-by: Sashiko Closes: https://sashiko.dev/#/patchset/20260616074931.3774929-1-zzzccc427%40gmail.com Reviewed-by: Joseph Qi Assisted-by: LLM Cc: Mark Fasheh Cc: Joel Becker Cc: Cen Zhang Cc: Junxiao Bi Cc: Heming Zhao --- fs/ocfs2/cluster/heartbeat.c | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/fs/ocfs2/cluster/heartbeat.c b/fs/ocfs2/cluster/heartbeat.c index 1c3def99bb0765..a4c8ea695f5cf1 100644 --- a/fs/ocfs2/cluster/heartbeat.c +++ b/fs/ocfs2/cluster/heartbeat.c @@ -1966,18 +1966,22 @@ static ssize_t o2hb_region_dev_store(struct config_item *item, atomic_set(®->hr_unsteady_iterations, (live_threshold * 3)); o2hb_set_region_stopping(reg, false); - hb_task = kthread_run(o2hb_thread, reg, "o2hb-%s", - reg->hr_item.ci_name); + hb_task = kthread_create(o2hb_thread, reg, "o2hb-%s", + reg->hr_item.ci_name); if (IS_ERR(hb_task)) { ret = PTR_ERR(hb_task); mlog_errno(ret); goto out; } + /* The thread may exit on its own, so pin it before it can run. */ + get_task_struct(hb_task); spin_lock(&o2hb_live_lock); reg->hr_task = hb_task; spin_unlock(&o2hb_live_lock); + wake_up_process(hb_task); + ret = wait_event_interruptible(o2hb_steady_queue, atomic_read(®->hr_steady_iterations) == 0 || reg->hr_node_deleted); @@ -2022,7 +2026,7 @@ static ssize_t o2hb_region_dev_store(struct config_item *item, spin_unlock(&o2hb_live_lock); if (hb_task) - kthread_stop(hb_task); + kthread_stop_put(hb_task); o2hb_unmap_slot_data(reg); @@ -2208,7 +2212,7 @@ static void o2hb_heartbeat_group_drop_item(struct config_group *group, spin_unlock(&o2hb_live_lock); if (hb_task) - kthread_stop(hb_task); + kthread_stop_put(hb_task); if (o2hb_global_heartbeat_active()) { spin_lock(&o2hb_live_lock); From a1b876904f7341cb424014b9dd8f87b036c5c74e Mon Sep 17 00:00:00 2001 From: Guixin Liu Date: Thu, 24 Sep 2026 11:39:23 +0800 Subject: [PATCH 0983/1012] checkpatch: don't flag ACQUIRE_ERR() assignments in if conditions ACQUIRE_ERR() and its wrappers, PM_RUNTIME_ACQUIRE_ERR() and IIO_DEV_ACQUIRE_FAILED(), report whether a conditional cleanup.h guard was acquired, and drivers consume the result directly in an if condition: if ((rc = ACQUIRE_ERR(mutex_intr, &lock))) return rc; That combined form is the established style at the 49 in-tree call sites under drivers/cxl and drivers/pci/tsm.c, so ASSIGN_IN_IF fires there only as a false positive, and every patch touching those lines carries noise that reviewers have to wave off manually. Skip the check only when every assignment in the condition assigns the result of such a call, matched by the *_ACQUIRE_ERR() / *_ACQUIRE_FAILED() naming convention of its wrappers. Plain assignments, mixed conditions and near-miss identifiers still get flagged. Link: https://lore.kernel.org/20260924033923.4140210-1-kanie@linux.alibaba.com Signed-off-by: Guixin Liu Signed-off-by: Andrew Morton Suggested-by: Alison Schofield Acked-by: Joe Perches Assisted-by: LLM Cc: Andy Whitcroft Cc: Jonathan Cameron --- scripts/checkpatch.pl | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/scripts/checkpatch.pl b/scripts/checkpatch.pl index 3614cfe4dcbb45..598e3ed743bd9e 100755 --- a/scripts/checkpatch.pl +++ b/scripts/checkpatch.pl @@ -5787,7 +5787,21 @@ sub process { my ($s, $c) = ($stat, $cond); my $fixed_assign_in_if = 0; + # ACQUIRE_ERR() and its wrappers, e.g. PM_RUNTIME_ACQUIRE_ERR() + # and IIO_DEV_ACQUIRE_FAILED(), are meant to be evaluated in an + # if condition, with the error assigned in the condition: + # if ((rc = ACQUIRE_ERR(name, &lock))) + # Allow that only when every assignment in the condition assigns + # the result of such a call, so that a mixed condition keeps + # getting flagged: + # if ((rc = regular_function()) || (ret = ACQUIRE_ERR(name, &lock))) + my $assign_in_if = 0; if ($c =~ /\bif\s*\(.*[^<>!=]=[^=].*/s) { + my $has_assignment = $c =~ /\b$Lval\s*=\s*[^,)&|=]+/; + my $has_other_assignment = $c =~ /\b$Lval\s*=\s*(?!\s*\w*ACQUIRE_(?:ERR|FAILED)\s*\()[^,)&|=]+/; + $assign_in_if = !$has_assignment || $has_other_assignment; + } + if ($assign_in_if) { if (ERROR("ASSIGN_IN_IF", "do not use assignment in if condition\n" . $herecurr) && $fix && $perl_version_ok) { From 9893f245a9f5ce76a119d37bd3eda9486040dc5f Mon Sep 17 00:00:00 2001 From: Andy Yan Date: Thu, 24 Sep 2026 18:50:33 +0800 Subject: [PATCH 0984/1012] mailmap: update entry for Andy Yan I will use andyshrk@163.com for future review and discussion. Link: https://lore.kernel.org/20260924105052.768760-1-andyshrk@163.com Signed-off-by: Andy Yan Signed-off-by: Andrew Morton Cc: Alexander Sverdlin Cc: Chuck Lever Cc: Jakub Kicinski Cc: Martin Kepplinger --- .mailmap | 1 + 1 file changed, 1 insertion(+) diff --git a/.mailmap b/.mailmap index 1a288972949802..3f9b79dac77fe8 100644 --- a/.mailmap +++ b/.mailmap @@ -103,6 +103,7 @@ Andy Chiu Andy Chiu Andy Shevchenko Andy Shevchenko +Andy Yan Anilkumar Kolli Anirudh Ghayal Antoine Tenart From 106f328ca4d6ed172bb17e14612eaf4f0eae0b42 Mon Sep 17 00:00:00 2001 From: Danish Khateeb Date: Fri, 25 Sep 2026 15:55:02 -0500 Subject: [PATCH 0985/1012] kselftest/filelock: plan for the five ofdlocks tests ofdlocks reports five results but plans for four, so even when they all pass it ends with # Planned tests != run tests (4 != 5) and exits with KSFT_FAIL, which run_kselftest.sh reports as a failure. Link: https://lore.kernel.org/20260925205502.115327-1-danishkhateeb03@gmail.com Fixes: 33d5b13098fb ("kselftest/filelock: report each test in oftlocks separately") Signed-off-by: Danish Khateeb Signed-off-by: Andrew Morton Assisted-by: LLM Cc: Shuah Khan Cc: Mark Brown Cc: Jeff Layton --- tools/testing/selftests/filelock/ofdlocks.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tools/testing/selftests/filelock/ofdlocks.c b/tools/testing/selftests/filelock/ofdlocks.c index 68bac28b234b7d..0ab484cb075bc7 100644 --- a/tools/testing/selftests/filelock/ofdlocks.c +++ b/tools/testing/selftests/filelock/ofdlocks.c @@ -40,7 +40,7 @@ int main(void) int fd2 = open("/tmp/aa", O_RDONLY); ksft_print_header(); - ksft_set_plan(4); + ksft_set_plan(5); unlink("/tmp/aa"); assert(fd != -1); From faa714760c621b192aa388782a636a4e7a38b217 Mon Sep 17 00:00:00 2001 From: Matthias Goergens Date: Mon, 28 Sep 2026 06:49:58 +0800 Subject: [PATCH 0986/1012] kdev_t: shift in dev_t in MKDEV() MKDEV(ma, mi) evaluates ((ma) << MINORBITS) in whatever type the caller passes, usually int. For majors >= 2048 (legal: majors go up to 4095) the result exceeds INT_MAX, which C11 leaves undefined for a signed left shift; shifting a negative ma is undefined regardless of the major's value. In practice, on the compilers and two's complement targets the kernel supports, both cases wrap to the intended bit pattern; this patch makes the macro well defined regardless by casting the major to dev_t before shifting. For major and minor numbers in range, the device number it produces is unchanged. isofs feeds the Rock Ridge 'PN' entry's dev_high and dev_low into MKDEV() while holding them in int variables (fs/isofs/rock.c), so a crafted image can drive this exact shift. Fixing the macro covers every caller instead of just that one. Link: https://lore.kernel.org/20260927224958.508964-1-matthias.goergens@gmail.com Signed-off-by: Matthias Goergens Signed-off-by: Andrew Morton Reviewed-by: Jan Kara Cc: Greg Kroah-Hartman --- include/linux/kdev_t.h | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/include/linux/kdev_t.h b/include/linux/kdev_t.h index 4856706fbfeb45..2dbbd47f1e68e1 100644 --- a/include/linux/kdev_t.h +++ b/include/linux/kdev_t.h @@ -9,7 +9,7 @@ #define MAJOR(dev) ((unsigned int) ((dev) >> MINORBITS)) #define MINOR(dev) ((unsigned int) ((dev) & MINORMASK)) -#define MKDEV(ma,mi) (((ma) << MINORBITS) | (mi)) +#define MKDEV(ma, mi) (((dev_t)(ma) << MINORBITS) | (mi)) #define print_dev_t(buffer, dev) \ sprintf((buffer), "%u:%u\n", MAJOR(dev), MINOR(dev)) From df3c608a271ccc14f82c14b18c61c394e20340c7 Mon Sep 17 00:00:00 2001 From: Karl Mehltretter Date: Sat, 26 Sep 2026 02:26:51 +0200 Subject: [PATCH 0987/1012] resource, kunit: stop selecting GET_FREE_REGION RESOURCE_KUNIT_TEST selects GET_FREE_REGION even when no other option needs it. Remove the selection to follow the dependency rule in Documentation/dev-tools/kunit/style.rst. Skip resource_test_region_intersects() when GET_FREE_REGION is disabled. GET_FREE_REGION has no prompt, so configurations without a production consumer cannot enable it. Most configurations will therefore skip this case; this is intentional. The union and intersection tests remain available. Since kunit_skip() does not return, the compiler drops the reference to the unavailable alloc_free_mem_region(), as in the CONFIG_OF_ADDRESS check in drivers/of/of_test.c. Link: https://lore.kernel.org/20260926002651.87267-1-kmehltretter@gmail.com Fixes: 99185c10d5d9 ("resource, kunit: add test case for region_intersects()") Signed-off-by: Karl Mehltretter Signed-off-by: Andrew Morton Reviewed-by: Bradley Morgan Tested-by: Bradley Morgan # Power10 Assisted-by: LLM Cc: Ying Huang Cc: Geert Uytterhoeven Cc: Brendan Higgins Cc: David Gow Cc: Rae Moar Cc: Andy Shevchenko --- kernel/resource_kunit.c | 3 +++ lib/Kconfig.debug | 1 - 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/kernel/resource_kunit.c b/kernel/resource_kunit.c index 42785796f1dbc5..9eeb2b1a85c01c 100644 --- a/kernel/resource_kunit.c +++ b/kernel/resource_kunit.c @@ -225,6 +225,9 @@ static void resource_test_region_intersects(struct kunit *test) struct resource *parent; resource_size_t start; + if (!IS_ENABLED(CONFIG_GET_FREE_REGION)) + kunit_skip(test, "CONFIG_GET_FREE_REGION is disabled"); + /* Find an iomem_resource hole to hold test resources */ parent = alloc_free_mem_region(&iomem_resource, RES_TEST_TOTAL_SIZE, SZ_1M, "test resources"); diff --git a/lib/Kconfig.debug b/lib/Kconfig.debug index c3f448f3b8f13c..56228eefdb4a9d 100644 --- a/lib/Kconfig.debug +++ b/lib/Kconfig.debug @@ -2776,7 +2776,6 @@ config RESOURCE_KUNIT_TEST tristate "KUnit test for resource API" if !KUNIT_ALL_TESTS depends on KUNIT default KUNIT_ALL_TESTS - select GET_FREE_REGION help This builds the resource API unit test. Tests the logic of API provided by resource.c and ioport.h. From dccce570524243ed8a53706e3ab0fe9ff1177523 Mon Sep 17 00:00:00 2001 From: Jim Cromie Date: Tue, 29 Sep 2026 12:07:30 -0600 Subject: [PATCH 0988/1012] kallsyms: match compressed tokens on the fly during binary search Patch series "kallsyms: Accelerate symbol name lookups by ~7x", v7. In 2022, commit 60443c88f3a8 ("kallsyms: Improve the performance of kallsyms_lookup_name()") introduced kallsyms_seqs_of_names[] (+550 KiB .rodata), transforming an O(N) linear scan into an O(log N) binary search (5.2 ms -> ~7.2 us). While this was a major step forward, the binary search inner loop was left decompressing full candidate names and scanning across sparse 256:1 markers on every probe. Modern fleet observability, security daemons (e.g. CrowdStrike Falcon, Cilium, Datadog, Falco), and tracing tools resolve thousands of kernel functions by name at boot or service start. CrowdStrike recently hit this in production: commit 93e8fd1a565e ("ftrace: Use kallsyms binary search for single-symbol lookup") Attaching just 50 kprobe.session programs caused an 858 ms attach stall with 25% CPU burned in kallsyms. That commit routed single-symbol libbpf attach directly to kallsyms_lookup_name(). In larger workloads (such as the BPF selftest serial_test_kprobe_multi_bench_attach across 64,000 symbols), kallsyms_lookup_names() spends ~390 ms in raw CPU spin. This series accelerates kallsyms_lookup_names() by 7.0x (from 6,102 ns down to 866 ns per lookup), cutting 64k-symbol attach from ~390 ms to ~55 ms, by fixing two inner-loop bottlenecks: 0. Candidate symbols are fully decompressed into a 512-byte stack buffer before calling strcmp(), even though ~16 of the 17 search steps mismatch at the first differing character (0..N-1, heavily front-loaded toward 0-2). 1. Probes scan sequentially from 256:1 markers in kallsyms_names[], decoding an average of 127.5 symbols per probe (~2,170 hops across a 17-step search). The 3-patch progression: 0. Patch 1 introduces kallsyms_strcmp_symbol() to compare ASCII queries against compressed tokens on the fly, bailing out on first mismatch. Drops the 512-byte stack buffer and saves ~530 ns. 1. Patch 2 increases marker density from 256:1 to 16:1, cutting average scan distance from 127.5 to 7.5 hops and dropping lookup latency from 6,102 ns to 866 ns for +42.2 KiB of .rodata. 2. Patch 3 inlines and unrolls get_symbol_seq() 24-bit reconstruction. Results (CONFIG_KALLSYMS_SELFTEST across ~184k symbols): - Baseline (256:1): 6,102 ns - Patch 1 (strcmp): 5,572 ns (-530 ns) - Patch 2 (16:1): 866 ns (7.0x faster) Trade-offs: - .rodata footprint: +42.2 KiB (+10,782 u32 entries for ~184k symbols, ~0.1% of loaded kernel image). - Runtime overhead: 0 bytes dynamic RAM (no kmalloc/kvmalloc), 0 RCU, 0 new locks, and 0 new algorithms. - Kernel stack: -512 bytes freed in kallsyms_lookup_names(). - Build tooling: scripts/kallsyms.c includes ../kernel/kallsyms_internal.h (guarded by #ifdef __KERNEL__) so host build and kernel runtime share KALLSYMS_MARKER_SHIFT 4 as a single source of truth. - Build time: Unmeasurable delta (< 1 ms in scripts/kallsyms.c). What's Unchanged: - Symbol table layout in address order remains identical. - Streaming decompression for /proc/kallsyms and sprint_symbol() is untouched. - 0 new user-facing APIs, 0 new locking primitives, 0 Kconfig options. This patch (of 3): kallsyms_lookup_names() runs a binary search across ~184k tokenized (compressed) symbols. For each of the ~17 comparisons in the search, it currently decompresses the candidate symbol into a temporary buffer on the stack before calling strcmp(). Comparing tokenized symbols directly in compressed space is impossible. The BPE token table assigns values by frequency, not alphabetical order (e.g. token 0x05 might expand to "zebra" while 0x42 expands to "apple"), so comparing raw token values scrambles lexicographical order. Even sorting the token table alphabetically wouldn't help; "bpf_" and "bpf_foo_" do not *have* a determinative sorting order, because the suffixes following those tokens would matter. However, full string expansion at every step is equally wasteful: of the ~17 strcmps in the binary search, only the last needs to check all N chars in both strings, earlier steps will know +/- outcome at char 0,1,2..N-1. However, full string expansion at every step is equally wasteful: of the ~17 strcmp()s in the binary search, only the final matching step needs to test all characters. Earlier non-matching steps diverge at the first differing character (0..N-1), but the baseline expands every candidate symbol to the stack unconditionally, before comparing. So we introduce kallsyms_strcmp_symbol() to compare ASCII search_name against tokenized symbols on the fly. Like strcmp, it tests the strings char by char, but when it hits a token in the symbol-string, it continues the char-test against that token-string, which is in kallsyms_token_table[]. It returns +- on 1st mismatch. Measured across all ~184k symbols via CONFIG_KALLSYMS_SELFTEST, this shaves ~530 ns (~14%) off average kallsyms_lookup_name() latency (from ~3810 ns to ~3280 ns on the default 256:1 baseline) and drops the 512-byte namebuf buffer stack-alloc in kallsyms_lookup_names(). Link: https://lore.kernel.org/20260929-ksyms-tune-v7-0-be568ceef41e@gmail.com Link: https://lore.kernel.org/20260929-ksyms-tune-v7-1-be568ceef41e@gmail.com Signed-off-by: Jim Cromie Signed-off-by: Andrew Morton Cc: Petr Mladek Cc: Zhen Lei Cc: Luis Chamberlain Cc: Andrey Grodzovsky Cc: Steven Rostedt Cc: Lorenzo Stoakes Cc: Kees Cook Cc: David Laight Cc: Masahiro Yamada Cc: Jiri Olsa Cc: Geert Uytterhoeven --- kernel/kallsyms.c | 94 +++++++++++++++++++++++++++++------------------ 1 file changed, 59 insertions(+), 35 deletions(-) diff --git a/kernel/kallsyms.c b/kernel/kallsyms.c index aec2f06858afdb..d18d78e626db26 100644 --- a/kernel/kallsyms.c +++ b/kernel/kallsyms.c @@ -34,6 +34,21 @@ #include "kallsyms_internal.h" +/* + * Get the compressed symbol length and data pointer. + */ +static inline const u8 *get_symbol_data(unsigned int off, unsigned int *len) +{ + const u8 *p = &kallsyms_names[off]; + unsigned int l = *p++; + + if (unlikely(l & 0x80)) + l = (l & 0x7F) | (*p++ << 7); + *len = l; + + return p; +} + /* * Expand a compressed symbol data into the resulting uncompressed string, * if uncompressed string is too long (>= maxlen), it will be truncated, @@ -42,28 +57,12 @@ static unsigned int kallsyms_expand_symbol(unsigned int off, char *result, size_t maxlen) { - int len, skipped_first = 0; + int skipped_first = 0; const char *tptr; - const u8 *data; + unsigned int len; + const u8 *data = get_symbol_data(off, &len); - /* Get the compressed symbol length from the first symbol byte. */ - data = &kallsyms_names[off]; - len = *data; - data++; - off++; - - /* If MSB is 1, it is a "big" symbol, so needs an additional byte. */ - if ((len & 0x80) != 0) { - len = (len & 0x7F) | (*data << 7); - data++; - off++; - } - - /* - * Update the offset to return the offset for the next symbol on - * the compressed stream. - */ - off += len; + off = (data - kallsyms_names) + len; /* * For every byte on the compressed symbol data, copy the table @@ -101,14 +100,43 @@ static unsigned int kallsyms_expand_symbol(unsigned int off, */ static char kallsyms_get_symbol_type(unsigned int off) { - /* - * Get just the first code, look it up in the token table, - * and return the first char from this token. If MSB of length - * is 1, it is a "big" symbol, so needs an additional byte. - */ - if (kallsyms_names[off] & 0x80) - off++; - return kallsyms_token_table[kallsyms_token_index[kallsyms_names[off + 1]]]; + unsigned int len; + const u8 *data = get_symbol_data(off, &len); + + return kallsyms_token_table[kallsyms_token_index[*data]]; +} + +/* + * Compare an uncompressed ASCII string against a compressed symbol table entry. + * Returns negative if name < sym, positive if name > sym, 0 if equal. + * Exits immediately on the first mismatched character without decompressing + * the rest of the symbol name. + */ +static int kallsyms_strcmp_symbol(unsigned int off, const char *name) +{ + const char *tptr; + unsigned int len; + const u8 *data = get_symbol_data(off, &len); + + tptr = &kallsyms_token_table[kallsyms_token_index[*data++]] + 1; + while (*tptr) { + int diff = (unsigned char)*name++ - (unsigned char)*tptr++; + + if (diff) + return diff; + } + + while (--len) { + tptr = &kallsyms_token_table[kallsyms_token_index[*data++]]; + do { + int diff = (unsigned char)*name++ - (unsigned char)*tptr++; + + if (diff) + return diff; + } while (*tptr); + } + + return (unsigned char)*name; } @@ -174,7 +202,6 @@ static int kallsyms_lookup_names(const char *name, int ret; int low, mid, high; unsigned int seq, off; - char namebuf[KSYM_NAME_LEN]; low = 0; high = kallsyms_num_syms - 1; @@ -183,8 +210,7 @@ static int kallsyms_lookup_names(const char *name, mid = low + (high - low) / 2; seq = get_symbol_seq(mid); off = get_symbol_offset(seq); - kallsyms_expand_symbol(off, namebuf, ARRAY_SIZE(namebuf)); - ret = strcmp(name, namebuf); + ret = kallsyms_strcmp_symbol(off, name); if (ret > 0) low = mid + 1; else if (ret < 0) @@ -200,8 +226,7 @@ static int kallsyms_lookup_names(const char *name, while (low) { seq = get_symbol_seq(low - 1); off = get_symbol_offset(seq); - kallsyms_expand_symbol(off, namebuf, ARRAY_SIZE(namebuf)); - if (strcmp(name, namebuf)) + if (kallsyms_strcmp_symbol(off, name) != 0) break; low--; } @@ -212,8 +237,7 @@ static int kallsyms_lookup_names(const char *name, while (high < kallsyms_num_syms - 1) { seq = get_symbol_seq(high + 1); off = get_symbol_offset(seq); - kallsyms_expand_symbol(off, namebuf, ARRAY_SIZE(namebuf)); - if (strcmp(name, namebuf)) + if (kallsyms_strcmp_symbol(off, name) != 0) break; high++; } From edd7005a42d42e8337fc4cae09118e5a53c12817 Mon Sep 17 00:00:00 2001 From: Jim Cromie Date: Tue, 29 Sep 2026 12:07:31 -0600 Subject: [PATCH 0989/1012] kallsyms: increase marker density to 16:1 to accelerate lookups kallsyms stores symbols with remarkably efficient packing, and simple streaming unpacking, laid out sequentially in address order. That said, variable-length records make arbitrary access inherently linear. kallsyms_markers[] addressed this by marking stream offsets every 256 symbols, reducing the scan distance by 256x down to an average of 127.5 sequential steps. While 127.5 hops was negligible for rare, single-shot oops backtraces, both table size (~184k symbols) and lookup traffic have expanded substantially. In alphabetical binary search (kallsyms_lookup_names), each of the ~17 comparison probes must locate candidate symbols via get_symbol_offset(), compounding into ~2,170 sequential symbol hops per lookup. In bulk tracing workloads (such as BPF multi-kprobe attach), this penalty compounds into multi-second latency. Without altering the underlying storage layout, we can retune this trade-off directly by increasing marker density from 256:1 down to 16:1 (KALLSYMS_MARKER_SHIFT 4) in kernel/kallsyms_internal.h, shared between scripts/kallsyms.c and kernel/kallsyms.c. This caps the remainder scan at 15 symbols and cuts average scan distance from 127.5 down to 7.5 hops (a 17x reduction). Across a 17-step binary search, total hops collapse from ~2,170 down to ~127. For a kernel with ~184,000 symbols, this adds ~10,800 u32 marker entries (+42 KiB) to write-protected .rodata. In-tree CONFIG_KALLSYMS_SELFTEST measurements across all ~184k symbols show average lookup latency dropping from 6,102 ns down to 866 ns (a 7.0x speedup). Link: https://lore.kernel.org/20260929-ksyms-tune-v7-2-be568ceef41e@gmail.com Signed-off-by: Jim Cromie Signed-off-by: Andrew Morton Cc: Petr Mladek Cc: Zhen Lei Cc: Luis Chamberlain Cc: Andrey Grodzovsky Cc: Steven Rostedt Cc: Lorenzo Stoakes Cc: Kees Cook Cc: David Laight Cc: Masahiro Yamada Cc: Jiri Olsa Cc: Geert Uytterhoeven --- kernel/kallsyms.c | 8 ++++---- kernel/kallsyms_internal.h | 11 +++++++++++ scripts/kallsyms.c | 14 +++++++++----- 3 files changed, 24 insertions(+), 9 deletions(-) diff --git a/kernel/kallsyms.c b/kernel/kallsyms.c index d18d78e626db26..91ced7aa797eb7 100644 --- a/kernel/kallsyms.c +++ b/kernel/kallsyms.c @@ -150,10 +150,10 @@ static unsigned int get_symbol_offset(unsigned long pos) int i, len; /* - * Use the closest marker we have. We have markers every 256 positions, - * so that should be close enough. + * Use the closest marker we have. We have markers every + * (1 << KALLSYMS_MARKER_SHIFT) positions, so that should be close enough. */ - name = &kallsyms_names[kallsyms_markers[pos >> 8]]; + name = &kallsyms_names[kallsyms_markers[pos >> KALLSYMS_MARKER_SHIFT]]; /* * Sequentially scan all the symbols up to the point we're searching @@ -161,7 +161,7 @@ static unsigned int get_symbol_offset(unsigned long pos) * so we just need to add the len to the current pointer for every * symbol we wish to skip. */ - for (i = 0; i < (pos & 0xFF); i++) { + for (i = 0; i < (pos & KALLSYMS_MARKER_MASK); i++) { len = *name; /* diff --git a/kernel/kallsyms_internal.h b/kernel/kallsyms_internal.h index 81a867dbe57d48..6a781e4cc77feb 100644 --- a/kernel/kallsyms_internal.h +++ b/kernel/kallsyms_internal.h @@ -2,6 +2,16 @@ #ifndef LINUX_KALLSYMS_INTERNAL_H_ #define LINUX_KALLSYMS_INTERNAL_H_ +/* + * Provide compile-constants for scripts/kallsyms.c + * so it can build the corresponding kallsyms_marker[] table. + * and wrap the rest in __KERNEL__ + */ +#define KALLSYMS_MARKER_SHIFT 4 +#define KALLSYMS_MARKER_SIZE (1U << KALLSYMS_MARKER_SHIFT) +#define KALLSYMS_MARKER_MASK (KALLSYMS_MARKER_SIZE - 1U) + +#ifdef __KERNEL__ #include extern const int kallsyms_offsets[]; @@ -14,5 +24,6 @@ extern const u16 kallsyms_token_index[]; extern const unsigned int kallsyms_markers[]; extern const u8 kallsyms_seqs_of_names[]; +#endif /* __KERNEL__ */ #endif // LINUX_KALLSYMS_INTERNAL_H_ diff --git a/scripts/kallsyms.c b/scripts/kallsyms.c index 494852ade6d87a..be42a911135007 100644 --- a/scripts/kallsyms.c +++ b/scripts/kallsyms.c @@ -29,6 +29,8 @@ #include +#include "../kernel/kallsyms_internal.h" + #define ARRAY_SIZE(arr) (sizeof(arr) / sizeof(arr[0])) #define KSYM_NAME_LEN 512 @@ -349,16 +351,18 @@ static void write_src(void) printf("\t.long\t%u\n", table_cnt); printf("\n"); - /* table of offset markers, that give the offset in the compressed stream - * every 256 symbols */ - markers_cnt = (table_cnt + 255) / 256; + /* + * Table of offset markers, giving the offset in the compressed stream + * every (1 << KALLSYMS_MARKER_SHIFT) symbols. + */ + markers_cnt = (table_cnt + KALLSYMS_MARKER_MASK) >> KALLSYMS_MARKER_SHIFT; markers = xmalloc(sizeof(*markers) * markers_cnt); output_label("kallsyms_names"); off = 0; for (i = 0; i < table_cnt; i++) { - if ((i & 0xFF) == 0) - markers[i >> 8] = off; + if ((i & KALLSYMS_MARKER_MASK) == 0) + markers[i >> KALLSYMS_MARKER_SHIFT] = off; table[i]->seq = i; /* There cannot be any symbol of length zero. */ From 692de2dc10b438e41d0d680e9777420d57631715 Mon Sep 17 00:00:00 2001 From: Jim Cromie Date: Tue, 29 Sep 2026 12:07:32 -0600 Subject: [PATCH 0990/1012] kallsyms: unroll 24-bit sequence reconstruction in get_symbol_seq() kallsyms_seqs_of_names[] stores 3-byte big-endian sequence indices that map alphabetical symbol positions to address-ordered symbol records. Currently, get_symbol_seq() reconstructs each 24-bit integer using a 3-iteration for-loop that shifts and bitwise-ORs each byte sequentially. During binary search in kallsyms_lookup_names() and duplicate boundary scans, this loop introduces branch and loop overhead on the hot lookup path. Mark get_symbol_seq() as static inline and unroll the 3-byte extraction into direct byte shifts: (p[0] << 16) | (p[1] << 8) | p[2]. This eliminates loop induction variable maintenance and allows the compiler to generate direct loads and constant shifts. Link: https://lore.kernel.org/20260929-ksyms-tune-v7-3-be568ceef41e@gmail.com Signed-off-by: Jim Cromie Signed-off-by: Andrew Morton Cc: Petr Mladek Cc: Zhen Lei Cc: Luis Chamberlain Cc: Andrey Grodzovsky Cc: Steven Rostedt Cc: Lorenzo Stoakes Cc: Kees Cook Cc: David Laight Cc: Masahiro Yamada Cc: Jiri Olsa Cc: Geert Uytterhoeven --- kernel/kallsyms.c | 9 +++------ 1 file changed, 3 insertions(+), 6 deletions(-) diff --git a/kernel/kallsyms.c b/kernel/kallsyms.c index 91ced7aa797eb7..52e41879c24bda 100644 --- a/kernel/kallsyms.c +++ b/kernel/kallsyms.c @@ -185,14 +185,11 @@ unsigned long kallsyms_sym_address(int idx) return (unsigned long)offset_to_ptr(kallsyms_offsets + idx); } -static unsigned int get_symbol_seq(int index) +static inline unsigned int get_symbol_seq(int index) { - unsigned int i, seq = 0; + const u8 *p = &kallsyms_seqs_of_names[3 * index]; - for (i = 0; i < 3; i++) - seq = (seq << 8) | kallsyms_seqs_of_names[3 * index + i]; - - return seq; + return (p[0] << 16) | (p[1] << 8) | p[2]; } static int kallsyms_lookup_names(const char *name, From cbb17a7a0afe4f2a766b25e75777f70f47dbc6ba Mon Sep 17 00:00:00 2001 From: Yuntao Wang Date: Wed, 30 Sep 2026 09:57:48 +0800 Subject: [PATCH 0991/1012] init: simplify initramfs.o build rule CONFIG_BLK_DEV_INITRD is known to be enabled in the else branch, so there is no need to use obj-$(CONFIG_BLK_DEV_INITRD). Use obj-y directly to make the build rule clearer. Also, reorder the condition to make the code easier to follow. No functional change. Link: https://lore.kernel.org/20260930015748.311366-1-yuntao.wang@linux.dev Signed-off-by: Yuntao Wang Signed-off-by: Andrew Morton Cc: Nathan Chancellor Cc: Nicolas Schier --- init/Makefile | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/init/Makefile b/init/Makefile index d6f75d8907e098..9102fc34c0cc0f 100644 --- a/init/Makefile +++ b/init/Makefile @@ -6,10 +6,10 @@ ccflags-y := -fno-function-sections -fno-data-sections obj-y := main.o version.o mounts.o -ifneq ($(CONFIG_BLK_DEV_INITRD),y) -obj-y += noinitramfs.o +ifeq ($(CONFIG_BLK_DEV_INITRD),y) +obj-y += initramfs.o else -obj-$(CONFIG_BLK_DEV_INITRD) += initramfs.o +obj-y += noinitramfs.o endif obj-$(CONFIG_GENERIC_CALIBRATE_DELAY) += calibrate.o obj-$(CONFIG_INITRAMFS_TEST) += initramfs_test.o From e68f74594fdba6b901ca404f583af74c89b7b3d8 Mon Sep 17 00:00:00 2001 From: Pagadala Yesu Anjaneyulu Date: Sat, 26 Sep 2026 20:17:25 +0300 Subject: [PATCH 0992/1012] wifi: iwlwifi: fw: harden UEFI reduced-power TLV parsing UEFI reduced-power parsing advanced by ALIGN(tlv_len, 4) but validated only tlv_len. When remaining bytes were between these values, len could underflow and parsing could continue past the buffer boundary. Validate remaining bytes against the aligned length before advancing in both TLV walkers. Also reject short PNVM_SKU TLVs before reading sku_id fields. Signed-off-by: Pagadala Yesu Anjaneyulu Assisted-by: GitHubCopilot:GPT-5.3-Codex Link: https://patch.msgid.link/20260926201527.2552432e98e8.I570750f27d9530d4c76922d2699a272d9c60a0ab@changeid Signed-off-by: Miri Korenblit --- drivers/net/wireless/intel/iwlwifi/fw/uefi.c | 49 +++++++++++++++----- 1 file changed, 37 insertions(+), 12 deletions(-) diff --git a/drivers/net/wireless/intel/iwlwifi/fw/uefi.c b/drivers/net/wireless/intel/iwlwifi/fw/uefi.c index 86825318c54389..dffd5e68f41da9 100644 --- a/drivers/net/wireless/intel/iwlwifi/fw/uefi.c +++ b/drivers/net/wireless/intel/iwlwifi/fw/uefi.c @@ -251,15 +251,24 @@ static int iwl_uefi_reduce_power_section(struct iwl_trans *trans, memset(pnvm_data, 0, sizeof(*pnvm_data)); while (len >= sizeof(*tlv)) { - u32 tlv_len, tlv_type; + u32 tlv_len, tlv_type, tlv_len_aligned; len -= sizeof(*tlv); tlv = (const void *)data; tlv_len = le32_to_cpu(tlv->length); tlv_type = le32_to_cpu(tlv->type); + tlv_len_aligned = ALIGN(tlv_len, 4); - if (len < tlv_len) { + /* Make sure ALIGN() did not overflow for a malformed TLV length. */ + if (tlv_len_aligned < tlv_len) { + IWL_ERR(trans, + "TLV len overflows on alignment: %u\n", + tlv_len); + return -EINVAL; + } + + if (len < tlv_len_aligned) { IWL_ERR(trans, "invalid TLV len: %zd/%u\n", len, tlv_len); return -EINVAL; @@ -278,13 +287,13 @@ static int iwl_uefi_reduce_power_section(struct iwl_trans *trans, "New REDUCE_POWER section started, stop parsing.\n"); goto done; default: - IWL_DEBUG_FW(trans, "Found TLV 0x%0x, len %d\n", + IWL_DEBUG_FW(trans, "Found TLV 0x%0x, len %u\n", tlv_type, tlv_len); break; } - len -= ALIGN(tlv_len, 4); - data += ALIGN(tlv_len, 4); + len -= tlv_len_aligned; + data += tlv_len_aligned; } done: @@ -305,15 +314,24 @@ int iwl_uefi_reduce_power_parse(struct iwl_trans *trans, IWL_DEBUG_FW(trans, "Parsing REDUCE_POWER data\n"); while (len >= sizeof(*tlv)) { - u32 tlv_len, tlv_type; + u32 tlv_len, tlv_type, tlv_len_aligned; len -= sizeof(*tlv); tlv = (const void *)data; tlv_len = le32_to_cpu(tlv->length); tlv_type = le32_to_cpu(tlv->type); + tlv_len_aligned = ALIGN(tlv_len, 4); + + /* Make sure ALIGN() did not overflow for a malformed TLV length. */ + if (tlv_len_aligned < tlv_len) { + IWL_ERR(trans, + "TLV len overflows on alignment: %u\n", + tlv_len); + return -EINVAL; + } - if (len < tlv_len) { + if (len < tlv_len_aligned) { IWL_ERR(trans, "invalid TLV len: %zd/%u\n", len, tlv_len); return -EINVAL; @@ -323,8 +341,15 @@ int iwl_uefi_reduce_power_parse(struct iwl_trans *trans, const struct iwl_sku_id *tlv_sku_id = (const void *)(data + sizeof(*tlv)); + if (tlv_len < sizeof(*tlv_sku_id)) { + IWL_ERR(trans, + "Invalid IWL_UCODE_TLV_PNVM_SKU len %u\n", + tlv_len); + return -EINVAL; + } + IWL_DEBUG_FW(trans, - "Got IWL_UCODE_TLV_PNVM_SKU len %d\n", + "Got IWL_UCODE_TLV_PNVM_SKU len %u\n", tlv_len); if (tlv_len < sizeof(*tlv_sku_id)) { IWL_ERR(trans, "invalid PNVM SKU TLV len: %u\n", @@ -337,8 +362,8 @@ int iwl_uefi_reduce_power_parse(struct iwl_trans *trans, le32_to_cpu(tlv_sku_id->data[1]), le32_to_cpu(tlv_sku_id->data[2])); - data += sizeof(*tlv) + ALIGN(tlv_len, 4); - len -= ALIGN(tlv_len, 4); + data += sizeof(*tlv) + tlv_len_aligned; + len -= tlv_len_aligned; if (sku_id[0] == tlv_sku_id->data[0] && sku_id[1] == tlv_sku_id->data[1] && @@ -352,8 +377,8 @@ int iwl_uefi_reduce_power_parse(struct iwl_trans *trans, IWL_DEBUG_FW(trans, "SKU ID didn't match!\n"); } } else { - data += sizeof(*tlv) + ALIGN(tlv_len, 4); - len -= ALIGN(tlv_len, 4); + data += sizeof(*tlv) + tlv_len_aligned; + len -= tlv_len_aligned; } } From 811ab7b55e08d108a664ce7534d9417f93b1dc86 Mon Sep 17 00:00:00 2001 From: Pagadala Yesu Anjaneyulu Date: Sat, 26 Sep 2026 20:17:26 +0300 Subject: [PATCH 0993/1012] wifi: iwlwifi: fw: add Samsung to TAS and PPAG allow lists Allow platforms reporting the exact manufacturer string "Samsung" to use TAS and PPAG, while retaining the existing Samsung Electronics entry. Signed-off-by: Pagadala Yesu Anjaneyulu Link: https://patch.msgid.link/20260926201527.e2e2e1f720af.I18a7292b5a8bd878d074b114b45ce99e1309897e@changeid Signed-off-by: Miri Korenblit --- drivers/net/wireless/intel/iwlwifi/fw/regulatory.c | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/drivers/net/wireless/intel/iwlwifi/fw/regulatory.c b/drivers/net/wireless/intel/iwlwifi/fw/regulatory.c index 27a5804dbd73a0..c7de45323935e8 100644 --- a/drivers/net/wireless/intel/iwlwifi/fw/regulatory.c +++ b/drivers/net/wireless/intel/iwlwifi/fw/regulatory.c @@ -49,11 +49,16 @@ static const struct dmi_system_id dmi_ppag_approved_list[] = { DMI_MATCH(DMI_SYS_VENDOR, "HP"), }, }, - { .ident = "SAMSUNG", + { .ident = "SAMSUNG_ELECTRONICS", .matches = { DMI_MATCH(DMI_SYS_VENDOR, "SAMSUNG ELECTRONICS CO., LTD"), }, }, + { .ident = "SAMSUNG", + .matches = { + DMI_EXACT_MATCH(DMI_SYS_VENDOR, "Samsung"), + }, + }, { .ident = "MSFT", .matches = { DMI_MATCH(DMI_SYS_VENDOR, "Microsoft Corporation"), @@ -126,10 +131,15 @@ static const struct dmi_system_id dmi_tas_approved_list[] = { DMI_MATCH(DMI_SYS_VENDOR, "HP"), }, }, - { .ident = "SAMSUNG", + { .ident = "SAMSUNG_ELECTRONICS", .matches = { DMI_MATCH(DMI_SYS_VENDOR, "SAMSUNG ELECTRONICS CO., LTD"), }, + }, + { .ident = "SAMSUNG", + .matches = { + DMI_EXACT_MATCH(DMI_SYS_VENDOR, "Samsung"), + }, }, { .ident = "LENOVO", .matches = { From 9e4440b00bee502c2257dd5ffb941725b758c91e Mon Sep 17 00:00:00 2001 From: Emmanuel Grumbach Date: Sat, 26 Sep 2026 20:17:27 +0300 Subject: [PATCH 0994/1012] wifi: iwlwifi: pcie: order RX reads after the write pointer iwl_pcie_rx_handle() reads the last closed RB index from rb_stts and then reads the completion descriptor and the RB contents that this index makes visible. Nothing orders those two reads: there is no barrier, the descriptor address derives from rxq->read rather than from the index just read, so there is no address dependency, and the "while (i != r)" test is only a control dependency, which does not order loads on arm64. The CPU can therefore speculate past the loop test, perform the completion descriptor load early and sample the value from before the device's DMA, then resolve the rb_stts load to the value from after it. On arm64 (Jetson AGX Orin) this showed up as recurring "Invalid rxb from HW " naming an in-range rbid the driver still owned, and "frame on invalid queue" for an RB tagged with the queue that used it before the wrap. Both force an NMI and a firmware restart. x86 never shows it because loads are not reordered with loads there. Add a dma_rmb() after reading the write pointer, as other NIC drivers do when consuming a DMA descriptor ring. It is a no-op on x86 and a dmb on arm64. Verified to stop the warnings on Orin. Signed-off-by: Emmanuel Grumbach Assisted-by: GitHub-Copilot:claude-opus-5 Link: https://patch.msgid.link/20260926201527.1bff8cebef69.I24dec81516e37a33674e87ed3359c41e59d12e5e@changeid Signed-off-by: Miri Korenblit --- drivers/net/wireless/intel/iwlwifi/pcie/rx.c | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/drivers/net/wireless/intel/iwlwifi/pcie/rx.c b/drivers/net/wireless/intel/iwlwifi/pcie/rx.c index eb70922cf51607..d1ef0365b932cf 100644 --- a/drivers/net/wireless/intel/iwlwifi/pcie/rx.c +++ b/drivers/net/wireless/intel/iwlwifi/pcie/rx.c @@ -1519,6 +1519,11 @@ static int iwl_pcie_rx_handle(struct iwl_trans *trans, int queue, int budget) r = iwl_get_closed_rb_stts(trans, rxq); i = rxq->read; + /* Order the read of the write pointer before any read of the + * completion descriptors and of the RB contents it makes visible. + */ + dma_rmb(); + /* W/A 9000 device step A0 wrap-around bug */ r &= (rxq->queue_size - 1); From 4425a91007332fc1e2779142f0dbc90c7961d2d4 Mon Sep 17 00:00:00 2001 From: Miri Korenblit Date: Sat, 26 Sep 2026 20:17:28 +0300 Subject: [PATCH 0995/1012] wifi: iwlwifi: pcie: remove iwl_dbgfs_fh_reg_read This is no longer needed, and adds lots of parsing code Reviewed-by: Emmanuel Grumbach Link: https://patch.msgid.link/20260926201527.3e2f229ac9f6.I122eaf364743ffc42bd18a84a766d7920ec5700d@changeid Signed-off-by: Miri Korenblit --- drivers/net/wireless/intel/iwlwifi/iwl-io.c | 177 ------------------ drivers/net/wireless/intel/iwlwifi/iwl-io.h | 3 - .../net/wireless/intel/iwlwifi/pcie/trans.c | 20 -- 3 files changed, 200 deletions(-) diff --git a/drivers/net/wireless/intel/iwlwifi/iwl-io.c b/drivers/net/wireless/intel/iwlwifi/iwl-io.c index bb746112ddadf5..544b8e686dfe8f 100644 --- a/drivers/net/wireless/intel/iwlwifi/iwl-io.c +++ b/drivers/net/wireless/intel/iwlwifi/iwl-io.c @@ -11,7 +11,6 @@ #include "iwl-csr.h" #include "iwl-debug.h" #include "iwl-prph.h" -#include "iwl-fh.h" void iwl_write8(struct iwl_trans *trans, u32 ofs, u8 val) { @@ -235,182 +234,6 @@ void iwl_force_nmi(struct iwl_trans *trans) } IWL_EXPORT_SYMBOL(iwl_force_nmi); -static const char *get_rfh_string(int cmd) -{ -#define IWL_CMD(x) case x: return #x -#define IWL_CMD_MQ(arg, reg, q) { if (arg == reg(q)) return #reg; } - - int i; - - for (i = 0; i < IWL_MAX_RX_HW_QUEUES; i++) { - IWL_CMD_MQ(cmd, RFH_Q_FRBDCB_BA_LSB, i); - IWL_CMD_MQ(cmd, RFH_Q_FRBDCB_WIDX, i); - IWL_CMD_MQ(cmd, RFH_Q_FRBDCB_RIDX, i); - IWL_CMD_MQ(cmd, RFH_Q_URBD_STTS_WPTR_LSB, i); - } - - switch (cmd) { - IWL_CMD(RFH_RXF_DMA_CFG); - IWL_CMD(RFH_GEN_CFG); - IWL_CMD(RFH_GEN_STATUS); - IWL_CMD(FH_TSSR_TX_STATUS_REG); - IWL_CMD(FH_TSSR_TX_ERROR_REG); - default: - return "UNKNOWN"; - } -#undef IWL_CMD_MQ -} - -struct reg { - u32 addr; - bool is64; -}; - -static int iwl_dump_rfh(struct iwl_trans *trans, char **buf) -{ - int i, q; - int num_q = trans->info.num_rxqs; - static const u32 rfh_tbl[] = { - RFH_RXF_DMA_CFG, - RFH_GEN_CFG, - RFH_GEN_STATUS, - FH_TSSR_TX_STATUS_REG, - FH_TSSR_TX_ERROR_REG, - }; - static const struct reg rfh_mq_tbl[] = { - { RFH_Q0_FRBDCB_BA_LSB, true }, - { RFH_Q0_FRBDCB_WIDX, false }, - { RFH_Q0_FRBDCB_RIDX, false }, - { RFH_Q0_URBD_STTS_WPTR_LSB, true }, - }; - -#ifdef CONFIG_IWLWIFI_DEBUGFS - if (buf) { - int pos = 0; - /* - * Register (up to 34 for name + 8 blank/q for MQ): 40 chars - * Colon + space: 2 characters - * 0X%08x: 10 characters - * New line: 1 character - * Total of 53 characters - */ - size_t bufsz = ARRAY_SIZE(rfh_tbl) * 53 + - ARRAY_SIZE(rfh_mq_tbl) * 53 * num_q + 40; - - *buf = kmalloc(bufsz, GFP_KERNEL); - if (!*buf) - return -ENOMEM; - - pos += scnprintf(*buf + pos, bufsz - pos, - "RFH register values:\n"); - - for (i = 0; i < ARRAY_SIZE(rfh_tbl); i++) - pos += scnprintf(*buf + pos, bufsz - pos, - "%40s: 0X%08x\n", - get_rfh_string(rfh_tbl[i]), - iwl_read_prph(trans, rfh_tbl[i])); - - for (i = 0; i < ARRAY_SIZE(rfh_mq_tbl); i++) - for (q = 0; q < num_q; q++) { - u32 addr = rfh_mq_tbl[i].addr; - - addr += q * (rfh_mq_tbl[i].is64 ? 8 : 4); - pos += scnprintf(*buf + pos, bufsz - pos, - "%34s(q %2d): 0X%08x\n", - get_rfh_string(addr), q, - iwl_read_prph(trans, addr)); - } - - return pos; - } -#endif - - IWL_ERR(trans, "RFH register values:\n"); - for (i = 0; i < ARRAY_SIZE(rfh_tbl); i++) - IWL_ERR(trans, " %34s: 0X%08x\n", - get_rfh_string(rfh_tbl[i]), - iwl_read_prph(trans, rfh_tbl[i])); - - for (i = 0; i < ARRAY_SIZE(rfh_mq_tbl); i++) - for (q = 0; q < num_q; q++) { - u32 addr = rfh_mq_tbl[i].addr; - - addr += q * (rfh_mq_tbl[i].is64 ? 8 : 4); - IWL_ERR(trans, " %34s(q %d): 0X%08x\n", - get_rfh_string(addr), q, - iwl_read_prph(trans, addr)); - } - - return 0; -} - -static const char *get_fh_string(int cmd) -{ - switch (cmd) { - IWL_CMD(FH_RSCSR_CHNL0_STTS_WPTR_REG); - IWL_CMD(FH_RSCSR_CHNL0_RBDCB_BASE_REG); - IWL_CMD(FH_RSCSR_CHNL0_WPTR); - IWL_CMD(FH_MEM_RCSR_CHNL0_CONFIG_REG); - IWL_CMD(FH_MEM_RSSR_SHARED_CTRL_REG); - IWL_CMD(FH_MEM_RSSR_RX_STATUS_REG); - IWL_CMD(FH_MEM_RSSR_RX_ENABLE_ERR_IRQ2DRV); - IWL_CMD(FH_TSSR_TX_STATUS_REG); - IWL_CMD(FH_TSSR_TX_ERROR_REG); - default: - return "UNKNOWN"; - } -#undef IWL_CMD -} - -int iwl_dump_fh(struct iwl_trans *trans, char **buf) -{ - int i; - static const u32 fh_tbl[] = { - FH_RSCSR_CHNL0_STTS_WPTR_REG, - FH_RSCSR_CHNL0_RBDCB_BASE_REG, - FH_RSCSR_CHNL0_WPTR, - FH_MEM_RCSR_CHNL0_CONFIG_REG, - FH_MEM_RSSR_SHARED_CTRL_REG, - FH_MEM_RSSR_RX_STATUS_REG, - FH_MEM_RSSR_RX_ENABLE_ERR_IRQ2DRV, - FH_TSSR_TX_STATUS_REG, - FH_TSSR_TX_ERROR_REG - }; - - if (trans->mac_cfg->mq_rx_supported) - return iwl_dump_rfh(trans, buf); - -#ifdef CONFIG_IWLWIFI_DEBUGFS - if (buf) { - int pos = 0; - size_t bufsz = ARRAY_SIZE(fh_tbl) * 48 + 40; - - *buf = kmalloc(bufsz, GFP_KERNEL); - if (!*buf) - return -ENOMEM; - - pos += scnprintf(*buf + pos, bufsz - pos, - "FH register values:\n"); - - for (i = 0; i < ARRAY_SIZE(fh_tbl); i++) - pos += scnprintf(*buf + pos, bufsz - pos, - " %34s: 0X%08x\n", - get_fh_string(fh_tbl[i]), - iwl_read_direct32(trans, fh_tbl[i])); - - return pos; - } -#endif - - IWL_ERR(trans, "FH register values:\n"); - for (i = 0; i < ARRAY_SIZE(fh_tbl); i++) - IWL_ERR(trans, " %34s: 0X%08x\n", - get_fh_string(fh_tbl[i]), - iwl_read_direct32(trans, fh_tbl[i])); - - return 0; -} - void iwl_trans_sync_nmi_with_addr(struct iwl_trans *trans, u32 inta_addr, u32 sw_err_bit) { diff --git a/drivers/net/wireless/intel/iwlwifi/iwl-io.h b/drivers/net/wireless/intel/iwlwifi/iwl-io.h index 6dce2e5267a645..e66b0c2a046764 100644 --- a/drivers/net/wireless/intel/iwlwifi/iwl-io.h +++ b/drivers/net/wireless/intel/iwlwifi/iwl-io.h @@ -59,9 +59,6 @@ void iwl_set_bits_mask_prph(struct iwl_trans *trans, u32 ofs, void iwl_clear_bits_prph(struct iwl_trans *trans, u32 ofs, u32 mask); void iwl_force_nmi(struct iwl_trans *trans); -/* Error handling */ -int iwl_dump_fh(struct iwl_trans *trans, char **buf); - /* * UMAC periphery address space changed from 0xA00000 to 0xD00000 starting from * device family AX200. So peripheries used in families above and below AX200 diff --git a/drivers/net/wireless/intel/iwlwifi/pcie/trans.c b/drivers/net/wireless/intel/iwlwifi/pcie/trans.c index 4771cf3b3e11d8..97bbc3faa98c8e 100644 --- a/drivers/net/wireless/intel/iwlwifi/pcie/trans.c +++ b/drivers/net/wireless/intel/iwlwifi/pcie/trans.c @@ -3016,24 +3016,6 @@ static ssize_t iwl_dbgfs_csr_write(struct file *file, return count; } -static ssize_t iwl_dbgfs_fh_reg_read(struct file *file, - char __user *user_buf, - size_t count, loff_t *ppos) -{ - struct iwl_trans *trans = file->private_data; - char *buf = NULL; - ssize_t ret; - - ret = iwl_dump_fh(trans, &buf); - if (ret < 0) - return ret; - if (!buf) - return -EINVAL; - ret = simple_read_from_buffer(user_buf, count, ppos, buf, ret); - kfree(buf); - return ret; -} - static ssize_t iwl_dbgfs_rfkill_read(struct file *file, char __user *user_buf, size_t count, loff_t *ppos) @@ -3133,7 +3115,6 @@ static ssize_t iwl_dbgfs_reset_write(struct file *file, } DEBUGFS_READ_WRITE_FILE_OPS(interrupt); -DEBUGFS_READ_FILE_OPS(fh_reg); DEBUGFS_READ_FILE_OPS(rx_queue); DEBUGFS_WRITE_FILE_OPS(csr); DEBUGFS_READ_WRITE_FILE_OPS(rfkill); @@ -3157,7 +3138,6 @@ void iwl_trans_pcie_dbgfs_register(struct iwl_trans *trans) DEBUGFS_ADD_FILE(tx_queue, dir, 0400); DEBUGFS_ADD_FILE(interrupt, dir, 0600); DEBUGFS_ADD_FILE(csr, dir, 0200); - DEBUGFS_ADD_FILE(fh_reg, dir, 0400); DEBUGFS_ADD_FILE(rfkill, dir, 0600); DEBUGFS_ADD_FILE(rf, dir, 0400); DEBUGFS_ADD_FILE(reset, dir, 0200); From 028bc654dea36fd42015dc7f5fb0913d5f450589 Mon Sep 17 00:00:00 2001 From: Miri Korenblit Date: Sat, 26 Sep 2026 20:17:29 +0300 Subject: [PATCH 0996/1012] wifi: iwlwifi: open code iwl_trans_sync_nmi_with_addr() iwl_trans_sync_nmi_with_addr() had a single caller, iwl_trans_pcie_sync_nmi(). There is no reason to keep it as a separate exported function in iwl-io.c, so embed its body directly in the caller and drop the now-unused declaration. Link: https://patch.msgid.link/20260926201527.a2048732f45f.I23611eb40d875c6a69df0c08add3834f6f7d8528@changeid Signed-off-by: Miri Korenblit --- drivers/net/wireless/intel/iwlwifi/iwl-io.c | 36 ------------------- .../net/wireless/intel/iwlwifi/iwl-trans.c | 5 --- .../net/wireless/intel/iwlwifi/iwl-trans.h | 5 --- .../net/wireless/intel/iwlwifi/pcie/trans.c | 33 +++++++++++++++-- 4 files changed, 31 insertions(+), 48 deletions(-) diff --git a/drivers/net/wireless/intel/iwlwifi/iwl-io.c b/drivers/net/wireless/intel/iwlwifi/iwl-io.c index 544b8e686dfe8f..ef4d97ef0b1b24 100644 --- a/drivers/net/wireless/intel/iwlwifi/iwl-io.c +++ b/drivers/net/wireless/intel/iwlwifi/iwl-io.c @@ -233,39 +233,3 @@ void iwl_force_nmi(struct iwl_trans *trans) UREG_DOORBELL_TO_ISR6_NMI_BIT); } IWL_EXPORT_SYMBOL(iwl_force_nmi); - -void iwl_trans_sync_nmi_with_addr(struct iwl_trans *trans, u32 inta_addr, - u32 sw_err_bit) -{ - unsigned long timeout = jiffies + IWL_TRANS_NMI_TIMEOUT; - bool interrupts_enabled = test_bit(STATUS_INT_ENABLED, &trans->status); - - /* if the interrupts were already disabled, there is no point in - * calling iwl_disable_interrupts - */ - if (interrupts_enabled) - iwl_trans_interrupts(trans, false); - - iwl_force_nmi(trans); - while (time_after(timeout, jiffies)) { - u32 inta_hw = iwl_read32(trans, inta_addr); - - /* Error detected by uCode */ - if (inta_hw & sw_err_bit) { - /* Clear causes register */ - iwl_write32(trans, inta_addr, inta_hw & sw_err_bit); - break; - } - - mdelay(1); - } - - /* enable interrupts only if there were already enabled before this - * function to avoid a case were the driver enable interrupts before - * proper configurations were made - */ - if (interrupts_enabled) - iwl_trans_interrupts(trans, true); - - iwl_trans_fw_error(trans, IWL_ERR_TYPE_NMI_FORCED); -} diff --git a/drivers/net/wireless/intel/iwlwifi/iwl-trans.c b/drivers/net/wireless/intel/iwlwifi/iwl-trans.c index 24bf45f06746d4..ad43cf53e8c5c2 100644 --- a/drivers/net/wireless/intel/iwlwifi/iwl-trans.c +++ b/drivers/net/wireless/intel/iwlwifi/iwl-trans.c @@ -523,11 +523,6 @@ int iwl_trans_d3_resume(struct iwl_trans *trans, bool reset) } IWL_EXPORT_SYMBOL(iwl_trans_d3_resume); -void iwl_trans_interrupts(struct iwl_trans *trans, bool enable) -{ - iwl_trans_pci_interrupts(trans, enable); -} - void iwl_trans_sync_nmi(struct iwl_trans *trans) { iwl_trans_pcie_sync_nmi(trans); diff --git a/drivers/net/wireless/intel/iwlwifi/iwl-trans.h b/drivers/net/wireless/intel/iwlwifi/iwl-trans.h index dfad839dd2e698..f933cba7eee8ab 100644 --- a/drivers/net/wireless/intel/iwlwifi/iwl-trans.h +++ b/drivers/net/wireless/intel/iwlwifi/iwl-trans.h @@ -1055,9 +1055,6 @@ static inline bool iwl_trans_fw_running(struct iwl_trans *trans) void iwl_trans_sync_nmi(struct iwl_trans *trans); -void iwl_trans_sync_nmi_with_addr(struct iwl_trans *trans, u32 inta_addr, - u32 sw_err_bit); - int iwl_trans_load_pnvm(struct iwl_trans *trans, const struct iwl_pnvm_image *pnvm_data, const struct iwl_ucode_capabilities *capa); @@ -1078,8 +1075,6 @@ static inline bool iwl_trans_dbg_ini_valid(struct iwl_trans *trans) trans->dbg.external_ini_cfg != IWL_INI_CFG_STATE_NOT_LOADED; } -void iwl_trans_interrupts(struct iwl_trans *trans, bool enable); - int iwl_trans_activate_nic(struct iwl_trans *trans); static inline void iwl_trans_finish_sw_reset(struct iwl_trans *trans) diff --git a/drivers/net/wireless/intel/iwlwifi/pcie/trans.c b/drivers/net/wireless/intel/iwlwifi/pcie/trans.c index 97bbc3faa98c8e..48ae2440cfe1d3 100644 --- a/drivers/net/wireless/intel/iwlwifi/pcie/trans.c +++ b/drivers/net/wireless/intel/iwlwifi/pcie/trans.c @@ -3574,8 +3574,10 @@ void iwl_trans_pci_interrupts(struct iwl_trans *trans, bool enable) void iwl_trans_pcie_sync_nmi(struct iwl_trans *trans) { - u32 inta_addr, sw_err_bit; + bool interrupts_enabled = test_bit(STATUS_INT_ENABLED, &trans->status); struct iwl_trans_pcie *trans_pcie = IWL_TRANS_GET_PCIE_TRANS(trans); + unsigned long timeout = jiffies + IWL_TRANS_NMI_TIMEOUT; + u32 inta_addr, sw_err_bit; if (trans_pcie->msix_enabled) { inta_addr = CSR_MSIX_HW_INT_CAUSES_AD; @@ -3588,7 +3590,34 @@ void iwl_trans_pcie_sync_nmi(struct iwl_trans *trans) sw_err_bit = CSR_INT_BIT_SW_ERR; } - iwl_trans_sync_nmi_with_addr(trans, inta_addr, sw_err_bit); + /* if the interrupts were already disabled, there is no point in + * calling iwl_disable_interrupts + */ + if (interrupts_enabled) + iwl_trans_pci_interrupts(trans, false); + + iwl_force_nmi(trans); + while (time_after(timeout, jiffies)) { + u32 inta_hw = iwl_read32(trans, inta_addr); + + /* Error detected by uCode */ + if (inta_hw & sw_err_bit) { + /* Clear causes register */ + iwl_write32(trans, inta_addr, inta_hw & sw_err_bit); + break; + } + + mdelay(1); + } + + /* enable interrupts only if there were already enabled before this + * function to avoid a case were the driver enable interrupts before + * proper configurations were made + */ + if (interrupts_enabled) + iwl_trans_pci_interrupts(trans, true); + + iwl_trans_fw_error(trans, IWL_ERR_TYPE_NMI_FORCED); } static int iwl_trans_pcie_alloc_txcmd_pool(struct iwl_trans *trans) From 400ff2d7582a78a38189a803f758125925021d00 Mon Sep 17 00:00:00 2001 From: Miri Korenblit Date: Sat, 26 Sep 2026 20:17:30 +0300 Subject: [PATCH 0997/1012] wifi: iwlwifi: move iwl_force_nmi() to where it belongs iwl_force_nmi() forces a firmware NMI and has nothing to do with the register I/O helpers in iwl-io.c. Move it to iwl-trans.c and rename it to iwl_trans_force_nmi() to match the transport API naming. Reviewed-by: Emmanuel Grumbach Link: https://patch.msgid.link/20260926201527.c3312fb2f2cc.Ie0d3ea9b23e35d7ff4b67e129c7b97ccb7c6c453@changeid Signed-off-by: Miri Korenblit --- drivers/net/wireless/intel/iwlwifi/fw/dbg.c | 5 +++-- drivers/net/wireless/intel/iwlwifi/iwl-io.c | 18 ------------------ drivers/net/wireless/intel/iwlwifi/iwl-io.h | 1 - drivers/net/wireless/intel/iwlwifi/iwl-trans.c | 18 ++++++++++++++++++ drivers/net/wireless/intel/iwlwifi/iwl-trans.h | 2 ++ .../net/wireless/intel/iwlwifi/mld/debugfs.c | 2 +- .../net/wireless/intel/iwlwifi/mvm/debugfs.c | 2 +- .../net/wireless/intel/iwlwifi/mvm/mac80211.c | 4 ++-- drivers/net/wireless/intel/iwlwifi/mvm/scan.c | 2 +- drivers/net/wireless/intel/iwlwifi/pcie/rx.c | 2 +- .../net/wireless/intel/iwlwifi/pcie/trans.c | 2 +- drivers/net/wireless/intel/iwlwifi/pcie/tx.c | 4 ++-- 12 files changed, 32 insertions(+), 30 deletions(-) diff --git a/drivers/net/wireless/intel/iwlwifi/fw/dbg.c b/drivers/net/wireless/intel/iwlwifi/fw/dbg.c index 8c8d15ac71ac03..897538066b7dea 100644 --- a/drivers/net/wireless/intel/iwlwifi/fw/dbg.c +++ b/drivers/net/wireless/intel/iwlwifi/fw/dbg.c @@ -2055,7 +2055,7 @@ int iwl_fw_dbg_collect(struct iwl_fw_runtime *fwrt, if (trigger->flags & IWL_FW_DBG_FORCE_RESTART) { IWL_WARN(fwrt, "Force restart: trigger %d fired.\n", trig); - iwl_force_nmi(fwrt->trans); + iwl_trans_force_nmi(fwrt->trans); return 0; } @@ -2233,7 +2233,8 @@ static void iwl_fw_dbg_collect_sync(struct iwl_fw_runtime *fwrt, u8 wk_idx) } if (fwrt->trans->dbg.last_tp_resetfw == IWL_FW_INI_RESET_FW_MODE_STOP_FW_ONLY) - iwl_force_nmi(fwrt->trans); + iwl_trans_force_nmi(fwrt->trans); + out: if (iwl_trans_dbg_ini_valid(fwrt->trans)) { iwl_fw_error_dump_data_free(dump_data); diff --git a/drivers/net/wireless/intel/iwlwifi/iwl-io.c b/drivers/net/wireless/intel/iwlwifi/iwl-io.c index ef4d97ef0b1b24..cc6bf4ea5055d5 100644 --- a/drivers/net/wireless/intel/iwlwifi/iwl-io.c +++ b/drivers/net/wireless/intel/iwlwifi/iwl-io.c @@ -10,7 +10,6 @@ #include "iwl-io.h" #include "iwl-csr.h" #include "iwl-debug.h" -#include "iwl-prph.h" void iwl_write8(struct iwl_trans *trans, u32 ofs, u8 val) { @@ -216,20 +215,3 @@ void iwl_clear_bits_prph(struct iwl_trans *trans, u32 ofs, u32 mask) } } IWL_EXPORT_SYMBOL(iwl_clear_bits_prph); - -void iwl_force_nmi(struct iwl_trans *trans) -{ - if (trans->mac_cfg->device_family < IWL_DEVICE_FAMILY_9000) - iwl_write_prph_delay(trans, DEVICE_SET_NMI_REG, - DEVICE_SET_NMI_VAL_DRV, 1); - else if (trans->mac_cfg->device_family < IWL_DEVICE_FAMILY_AX210) - iwl_write_umac_prph(trans, UREG_NIC_SET_NMI_DRIVER, - UREG_NIC_SET_NMI_DRIVER_NMI_FROM_DRIVER); - else if (trans->mac_cfg->device_family < IWL_DEVICE_FAMILY_BZ) - iwl_write_umac_prph(trans, UREG_DOORBELL_TO_ISR6, - UREG_DOORBELL_TO_ISR6_NMI_BIT); - else - iwl_write32(trans, CSR_DOORBELL_VECTOR, - UREG_DOORBELL_TO_ISR6_NMI_BIT); -} -IWL_EXPORT_SYMBOL(iwl_force_nmi); diff --git a/drivers/net/wireless/intel/iwlwifi/iwl-io.h b/drivers/net/wireless/intel/iwlwifi/iwl-io.h index e66b0c2a046764..e4a9f2ba0de93d 100644 --- a/drivers/net/wireless/intel/iwlwifi/iwl-io.h +++ b/drivers/net/wireless/intel/iwlwifi/iwl-io.h @@ -57,7 +57,6 @@ void iwl_set_bits_prph(struct iwl_trans *trans, u32 ofs, u32 mask); void iwl_set_bits_mask_prph(struct iwl_trans *trans, u32 ofs, u32 bits, u32 mask); void iwl_clear_bits_prph(struct iwl_trans *trans, u32 ofs, u32 mask); -void iwl_force_nmi(struct iwl_trans *trans); /* * UMAC periphery address space changed from 0xA00000 to 0xD00000 starting from diff --git a/drivers/net/wireless/intel/iwlwifi/iwl-trans.c b/drivers/net/wireless/intel/iwlwifi/iwl-trans.c index ad43cf53e8c5c2..efe657857772bb 100644 --- a/drivers/net/wireless/intel/iwlwifi/iwl-trans.c +++ b/drivers/net/wireless/intel/iwlwifi/iwl-trans.c @@ -10,6 +10,7 @@ #include "iwl-trans.h" #include "iwl-drv.h" +#include "iwl-prph.h" #include #include "fw/api/commands.h" #include "pcie/internal.h" @@ -452,6 +453,23 @@ void iwl_trans_write_prph(struct iwl_trans *trans, u32 ofs, u32 val) return iwl_trans_pcie_write_prph(trans, ofs, val); } +void iwl_trans_force_nmi(struct iwl_trans *trans) +{ + if (trans->mac_cfg->device_family < IWL_DEVICE_FAMILY_9000) + iwl_write_prph_delay(trans, DEVICE_SET_NMI_REG, + DEVICE_SET_NMI_VAL_DRV, 1); + else if (trans->mac_cfg->device_family < IWL_DEVICE_FAMILY_AX210) + iwl_write_umac_prph(trans, UREG_NIC_SET_NMI_DRIVER, + UREG_NIC_SET_NMI_DRIVER_NMI_FROM_DRIVER); + else if (trans->mac_cfg->device_family < IWL_DEVICE_FAMILY_BZ) + iwl_write_umac_prph(trans, UREG_DOORBELL_TO_ISR6, + UREG_DOORBELL_TO_ISR6_NMI_BIT); + else + iwl_write32(trans, CSR_DOORBELL_VECTOR, + UREG_DOORBELL_TO_ISR6_NMI_BIT); +} +IWL_EXPORT_SYMBOL(iwl_trans_force_nmi); + int iwl_trans_read_mem(struct iwl_trans *trans, u32 addr, void *buf, int dwords) { diff --git a/drivers/net/wireless/intel/iwlwifi/iwl-trans.h b/drivers/net/wireless/intel/iwlwifi/iwl-trans.h index f933cba7eee8ab..a7bb6cee81880f 100644 --- a/drivers/net/wireless/intel/iwlwifi/iwl-trans.h +++ b/drivers/net/wireless/intel/iwlwifi/iwl-trans.h @@ -1055,6 +1055,8 @@ static inline bool iwl_trans_fw_running(struct iwl_trans *trans) void iwl_trans_sync_nmi(struct iwl_trans *trans); +void iwl_trans_force_nmi(struct iwl_trans *trans); + int iwl_trans_load_pnvm(struct iwl_trans *trans, const struct iwl_pnvm_image *pnvm_data, const struct iwl_ucode_capabilities *capa); diff --git a/drivers/net/wireless/intel/iwlwifi/mld/debugfs.c b/drivers/net/wireless/intel/iwlwifi/mld/debugfs.c index 351a4f177e920f..dd4bf7cb6a7911 100644 --- a/drivers/net/wireless/intel/iwlwifi/mld/debugfs.c +++ b/drivers/net/wireless/intel/iwlwifi/mld/debugfs.c @@ -69,7 +69,7 @@ static ssize_t iwl_dbgfs_fw_nmi_write(struct iwl_mld *mld, char *buf, if (count == 6 && !strcmp(buf, "nolog\n")) mld->fw_status.do_not_dump_once = true; - iwl_force_nmi(mld->trans); + iwl_trans_force_nmi(mld->trans); return count; } diff --git a/drivers/net/wireless/intel/iwlwifi/mvm/debugfs.c b/drivers/net/wireless/intel/iwlwifi/mvm/debugfs.c index 4db0f29dd21af8..06502f57e823e3 100644 --- a/drivers/net/wireless/intel/iwlwifi/mvm/debugfs.c +++ b/drivers/net/wireless/intel/iwlwifi/mvm/debugfs.c @@ -1102,7 +1102,7 @@ static ssize_t iwl_dbgfs_fw_nmi_write(struct iwl_mvm *mvm, char *buf, if (count == 6 && !strcmp(buf, "nolog\n")) set_bit(IWL_MVM_STATUS_SUPPRESS_ERROR_LOG_ONCE, &mvm->status); - iwl_force_nmi(mvm->trans); + iwl_trans_force_nmi(mvm->trans); return count; } diff --git a/drivers/net/wireless/intel/iwlwifi/mvm/mac80211.c b/drivers/net/wireless/intel/iwlwifi/mvm/mac80211.c index 5bd246e3794327..a4f3d3c2bd350a 100644 --- a/drivers/net/wireless/intel/iwlwifi/mvm/mac80211.c +++ b/drivers/net/wireless/intel/iwlwifi/mvm/mac80211.c @@ -5258,7 +5258,7 @@ iwl_mvm_switch_vif_chanctx_swap(struct iwl_mvm *mvm, out_restart: /* things keep failing, better restart the hw */ - iwl_force_nmi(mvm->trans); + iwl_trans_force_nmi(mvm->trans); return ret; } @@ -5294,7 +5294,7 @@ iwl_mvm_switch_vif_chanctx_reassign(struct iwl_mvm *mvm, out_restart: /* things keep failing, better restart the hw */ - iwl_force_nmi(mvm->trans); + iwl_trans_force_nmi(mvm->trans); return ret; } diff --git a/drivers/net/wireless/intel/iwlwifi/mvm/scan.c b/drivers/net/wireless/intel/iwlwifi/mvm/scan.c index 817c6dacfaccd7..a94eaca0b3ed51 100644 --- a/drivers/net/wireless/intel/iwlwifi/mvm/scan.c +++ b/drivers/net/wireless/intel/iwlwifi/mvm/scan.c @@ -2703,7 +2703,7 @@ void iwl_mvm_scan_timeout_wk(struct work_struct *work) IWL_ERR(mvm, "regular scan timed out\n"); - iwl_force_nmi(mvm->trans); + iwl_trans_force_nmi(mvm->trans); } static void iwl_mvm_fill_scan_type(struct iwl_mvm *mvm, diff --git a/drivers/net/wireless/intel/iwlwifi/pcie/rx.c b/drivers/net/wireless/intel/iwlwifi/pcie/rx.c index d1ef0365b932cf..db3933b0d1979e 100644 --- a/drivers/net/wireless/intel/iwlwifi/pcie/rx.c +++ b/drivers/net/wireless/intel/iwlwifi/pcie/rx.c @@ -1493,7 +1493,7 @@ static struct iwl_rx_mem_buffer *iwl_pcie_get_rxb(struct iwl_trans *trans, out_err: WARN(1, "Invalid rxb from HW %u\n", (u32)vid); - iwl_force_nmi(trans); + iwl_trans_force_nmi(trans); return NULL; } diff --git a/drivers/net/wireless/intel/iwlwifi/pcie/trans.c b/drivers/net/wireless/intel/iwlwifi/pcie/trans.c index 48ae2440cfe1d3..1304d89164240a 100644 --- a/drivers/net/wireless/intel/iwlwifi/pcie/trans.c +++ b/drivers/net/wireless/intel/iwlwifi/pcie/trans.c @@ -3596,7 +3596,7 @@ void iwl_trans_pcie_sync_nmi(struct iwl_trans *trans) if (interrupts_enabled) iwl_trans_pci_interrupts(trans, false); - iwl_force_nmi(trans); + iwl_trans_force_nmi(trans); while (time_after(timeout, jiffies)) { u32 inta_hw = iwl_read32(trans, inta_addr); diff --git a/drivers/net/wireless/intel/iwlwifi/pcie/tx.c b/drivers/net/wireless/intel/iwlwifi/pcie/tx.c index 26d1530cfa373e..496435ea93ee83 100644 --- a/drivers/net/wireless/intel/iwlwifi/pcie/tx.c +++ b/drivers/net/wireless/intel/iwlwifi/pcie/tx.c @@ -715,7 +715,7 @@ static void iwl_txq_stuck_timer(struct timer_list *t) iwl_txq_log_scd_error(trans, txq); - iwl_force_nmi(trans); + iwl_trans_force_nmi(trans); } int iwl_pcie_txq_alloc(struct iwl_trans *trans, struct iwl_txq *txq, @@ -1105,7 +1105,7 @@ static void iwl_pcie_cmdq_reclaim(struct iwl_trans *trans, int txq_id, int idx) if (nfreed++ > 0) { IWL_ERR(trans, "HCMD skipped: index (%d) %d %d\n", idx, txq->write_ptr, r); - iwl_force_nmi(trans); + iwl_trans_force_nmi(trans); } } From f135428295893e9012c4f535cd336d6be531ec73 Mon Sep 17 00:00:00 2001 From: Emmanuel Grumbach Date: Sat, 26 Sep 2026 20:17:31 +0300 Subject: [PATCH 0998/1012] wifi: iwlwifi: add support for the new MPDU descriptor The firmware now reports in DW-12 bits 31:24 the RU/MRU/DRU that was used for the received frame, in trigger-like format. The MAC context byte moved to bits 31:24 of DW-10. It isn't used by the driver, so just follow the new layout. Assisted-by: LLM Signed-off-by: Emmanuel Grumbach Link: https://patch.msgid.link/20260926201527.cf236c23d1aa.Ie2d919c884f7db8cd76b4cb6a9f4584eb285eea5@changeid Signed-off-by: Miri Korenblit --- .../net/wireless/intel/iwlwifi/fw/api/rx.h | 19 +++++++++++++++---- 1 file changed, 15 insertions(+), 4 deletions(-) diff --git a/drivers/net/wireless/intel/iwlwifi/fw/api/rx.h b/drivers/net/wireless/intel/iwlwifi/fw/api/rx.h index a8f1edf19bbefc..20460534fc99d0 100644 --- a/drivers/net/wireless/intel/iwlwifi/fw/api/rx.h +++ b/drivers/net/wireless/intel/iwlwifi/fw/api/rx.h @@ -583,7 +583,11 @@ struct iwl_rx_mpdu_desc_v3 { /** * @reserved_xsum: reserved high bits in the raw checksum */ - __le16 reserved_xsum; + u8 reserved_xsum; + /** + * @mac_context: MAC context mask, in this DW only from API version 10 + */ + u8 mac_context; /* DW11 */ /** * @rate_n_flags: RX rate/flags encoding @@ -603,9 +607,10 @@ struct iwl_rx_mpdu_desc_v3 { */ u8 channel; /** - * @mac_context: MAC context mask + * @ru: RU/MRU/DRU used for the frame, in trigger-like format; + * only from API version 10, this byte held @mac_context before */ - u8 mac_context; + u8 ru; /* DW13 */ /** * @gp2_on_air_rise: GP2 timer value on air rise (INA) @@ -647,7 +652,11 @@ struct iwl_rx_mpdu_desc_v3 { __le32 reserved[1]; } __packed; /* RX_MPDU_RES_START_API_S_VER_3, * RX_MPDU_RES_START_API_S_VER_5, - * RX_MPDU_RES_START_API_S_VER_6 + * RX_MPDU_RES_START_API_S_VER_6, + * RX_MPDU_RES_START_API_S_VER_7, + * RX_MPDU_RES_START_API_S_VER_8, + * RX_MPDU_RES_START_API_S_VER_9, + * RX_MPDU_RES_START_API_S_VER_10 */ /** @@ -737,6 +746,8 @@ struct iwl_rx_mpdu_desc { * RX_MPDU_RES_START_API_S_VER_6 * RX_MPDU_RES_START_API_S_VER_7 * RX_MPDU_RES_START_API_S_VER_8 + * RX_MPDU_RES_START_API_S_VER_9 + * RX_MPDU_RES_START_API_S_VER_10 */ #define IWL_RX_DESC_SIZE_V1 offsetofend(struct iwl_rx_mpdu_desc, v1) From 8a7c9bfcaa61b8737fc66e84d7df4712fbdbd938 Mon Sep 17 00:00:00 2001 From: Emmanuel Grumbach Date: Sat, 26 Sep 2026 20:17:32 +0300 Subject: [PATCH 0999/1012] wifi: iwlwifi: support RX BAID modify command version 3 Version 3 of the RX BAID allocation modify command shrinks the tid field to a u8 and adds a window size field, so that the window size can be updated while moving a BAID to a new AP. A window size of 0 tells the firmware to keep the current one, which is what we need for now. Add the new layout and pick it based on the command version. Assisted-by: LLM Signed-off-by: Emmanuel Grumbach Link: https://patch.msgid.link/20260926201527.46c77de75c0b.Ie41b1d43f945a0c5e2bb68981a0cec0c62111233@changeid Signed-off-by: Miri Korenblit --- .../wireless/intel/iwlwifi/fw/api/datapath.h | 29 +++++++++++++++---- drivers/net/wireless/intel/iwlwifi/mld/agg.c | 7 ++++- 2 files changed, 30 insertions(+), 6 deletions(-) diff --git a/drivers/net/wireless/intel/iwlwifi/fw/api/datapath.h b/drivers/net/wireless/intel/iwlwifi/fw/api/datapath.h index b17babef71c018..3fbcb2e4bfe069 100644 --- a/drivers/net/wireless/intel/iwlwifi/fw/api/datapath.h +++ b/drivers/net/wireless/intel/iwlwifi/fw/api/datapath.h @@ -522,17 +522,34 @@ struct iwl_rx_baid_cfg_cmd_alloc { } __packed; /* RX_BAID_ALLOCATION_ADD_CMD_API_S_VER_1 */ /** - * struct iwl_rx_baid_cfg_cmd_modify - BAID modification data + * struct iwl_rx_baid_cfg_cmd_modify_v2 - BAID modification data * @old_sta_id_mask: old station ID mask * @new_sta_id_mask: new station ID mask * @tid: TID of the BAID */ -struct iwl_rx_baid_cfg_cmd_modify { +struct iwl_rx_baid_cfg_cmd_modify_v2 { __le32 old_sta_id_mask; __le32 new_sta_id_mask; __le32 tid; } __packed; /* RX_BAID_ALLOCATION_MODIFY_CMD_API_S_VER_2 */ +/** + * struct iwl_rx_baid_cfg_cmd_modify - BAID modification data + * @old_sta_id_mask: old station ID mask + * @new_sta_id_mask: new station ID mask + * @tid: TID of the BAID + * @reserved: reserved + * @win_size: RX BA session window size to apply to the modified BAID, + * 0 means keep the current one + */ +struct iwl_rx_baid_cfg_cmd_modify { + __le32 old_sta_id_mask; + __le32 new_sta_id_mask; + u8 tid; + u8 reserved; + __le16 win_size; +} __packed; /* RX_BAID_ALLOCATION_MODIFY_CMD_API_S_VER_3 */ + /** * struct iwl_rx_baid_cfg_cmd_remove_v1 - BAID removal data * @baid: the BAID to remove @@ -555,7 +572,8 @@ struct iwl_rx_baid_cfg_cmd_remove { * struct iwl_rx_baid_cfg_cmd - BAID allocation/config command * @action: the action, from &enum iwl_rx_baid_action * @alloc: allocation data - * @modify: modify data + * @modify_v2: modify data (version 2) + * @modify: modify data (version 3) * @remove_v1: remove data (version 1) * @remove: remove data */ @@ -563,11 +581,12 @@ struct iwl_rx_baid_cfg_cmd { __le32 action; union { struct iwl_rx_baid_cfg_cmd_alloc alloc; + struct iwl_rx_baid_cfg_cmd_modify_v2 modify_v2; struct iwl_rx_baid_cfg_cmd_modify modify; struct iwl_rx_baid_cfg_cmd_remove_v1 remove_v1; struct iwl_rx_baid_cfg_cmd_remove remove; - }; /* RX_BAID_ALLOCATION_OPERATION_API_U_VER_2 */ -} __packed; /* RX_BAID_ALLOCATION_CONFIG_CMD_API_S_VER_2 */ + }; /* RX_BAID_ALLOCATION_OPERATION_API_U_VER_3 */ +} __packed; /* RX_BAID_ALLOCATION_CONFIG_CMD_API_S_VER_3 */ /** * struct iwl_rx_baid_cfg_resp - BAID allocation response diff --git a/drivers/net/wireless/intel/iwlwifi/mld/agg.c b/drivers/net/wireless/intel/iwlwifi/mld/agg.c index c45c47337509e9..ea1f02380ad4ef 100644 --- a/drivers/net/wireless/intel/iwlwifi/mld/agg.c +++ b/drivers/net/wireless/intel/iwlwifi/mld/agg.c @@ -660,6 +660,7 @@ int iwl_mld_update_sta_baids(struct iwl_mld *mld, .modify.new_sta_id_mask = cpu_to_le32(new_sta_mask), }; u32 cmd_id = WIDE_ID(DATA_PATH_GROUP, RX_BAID_ALLOCATION_CONFIG_CMD); + u8 cmd_ver = iwl_fw_lookup_cmd_ver(mld->fw, cmd_id, 2); int baid; /* mac80211 will remove sessions later, but we ignore all that */ @@ -667,6 +668,7 @@ int iwl_mld_update_sta_baids(struct iwl_mld *mld, return 0; BUILD_BUG_ON(sizeof(struct iwl_rx_baid_cfg_resp) != sizeof(baid)); + BUILD_BUG_ON(sizeof(cmd.modify) != sizeof(cmd.modify_v2)); for (baid = 0; baid < ARRAY_SIZE(mld->fw_id_to_ba); baid++) { struct iwl_mld_baid_data *data; @@ -683,7 +685,10 @@ int iwl_mld_update_sta_baids(struct iwl_mld *mld, "BAID data for %d corrupted - expected 0x%x found 0x%x\n", baid, old_sta_mask, data->sta_mask); - cmd.modify.tid = cpu_to_le32(data->tid); + if (cmd_ver >= 3) + cmd.modify.tid = data->tid; + else + cmd.modify_v2.tid = cpu_to_le32(data->tid); ret = iwl_mld_send_cmd_pdu(mld, cmd_id, &cmd); if (ret) From 18925f7a61206e383bdf852faf4f306a9c717e09 Mon Sep 17 00:00:00 2001 From: Miri Korenblit Date: Sat, 26 Sep 2026 20:17:33 +0300 Subject: [PATCH 1000/1012] wifi: iwlwifi: drop the orphaned nic_access annotations commit 5b63d0ae94cc ("compiler-context-analysis: Remove Sparse support") deleted the Sparse implementation of the lock annotations. __acquires()/__releases()/__acquire()/__release() are gone; the macros now only map onto clang's context analysis and expand to nothing for gcc. The nic_access annotations describe nothing anymore: - Every one of them is a __releases() with no matching __acquires(). The acquire side was the __cond_lock() wrapper around _iwl_trans_grab_nic_access(), which went away together with __cond_lock() itself. The names don't even agree - PCIe releases nic_access_nobh while the generic wrapper calling it claims nic_access. - The __acquire()/__release(reg_lock) pairs only existed to silence Sparse Use lockdep_assert_held() instead, which unlike the annotations actually runs, and add the missing assertion to iwl_trans_pcie_resched_with_nic_access(). Reviewed-by: Johannes Berg Link: https://patch.msgid.link/20260926201527.f818a7e09def.Iccecaabe0f37525430377ae1c0db6fd7bbf7625a@changeid Signed-off-by: Miri Korenblit --- drivers/net/wireless/intel/iwlwifi/iwl-trans.c | 3 +-- drivers/net/wireless/intel/iwlwifi/iwl-trans.h | 3 +-- .../net/wireless/intel/iwlwifi/pcie/internal.h | 3 +-- .../net/wireless/intel/iwlwifi/pcie/trans.c | 18 ++++++------------ 4 files changed, 9 insertions(+), 18 deletions(-) diff --git a/drivers/net/wireless/intel/iwlwifi/iwl-trans.c b/drivers/net/wireless/intel/iwlwifi/iwl-trans.c index efe657857772bb..0a6b7b264aa230 100644 --- a/drivers/net/wireless/intel/iwlwifi/iwl-trans.c +++ b/drivers/net/wireless/intel/iwlwifi/iwl-trans.c @@ -576,8 +576,7 @@ void iwl_trans_resched_with_nic_access(struct iwl_trans *trans) iwl_trans_pcie_resched_with_nic_access(trans); } -void __releases(nic_access) -iwl_trans_release_nic_access(struct iwl_trans *trans) +void iwl_trans_release_nic_access(struct iwl_trans *trans) { iwl_trans_pcie_release_nic_access(trans); } diff --git a/drivers/net/wireless/intel/iwlwifi/iwl-trans.h b/drivers/net/wireless/intel/iwlwifi/iwl-trans.h index a7bb6cee81880f..56c5e61b614ad0 100644 --- a/drivers/net/wireless/intel/iwlwifi/iwl-trans.h +++ b/drivers/net/wireless/intel/iwlwifi/iwl-trans.h @@ -992,8 +992,7 @@ bool iwl_trans_grab_nic_access(struct iwl_trans *trans); */ void iwl_trans_resched_with_nic_access(struct iwl_trans *trans); -void __releases(nic_access) -iwl_trans_release_nic_access(struct iwl_trans *trans); +void iwl_trans_release_nic_access(struct iwl_trans *trans); static inline void iwl_trans_schedule_reset(struct iwl_trans *trans, enum iwl_fw_error_type type) diff --git a/drivers/net/wireless/intel/iwlwifi/pcie/internal.h b/drivers/net/wireless/intel/iwlwifi/pcie/internal.h index e10aa8e7e06190..e6b1cb559a0a68 100644 --- a/drivers/net/wireless/intel/iwlwifi/pcie/internal.h +++ b/drivers/net/wireless/intel/iwlwifi/pcie/internal.h @@ -1161,8 +1161,7 @@ int iwl_trans_pcie_read_config32(struct iwl_trans *trans, u32 ofs, u32 *val); bool iwl_trans_pcie_grab_nic_access(struct iwl_trans *trans); void iwl_trans_pcie_resched_with_nic_access(struct iwl_trans *trans); -void __releases(nic_access_nobh) -iwl_trans_pcie_release_nic_access(struct iwl_trans *trans); +void iwl_trans_pcie_release_nic_access(struct iwl_trans *trans); void iwl_pcie_alloc_fw_monitor(struct iwl_trans *trans, u8 max_power); int _iwl_pci_probe(struct pci_dev *pdev, const struct pci_device_id *ent, const struct iwl_mac_cfg *mac_cfg, diff --git a/drivers/net/wireless/intel/iwlwifi/pcie/trans.c b/drivers/net/wireless/intel/iwlwifi/pcie/trans.c index 1304d89164240a..bda069a6f7ebfd 100644 --- a/drivers/net/wireless/intel/iwlwifi/pcie/trans.c +++ b/drivers/net/wireless/intel/iwlwifi/pcie/trans.c @@ -2402,10 +2402,10 @@ bool _iwl_trans_pcie_grab_nic_access(struct iwl_trans *trans, bool silent) out: /* - * Fool sparse by faking we release the lock - sparse will - * track nic_access anyway. + * Deliberately return with reg_lock held; the caller must drop it via + * iwl_trans_pcie_release_nic_access(), or explicitly if it wants to + * keep the NIC awake past the critical section (cmd_hold_nic_awake). */ - __release(&trans_pcie->reg_lock); return true; } @@ -2427,24 +2427,19 @@ void iwl_trans_pcie_resched_with_nic_access(struct iwl_trans *trans) { struct iwl_trans_pcie *trans_pcie = IWL_TRANS_GET_PCIE_TRANS(trans); + lockdep_assert_held(&trans_pcie->reg_lock); + spin_unlock_bh(&trans_pcie->reg_lock); cond_resched(); spin_lock_bh(&trans_pcie->reg_lock); } -void __releases(nic_access_nobh) -iwl_trans_pcie_release_nic_access(struct iwl_trans *trans) +void iwl_trans_pcie_release_nic_access(struct iwl_trans *trans) { struct iwl_trans_pcie *trans_pcie = IWL_TRANS_GET_PCIE_TRANS(trans); lockdep_assert_held(&trans_pcie->reg_lock); - /* - * Fool sparse by faking we acquiring the lock - sparse will - * track nic_access anyway. - */ - __acquire(&trans_pcie->reg_lock); - if (trans_pcie->cmd_hold_nic_awake) goto out; if (trans->mac_cfg->device_family >= IWL_DEVICE_FAMILY_BZ) @@ -2460,7 +2455,6 @@ iwl_trans_pcie_release_nic_access(struct iwl_trans *trans) * scheduled on different CPUs (after we drop reg_lock). */ out: - __release(nic_access_nobh); spin_unlock_bh(&trans_pcie->reg_lock); } From b75dd71d260eee7f77b661fc458c5ce767617680 Mon Sep 17 00:00:00 2001 From: Shahar Tzarfati Date: Sat, 26 Sep 2026 20:17:34 +0300 Subject: [PATCH 1001/1012] wifi: iwlwifi: disable HE ER mode HE ER mode (~3 Mbps) provides marginal range benefit over legacy rates, while effective coverage remains constrained by beacon reception. In addition, ~3 Mbps is insufficient for laptop connectivity. Furthermore, any range benefit in 5 and 6 GHz is more easily achieved by roaming to 2.4 GHz, while avoiding degrading 5 and 6 GHz aggregate network capacity with low-rate links. Therefore, disable HE ER mode. Signed-off-by: Shahar Tzarfati Link: https://patch.msgid.link/20260926201527.aac48f46b60e.I5dcf9ed88b2b73bc09c81c0d9c841dc624447faf@changeid Signed-off-by: Miri Korenblit --- drivers/net/wireless/intel/iwlwifi/iwl-nvm-parse.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/drivers/net/wireless/intel/iwlwifi/iwl-nvm-parse.c b/drivers/net/wireless/intel/iwlwifi/iwl-nvm-parse.c index 6bb27d7f1166ef..413dd55115f8f5 100644 --- a/drivers/net/wireless/intel/iwlwifi/iwl-nvm-parse.c +++ b/drivers/net/wireless/intel/iwlwifi/iwl-nvm-parse.c @@ -612,7 +612,6 @@ static const struct ieee80211_sband_iftype_data iwl_iftype_cap[] = { IEEE80211_HE_PHY_CAP7_POWER_BOOST_FACTOR_SUPP | IEEE80211_HE_PHY_CAP7_HE_SU_MU_PPDU_4XLTF_AND_08_US_GI, .phy_cap_info[8] = - IEEE80211_HE_PHY_CAP8_HE_ER_SU_PPDU_4XLTF_AND_08_US_GI | IEEE80211_HE_PHY_CAP8_20MHZ_IN_40MHZ_HE_PPDU_IN_2G | IEEE80211_HE_PHY_CAP8_20MHZ_IN_160MHZ_HE_PPDU | IEEE80211_HE_PHY_CAP8_80MHZ_IN_160MHZ_HE_PPDU | @@ -747,7 +746,6 @@ static const struct ieee80211_sband_iftype_data iwl_iftype_cap[] = { .phy_cap_info[7] = IEEE80211_HE_PHY_CAP7_HE_SU_MU_PPDU_4XLTF_AND_08_US_GI, .phy_cap_info[8] = - IEEE80211_HE_PHY_CAP8_HE_ER_SU_PPDU_4XLTF_AND_08_US_GI | IEEE80211_HE_PHY_CAP8_DCM_MAX_RU_242, .phy_cap_info[9] = IEEE80211_HE_PHY_CAP9_TX_1024_QAM_LESS_THAN_242_TONE_RU | From 039024d55c059b6372f5469fc59f1a5146f155f1 Mon Sep 17 00:00:00 2001 From: Mark Brown Date: Wed, 30 Sep 2026 14:22:55 +0100 Subject: [PATCH 1002/1012] Revert "vdso/gettimeofday: Assert that the clockid fits into the u32 bitmask" This reverts commit ca46a07c4b68246c4e400ce3649ab7228a878d8f due to. /tmp/next/build/lib/vdso/gettimeofday.c: In function '__cvdso_clock_gettime_common': /tmp/next/build/lib/vdso/gettimeofday.c:288:31: error: implicit declaration of function 'BITS_PER_TYPE'; did you mean 'BITS_PER_LONG'? [-Wimplicit-function-declaration] 288 | BUILD_BUG_ON(clock >= BITS_PER_TYPE(msk)); | ^~~~~~~~~~~~~ /tmp/next/build/include/linux/compiler_types.h:682:23: note: in definition of macro '__compiletime_assert' 682 | if (!(condition)) \ | ^~~~~~~~~ /tmp/next/build/include/linux/compiler_types.h:702:9: note: in expansion of macro '_compiletime_assert' 702 | _compiletime_assert(condition, msg, __compiletime_assert_, __COUNTER__) | ^~~~~~~~~~~~~~~~~~~ /tmp/next/build/include/linux/build_bug.h:40:37: note: in expansion of macro 'compiletime_assert' 40 | #define BUILD_BUG_ON_MSG(cond, msg) compiletime_assert(!(cond), msg) | ^~~~~~~~~~~~~~~~~~ /tmp/next/build/include/linux/build_bug.h:51:9: note: in expansion of macro 'BUILD_BUG_ON_MSG' 51 | BUILD_BUG_ON_MSG(condition, "BUILD_BUG_ON failed: " #condition) | ^~~~~~~~~~~~~~~~ /tmp/next/build/lib/vdso/gettimeofday.c:288:9: note: in expansion of macro 'BUILD_BUG_ON' 288 | BUILD_BUG_ON(clock >= BITS_PER_TYPE(msk)); | ^~~~~~~~~~~~ Signed-off-by: Mark Brown --- lib/vdso/gettimeofday.c | 2 -- 1 file changed, 2 deletions(-) diff --git a/lib/vdso/gettimeofday.c b/lib/vdso/gettimeofday.c index ef4dcc61448916..f7a591aba59f11 100644 --- a/lib/vdso/gettimeofday.c +++ b/lib/vdso/gettimeofday.c @@ -285,7 +285,6 @@ __cvdso_clock_gettime_common(const struct vdso_time_data *vd, clockid_t clock, * Convert the clockid to a bitmask and use it to check which * clocks are handled in the VDSO directly. */ - BUILD_BUG_ON(clock >= BITS_PER_TYPE(msk)); msk = 1U << clock; if (likely(msk & VDSO_HRES)) vc = &vc[CS_HRES_COARSE]; @@ -439,7 +438,6 @@ bool __cvdso_clock_getres_common(const struct vdso_time_data *vd, clockid_t cloc * Convert the clockid to a bitmask and use it to check which * clocks are handled in the VDSO directly. */ - BUILD_BUG_ON(clock >= BITS_PER_TYPE(msk)); msk = 1U << clock; if (msk & (VDSO_HRES | VDSO_RAW)) { /* From 6c2cb8b8b843d216ab549b678a0d8831c43153e0 Mon Sep 17 00:00:00 2001 From: Mark Brown Date: Wed, 30 Sep 2026 14:54:37 +0100 Subject: [PATCH 1003/1012] Add linux-next specific files for 20260930 Signed-off-by: Mark Brown --- Next/SHA1s | 433 ++ Next/Trees | 433 ++ Next/merge.log | 17408 ++++++++++++++++++++++++++++++++++++++++++++ localversion-next | 1 + 4 files changed, 18275 insertions(+) create mode 100644 Next/SHA1s create mode 100644 Next/Trees create mode 100644 Next/merge.log create mode 100644 localversion-next diff --git a/Next/SHA1s b/Next/SHA1s new file mode 100644 index 00000000000000..0ae8ae9e27a2c7 --- /dev/null +++ b/Next/SHA1s @@ -0,0 +1,433 @@ +Name SHA1 +---- ---- +origin 551c722f40809618230001baccf219193e22fc5a +ext4-fixes 981fcc5674e67158d24d23e841523eccba19d0e7 +vfs-brauner-fixes b78b728e21c32ec4c330b299f657fb1eb02dffc2 +fscrypt-current cee9395acd8043be0644b25c34bfa86623f2b935 +fsverity-current cee9395acd8043be0644b25c34bfa86623f2b935 +btrfs-fixes 50a31f73ef35e06375895ba37814acc3bd9479db +vfs-fixes 49c5d168a3a8f4eb27d44a2a22b7e8a856ca601f +erofs-fixes 135d84c66f85426299db01a09d93a79a87af18ba +nfsd-fixes f76017a7663c4ce5e379f8a8d39f032bdb1fd865 +v9fs-fixes 028ef9c96e96197026887c0f092424679298aae8 +overlayfs-fixes 4549871118cf616eecdd2d939f78e3b9e1dddc48 +fscrypt a63883d2c3d47ce6d41918c2453a4513b921eaaa +btrfs 55d05b596923df9445a9f6179b8807d6d4c5a50d +ceph dc173b37415e8f738fc4de477490056b479ddc9f +cifs f14572c203d57492e1d4e5d7851a3b143e083b82 +configfs 620938e7da8943970ec25e1c8cdbf44ba1d44fa1 +configfs-rust 6dc9fb1516aa4930329ef561f078c7de9257abf2 +ecryptfs f81cb44f9a4b88d73ee5dec4a1ccdb0232fd2e3f +dlm ed9b6a1296f10e4881d93dfe6d76013fbbaeee87 +erofs a7d28aa0e9b2c983b915d091f9596f0e3b253355 +exfat 216426aff8f69379f03337857551ca0ed3190361 +ext3 ba5855e74bcd761123e39f4708834a0015a74a8b +ext4 9091c97be34083587a75db174aab51551d8e8543 +f2fs a4b9fb9e69116e9ce35e1de2ca2bd940552b8d1d +fsverity b08e4ee274917074633a41b3598d5541cee2b914 +fuse 29b60c6c561f71784c43d4c5d65565ce46e223c6 +gfs2 d0c4bce31a579216780b990a09250f15e00f38ff +jfs dad98c5b2a05ef744af4c884c97066a3c8cdad61 +ksmbd d30f0c30c87d62a7be7fdad295a01860a0d846ef +nfs fd73f4a6659897191fa0d40695fe370925dd3780 +nfs-anna 9bafc322b7fca03717db78615b07bd401367b7c2 +nfsd ac04dab23b5ff28fc7e41957824c5c439ae99887 +ntfs 708f9d56cacae21aeee98d16bcdd50a66edc04a0 +ntfs3 f3b8ee6c05bed24a06192ea8e4cafcc06946c1bd +orangefs 2bc09cb0e9e44c7e622b956332ebfb9b638fd5f7 +overlayfs 1f6ee9be92f8df85a8c9a5a78c20fd39c0c21a95 +ubifs a5e0055eac837a1168c781653943d4a0d9920af3 +v9fs c60ae98c5aa64021751b38ab1313b19d620bf640 +v9fs-ericvh 028ef9c96e96197026887c0f092424679298aae8 +xfs b4787e7b9730d52f33cf8dd23b3c42f1237f9753 +zonefs 3a8389d42bdf4213730f4067f8bfa78bae6564ef +vfs-brauner 84086827932b58e7645d93d970bbc566c4ee408b +vfs 4dda01b67c8662c5d0c53034974cb8280c575a5d +mm-fixes ffd79f732828e5fe7c9f97f9e5930f95db70583b +fs-current 5a9410e0a34d2e31af3e2c97d2dc428d08e41f39 +kbuild-current fd73f4a6659897191fa0d40695fe370925dd3780 +clang-fixes-current df2908090cda368b01ff43709f51890076c56157 +arc-current f050c3e61d2a1aaece3170d459e4ce5fa2486c12 +arm-current 1039bffd6ae9c75b42b7d148d6c1106134107b66 +arm64-fixes 3872cc6b92940af0f73c6f9a45d65300492d648d +arm-soc-fixes 4dd1999783d7d12434006289338373e49492dc96 +davinci-current cee9395acd8043be0644b25c34bfa86623f2b935 +realtek-fixes dc59e4fea9d83f03bad6bddf3fa2e52491777482 +drivers-memory-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +sophgo-fixes 19272b37aa4f83ca52bdf9c16d5d81bdd1354494 +sophgo-soc-fixes 0af2f6be1b4281385b618cb86ad946eded089ac8 +m68k-current 2f8e3cad53b5c36ab0ed5d3195bfc55c59ea61a5 +powerpc-fixes 93f51579e7df248780214094418f205253383cc5 +s390-fixes 5b76268dac968612f7283d59b539036de955b7d9 +net 99b43ede9e355ba35244cc9470bf1819774ce39d +bpf 564b7e8e09ec9b47921dc11de0de05da358335d9 +ipsec 868f63c8bfafa9b827168c9126f85264c39c02ec +netfilter 9c572a83037a7dcd653ba3a9cc468c16b857d0c9 +ipvs a401a9d547c50ef34db1088cc1fb9a201a7af657 +bluetooth-fixes 86ef0f58bdecdedb3a1240971c56d71b7e4ce3fc +wireless 6f63e919fe1e335b8abcb3a28bfd4804a98d875a +ath 6f63e919fe1e335b8abcb3a28bfd4804a98d875a +iwlwifi 6f63e919fe1e335b8abcb3a28bfd4804a98d875a +wpan 2b4707a149a55e8fa75c9ef32b359d60f470a566 +rdma-fixes 72d3fcf802c45d00b300f25b848a93c3a2bd7c7e +sound-current a154f7b9f197b3d5e41fa3ad62083f5aa93b1fe8 +sound-asoc-fixes b2047b8cadadccb1c9269ce756c399cd9aef2c82 +regmap-fixes 2e42cade8ff1ff579e77976d9869db4df1feaf74 +regulator-fixes f3e6ef13e24c9f26dca0d35de57fcdf04f78e378 +spi-fixes 3d743adf090cd4c9a2120c1e02b0482e88aa0d2d +pci-current cee9395acd8043be0644b25c34bfa86623f2b935 +driver-core.current 72d3fcf802c45d00b300f25b848a93c3a2bd7c7e +tty.current 5dad87615c9861cfa366ca984b52f581e861df20 +usb.current abc36cbda29d8f19cf3a580cd86ca9e865186a41 +usb-serial-fixes ec06546b43b612cefc851558981b4955c0ef9e46 +phy 93f51579e7df248780214094418f205253383cc5 +staging.current df2908090cda368b01ff43709f51890076c56157 +iio-fixes d2f2868d2f487b418f68db2ef5bd7cf41722ee68 +watchdog-fixes 2686e13649a9e846f6c08cf4ada0256ca19c3475 +counter-current fd73f4a6659897191fa0d40695fe370925dd3780 +char-misc.current 540f55de8c9f79c83f13e44910955faf82b0d79c +soundwire-fixes 93f51579e7df248780214094418f205253383cc5 +thunderbolt-fixes 395e9f2967a7ac6898026e8f136c369805dc2fd9 +input-current 309731e95917125bbd13626a7a5600490a5bf44f +crypto-current 10396a2d6d41d594975b6ece712278570c3c970c +libcrypto-fixes 6d996c3b974c501f347e8bad1644afd1180b2795 +vfio-fixes e242e974e812e7a47e3088860c80d9492fac314f +kselftest-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +dmaengine-fixes 93f51579e7df248780214094418f205253383cc5 +backlight-fixes dc59e4fea9d83f03bad6bddf3fa2e52491777482 +mtd-fixes 1c1a342aceec79528b3ba51f376eca2be5928fe7 +mfd-fixes d5d2d7a8d8be18681a0864f58e3875f1c639e11c +v4l-dvb-fixes 2579cbe68005f46fc7f8f95364f6b101b07b9d1c +reset-fixes 71827776667f4e4677a4fa806bcfb24d4b8dd9d7 +mips-fixes 93f51579e7df248780214094418f205253383cc5 +at91-fixes afb1ecfeda3fc31d6300a176f4854cc62e289bf3 +omap-fixes 2fabd2f406d0cf787be47b3bbeea99b30004033d +tegra-fixes 52574866649979631b76b601053be903ab2453bd +kvm-fixes 973ea70393e885e540f714904e51bc6cac80e3d7 +kvms390-fixes f47190b08b71e8482072978373ee88cb2dfbdaf4 +kvm-arm-fixes afb4334fb52c7b416bbc9c4b166f6806509f734b +hwmon-fixes 61406e9cac695b19b979b002d790d346b9f07887 +nvdimm-fixes a8aec14230322ed8f1e8042b6d656c1631d41163 +cxl-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +dma-mapping-fixes 057e5e07420c753f248f9ce148040ab6dcf8359f +drivers-x86-fixes d144a494d81fcf2d1c5cf58b01c655bb8bafc701 +samsung-krzk-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pinctrl-samsung-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pinctrl-qcom-fixes 19fc4240358be2a25ce8e0a49a2819588ed61c5d +devicetree-fixes 21c89ff1fc86ffa517f3a9ca1ba5c48e9e37c1ba +dt-krzk-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +scsi-fixes 42d1221d321e55afc7bba9109a77aaf5a817c8a3 +drm-fixes a9ed3aa9b87ee41e8ab3ff471b1331c154767e45 +drm-intel-fixes c034e8a46e4cb703018fe7e10fe5a3a974d6a7c3 +mmc-fixes aa0a37b5024d921e96ac4f8f487c8bfbf1f1a582 +rtc-fixes 055ef5ce9f67a6a3a1363fa663007f8196cc0fb8 +gnss-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +hyperv-fixes ca039df94983c441fdee0e2488263fc744e8ecb6 +risc-v-fixes e1116094833337fe9480fb9527a59c7637b96118 +riscv-dt-fixes 0f70fd6f4c1d6c1f2ea298d8dd07c4a4d5dff9af +riscv-soc-fixes dc59e4fea9d83f03bad6bddf3fa2e52491777482 +fpga-fixes 19272b37aa4f83ca52bdf9c16d5d81bdd1354494 +spdx cee9395acd8043be0644b25c34bfa86623f2b935 +gpio-brgl-fixes ff82fc3a4a1d417dee1229681cc5285986edb1fa +gpio-intel-fixes 8d37f20173a2760718bb02b58bafc5a7cac93d67 +pinctrl-intel-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +auxdisplay-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +kunit-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +renesas-fixes 8dc2615d5702059b2b71fca6f93c0d7d10ae54cb +perf-current aadea57f532882d8bab444646863c7ef8a778ff1 +efi-fixes d8809f6931065cbbf3554647a50a65a471ab5983 +battery-fixes a58cbc8b36ec0cd00d2ba3de7d003de818a27523 +iommufd-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +rust-fixes 551c722f40809618230001baccf219193e22fc5a +w1-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pmdomain-fixes 4c66c423593e26273ea0d644906ed007630c6f6d +i2c-andi-fixes 840d8acc87925deb3c75566fde9ca81d2f47f98c +i2c-rust-fixes 4eb422482ca5d924d7212ad2ca1cb7ea6f5b524d +sparc-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +clk-fixes 81493c1dd1b1ebecb2843a7815973a5bd9a37e5a +thead-clk-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +tenstorrent-clk-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +fustini-config-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +pwrseq-fixes 58a00330248154461c52a6f3bd5c01e5b67f9560 +thead-dt-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +ftrace-fixes 1650a1b6cb1ae6cb99bb4fce21b30ebdf9fc238e +ring-buffer-fixes 057caace5214da3b457bbd295e1a2ad34d3685ea +trace-fixes d860c67c051685abb0460b593b193f0f45f4fa92 +tracefs-fixes 07004a8c4b572171934390148ee48c4175c77eed +spacemit-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +tip-fixes 3ebb3531c6d0d533fddc65fa3bed174291d71956 +kexec-fixes a901b0778ae82d46084b55237e65d3a7bb5f0c86 +liveupdate-fixes 3a0b8fa2eb36afc88b62a95f33f0c77c71fa5ded +drm-msm-fixes a15fac810c76397ec9f62a6fc26c4d7ab6e238a7 +uml-fixes af421e9aed3920c7ac88c24daa48606c7112feca +fwctl-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +devsec-tsm-fixes c3fd16c3b98ed726294feab2f94f876290bf7b61 +drm-rust-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +tenstorrent-dt-fixes cee9395acd8043be0644b25c34bfa86623f2b935 +nfc-fixes b61732f47316d45f27706db7812950145d3327b5 +mm-nonmm-hotfixes-stable a243ede718463c7b481878656f1ff32a0ce0fd54 +mm-nonmm-hotfixes-unstable 844c67369f72de6c89ad783b3fd1d59f8e8f4902 +drm-misc-fixes e78a9fb7a40c55ec2a70ae39903dde4f898af2ef +rust 72d3fcf802c45d00b300f25b848a93c3a2bd7c7e +rust-interop 05f7e89ab9731565d8a62e3b5d1ec206485eeb0b +rust-alloc 6e8339118040f24ebcc21db22c9c1b7398b7e39e +rust-io 86731a2a651e58953fc949573895f2fa6d456841 +rust-pin-init 95593ed3616194c9498cc190cd1a3693ff4bca37 +rust-timekeeping 2ea0119f72dba597aa8a98cbdb72c564bfc5cb38 +rust-xarray c455f19bbe6104debd980bb15515faf716bd81b8 +rust-analyzer 5f45afb8ab04d934fc0601a202f95267ebc20059 +mm 7a6c9825d917f0428982af72a233a36dbf45cf8b +mm-nonmm-stable a243ede718463c7b481878656f1ff32a0ce0fd54 +mm-nonmm-unstable cbb17a7a0afe4f2a766b25e75777f70f47dbc6ba +kbuild 59ba9b8f8803c132f59db74bdd4b809c30dbe3c7 +clang-fixes 0bb666d5f5a2339a5692afb312c2161df9620503 +clang-format 8f0b4cce4481fb22653697cced8d0d04027cb1e8 +perf 45d15e89a783a0a279b7a54f9018c230127380d7 +compiler-attributes 8f0b4cce4481fb22653697cced8d0d04027cb1e8 +dma-mapping 57a57ae077f0f167056fc08f1666678f78e5b147 +asm-generic adbbd9714f8058730f93c8df5c5bf1679456424b +alpha d58041d2c63e09a1c9083e0e9f4151e487c4e16a +arm 1a89abc009cb5035d0cb4e5ba48d32b94f30e8e4 +arm64 6c360697fd4d17a53ed977a1341add3520a19269 +arm-perf cee9395acd8043be0644b25c34bfa86623f2b935 +arm-soc b9bb4134a9583a388704433eef22c82c49963585 +amlogic a2530c83c458a503e55219966217dfd1c36a385e +asahi-soc 9378cd5ddcebc06fdd803373487ed1a376532391 +at91 bd49a1c72abbdddbc7b3d691d926623e33a7c1a0 +bmc cd7d1ef7d74ed5f5a1790a40389bc38382924556 +broadcom c0f2033a9ce4b843ca3e947536a0607ecf982be1 +cix a0cffbd8878c55b12dc4555f883adf3372cecd91 +davinci cee9395acd8043be0644b25c34bfa86623f2b935 +drivers-memory a22355280361d2a376a2020059a2bae11e6cea10 +fsl 7b3b0598c00e67f2df85d0b4cd99eaa3be6c9bfd +imx-mxs 5d034f17b24642b2a764c0e3a68b4c6937c9ba75 +mediatek 4bfec57314396c8c816ee78f4803987882b668a7 +mvebu 6dfda79476538a9aad1d2dac45453e62c6939522 +omap 7cdd46c9d6c5a7f4800195af82bb718081e7c84d +qcom e081f1e68766ca772101bf1f9e14368166fe0ae7 +realtek 3c778f0c9fa36b6b9a26bd2146df24ed4e830ba4 +renesas 7687cd01a980ac59572031b1e3cbe30bf09e90fa +reset d373605cd514837d8a6de3d00c786d4bae6dbaf8 +rockchip 23d8f49fcae5d72b77744c7f349c12462c3bb748 +samsung-krzk 08df370772f32b2bce3f88eed253b6a1b0035e40 +scmi 0dcfd0f40b5bee8583eb59ed546189523452c0ae +sophgo 76acfee87c74dc0dc7a68a00b70b9c39d1c8428f +sophgo-soc c8754c7deab4cbfa947fa2d656cbaf83771828ef +spacemit 4b98722be2911a922fb29b54adccdee46f6a9f41 +stm32 e6cdc9f49eebe486c256e134be84175865a357e9 +sunxi afb622bddaa6fc7b6d205fd080b9c0d3ec853431 +tee a4df08db64528b4b605924a0a1f874ffdb99545f +tegra 11430072551f5d5a37f58e390b3d042035992ca4 +tenstorrent-dt cee9395acd8043be0644b25c34bfa86623f2b935 +fustini-config cee9395acd8043be0644b25c34bfa86623f2b935 +thead-dt cee9395acd8043be0644b25c34bfa86623f2b935 +ti ba727be7fe4a64ddc6aa2c178204cf6a36d44cc4 +xilinx bf126d9aa20f2d8bb1c71f5b7f7ba216178fd80d +socfpga 04a1d21330501b113ba04efb81724f1c95ad2db8 +clk b1470947e92aeedd522ddb7363369c956e55d6b6 +clk-imx 39ec460b56b26319d1f31b459e8be7ea34ae9c67 +clk-renesas 21cbd7ec29930f8e7e156e18b81736f099f03849 +thead-clk cee9395acd8043be0644b25c34bfa86623f2b935 +tenstorrent-clk c64b54ddb692c30b8fb139a2ead32eb4db7faef5 +csky abb81e5ce7d995baa41556b8125fa59e28ba3be8 +loongarch a2628ce4ddb6873e35380a42396d17a66e704a1a +m68k af32c3cb72529b14bfb2685bbdaa2b1514496b58 +m68knommu 29036c5910df0f0a4c824d7da5b6254c7d4ce500 +microblaze 6b125c73aaa0dab9ce625437c63b176afe453776 +mips f552e14f77bfce3e7ee0bb268ab023951baec3ff +openrisc 6620f5e8c11c4f7e41222a86f5c97150cc5f84a5 +parisc-hd a5f6df25261b31eb75f61047a890515671d95a29 +powerpc 12d238d584ba485a99e02e033746100fc7099184 +risc-v a06776565b9e73512e880a95b66c198ae8e7e24a +riscv-dt f55a54f07a367ed5c73ae53398a60c01219a6df3 +riscv-soc 00bdd4d4daaae140ff747475ef475facee61a950 +s390 c57ae907d4f9ae822bdcd7b16f2d0dc56fb7973b +sh be5a19d95030ffa7a37d2981d0ab4eac8a6c89fa +sparc cee9395acd8043be0644b25c34bfa86623f2b935 +uml 2f88f5689de1a039764d00f466209acf0c010eaf +xtensa 28722ed2527aa63ac1454013defe718c56032886 +fs-next bf234c28d9e24e3d6c42a6202e3f92fba514c2ae +printk 3fa6f22c1dc835fe1757b06e7d51d8c424ac978a +pci 81bebadfa0f7f9337a0b4e8e218731457b7eb9af +pstore 7c756181175d50200be487affa905e973f6cbaf5 +hid 145c2b2e9a5c0f794fb4009bcb072ab19f8ccfcd +i2c 8cd9520d35a6c38db6567e97dd93b1f11f185dc6 +i2c-andi e33eb6c27c095649c09465fbf75ada7d492ba83a +i2c-rust 61ddec70c9bcc3d4b0471c8683e57d3b03cecf0a +i3c bf9ec06c90dcd8ccba116fbc065b5d918de47644 +dmi 1afafbaf749d8e8ec53f8e38efdc731131902b5b +hwmon-staging 4781ca52761e666cf18b591e6bb0478396c90320 +jc_docs 2d72a4c09867a8f9131afc2e705a728f9fba2417 +v4l-dvb 58348f64125e9a3e44d3abb275ca7f4e6c9641e5 +v4l-dvb-next adc218676eef25575469234709c2d87185ca223a +pm 114920a3e82922869f8c27916e852b4ba0d560c4 +cpufreq-arm d82d896f00e7bf697b61d5df0555941afa8d0657 +cpupower cee9395acd8043be0644b25c34bfa86623f2b935 +devfreq 9a222650d9e70d0027126e4df5b00b1b5b678a97 +pmdomain d1a93cf1d3bed4667de58a561afb9682430e4f50 +opp 1f6de65e3314519ce462bfd3a1d29e918f97e4e5 +thermal e856ca3013b3074fd07cb81c90b1de332f69153c +rdma 485effd117d0310a7f589defb4f442fa3bb52ae2 +net-next 47a1446725732cd3996edf607e8739334bbf4d78 +bpf-next acff58e305175df35985082b0e79103a2497f702 +ipsec-next 014d795c73837ea2339a4ea8e8f82c6e959b845d +mlx5-next 36b1d3299d0b6aa51485cc9a79b8d94948d5f3d4 +netfilter-next 87b80c2f6b05cad9f0ff9136709c62a0f59923e3 +ipvs-next 6ebcf5074cff0402730c6981d2397139fee6322d +bluetooth fdd5964bd389936856971647fcac3442117009dc +wireless-next 21b4248bfa0f410fa22202fc2c82f4808d672cda +ath-next 21b4248bfa0f410fa22202fc2c82f4808d672cda +iwlwifi-next b75dd71d260eee7f77b661fc458c5ce767617680 +wpan-next a6bfdfcc6711d1d5a92e98644359dedc67c0c858 +wpan-staging a6bfdfcc6711d1d5a92e98644359dedc67c0c858 +mtd 112a666bd82e96e4a01e0cd8a0fb9c88dd0bce38 +nand 55c5b6d5f59f59a8c98a7effc3195226cc20175c +spi-nor 68d7115d77a873c28d8c7d2f239c3d9ca83bf39f +crypto 67aefeccc4101a93627f36abbd049f191b54f903 +libcrypto b63b9b3d5ddaf0f1767f31aeef7aef08bb36225e +drm 2abe8e7e33973dda5dd24b89aa3ee52470a78265 +drm-exynos 3a8660878839faadb4f1a6dd72c3179c1df56787 +drm-misc 744f262401cc2e1f3827496c72a47e072d31a852 +amdgpu 76b9706e7fa1a6e374703170128b1be2f590cda7 +drm-intel c9e608d1247cd4c286a54d7f26188c41e8f0a6c2 +drm-msm d33622598496c8994c35ba0a8a913de064dfc1a4 +drm-msm-lumag 140b13475302601368c0cf4e193e66126a49feb3 +drm-xe 3fc93d311249d5ad969a520dd3aae73bc007d431 +drm-rust 658c5f0042e6e897338dd04330a363aa7022c85d +drm-nova 93296e9d9528f0d87f2cf3fee494599060a0f14a +etnaviv 6bde14ba5f7ef59e103ac317df6cc5ac4291ff4a +fbdev 9f4c6043f33c95db7e46d6ca198344ce0ca2e6c5 +regmap 117e6a5fd98fb7002e70ba39c3ba393498399982 +sound d4febca239be3c91498762dd0535a87d4af1eabc +ieee1394 a6e7c3836b81235df08814bd409f7a23efcfb34d +sound-asoc 0fbfaeecfd6f5925bfcd7cfcbd62860a9e868619 +modules d6bfb4a05affc8083a45feefe0bbf715968e0d9e +input daae2ab46e0cb612f50ca1d86cf50e5962461ae5 +block 0183ac11c13abc35710b3356919f42806bd961e8 +device-mapper bec2fa7bdf06d6523e40e05fd93117b0c1c8bd30 +libata cfce1dc635041d6b9dd5fad5367a331faaf644fe +pcmcia b3c26ea81ccc522e77ed0b1707add61fc9206216 +mmc c2063613bceaaffe864ff781082845d322ffb2e1 +mfd 319633b06ff2bfc2a7a60d2cfcab2e681a343b27 +backlight 5e1631df673f3d5355acd06263a884e00d8d007b +battery 4fc88ba435dadbc05990951e3f3fbd8ccd2df140 +regulator 26f25a65cda68bab7db1152b1f37d856c8a49ae1 +security 881f19c2ffbc73f351b257e09c71e6059939b770 +apparmor cee9395acd8043be0644b25c34bfa86623f2b935 +integrity 1dc317438308c79cf90a6f14d3e4116347ea3716 +selinux f2b37cf384b0111b975239f4dc75930368023058 +smack fedc88e38ce979a720cd2de042578cb5df3dc8de +tomoyo 72d3fcf802c45d00b300f25b848a93c3a2bd7c7e +tpmdd-tpm 015fb29a748342a186f37b602ac70017098c8251 +tpmdd-keys e7ac8fd885298225076abd92b82c1f31a6bc8320 +watchdog 8b5a9f09037e3c372c4e9f2fbda85fcfbf5ea5f0 +iommu cec2dc7663e7182c311162163d71b1bbdfdbf09b +audit 88bd5e852addf09db73a38d93f7911863ef5a104 +devicetree 833aaa4790eee9812269b776eb8672acf1524741 +dt-krzk dfcea1641506d7ba096a9e6ead5d15f5573206ec +mailbox 14af7a96afa39f4f3c1972705489b9ba15c01857 +spi 168b675c187d40141944d76cedee8a1061beba68 +tip 1aeb52f7869a680c042fc9ae806281e8f60469f7 +kexec 9d0b028715b007188a61ce70b9656ead84a2b3ef +liveupdate 5221d141653a83a974595b4e5eedd9d6983ab270 +clockevents 1b8b356b4b06e3a28feb852c04e06caeac07bc87 +edac 898e6a2c5ce0e4aba55d126bbe610f0f77e71226 +ftrace 6dc993d4520fdaa083c8871e5fee443117d87057 +rcu aebf6c4777d534ccd089b3ccab98d71ea3167a0d +paulmck c87605b21fdd40aef7d647baff6685fa1aefee1a +kvm d4b7fb647204f0c81dfeae2d1a708e4d858e0c94 +kvm-arm 8c00199d322b9e5e936bb0ff624ac95b01c740fe +kvms390 044ae0767d8cc1fc2a53930361f05c9ed2c3c081 +kvm-ppc 93f51579e7df248780214094418f205253383cc5 +kvm-riscv 41e81f7e3ef96594fb840445343c0ee7723aa550 +kvm-x86 b378201ccd5280d0fff89bbe55e1eb00620ec0d5 +xen-tip 93f51579e7df248780214094418f205253383cc5 +percpu 8f0b4cce4481fb22653697cced8d0d04027cb1e8 +workqueues fee1265a725725dac0ff7459f912d69558798b36 +sched-ext 91a186b9e599f08f79abfbdbbbcf22443f966a6a +drivers-x86 fe5030c8cc7156223f48530e9b49aa87c0305bcd +chrome-platform 5859f97c6f404d07e89fe9c0d20a318ebf200bd1 +chrome-platform-firmware 0e30b98a545991e440a947fb24f9df4ded9798f0 +hsi e81250ec6b69248b00d38c523dc6a13efaf38aab +leds-lj 05b4738b0078f7d6f154f68068a11c8a0635e9df +ipmi 89a312991dc6e638a36adc43ccb91dbc25504c04 +driver-core f1850e443b0e4f2429ddf42a8d5033ea54ae8a90 +usb d58dffe9ee2c8883193959ff4ef995ec07932874 +thunderbolt a93a8e3200002f0c345fec4c340390e9a60aabc7 +usb-serial 6583f9741341b98ade67aa764794a02fea1eb7db +tty bf961847813d9edcb5a9c089714069428d141fa9 +char-misc 315860c4f912990c75ef228775ea8871fc5bcc9a +coresight 9e3604d7369cfc0110100eb1a0acab1865ee2d18 +fastrpc ef071c4906eb45d16b60f09154cf0bc6ec8f5435 +fpga 093da48782df3d50953c520626b345ca70af8cbd +icc 3a6d690152557706cc3d44f6402d40c6c1ea7d74 +iio a3b3580713f3ac5a32dc2874ee546828977a1d68 +nfc fd73f4a6659897191fa0d40695fe370925dd3780 +phy-next c7f2322431cb6d108b18fb4154606b49e2bc50f7 +soundwire 90b63b309fd6c0f192337659e73e3047a67df99f +extcon 8d3ae59288f1e7d58d76558a6ee96d533bc5019f +gnss cee9395acd8043be0644b25c34bfa86623f2b935 +vfio b30b52c2fb82e6d38ee2845a4d85e8254b44f3d8 +w1 813a5b9a3c9f6cee830b1e23343fbd2359683557 +spmi 8cdeaa50eae8dad34885515f62559ee83e7e8dda +staging 8444548bd905f22093729065408284a6b46f7eee +counter-next edac5cf35699492027fb54e348dc6234b2661617 +mux ac7bde3c53166656d80e3aacc7d274d3c60a6de4 +dmaengine 0a8dda0a15d3926422d286567f945a05328a4ac6 +cgroup 8deb0752fa76101ff2a0c5cdaf837d95ed0d0046 +scsi 6147f16c23efb58a98fdfc71b794b063dc01c767 +scsi-mkp f09d2c7485b32adb82336d0d748935c8237a649e +vhost 8f2c2fb94a01320e5136c6e89a47fb12355ad27c +rpmsg 7fbd9a5319c33204a719294d90c08abb15a5a46c +gpio-brgl c4e74a7058b573617f142e07b4b5bc9eee04f4c8 +gpio-intel 0fc424b6a8ef447450f6087db42abce3a6feb328 +pinctrl c34fbaaac34ba9a85af0f16a0e1d2b98b2d05b25 +pinctrl-intel c016587866e573fa8dff50c3bdae9734c4418099 +pinctrl-renesas 0cd4a7b3a4883970df3ae8814d9aabdecf65a81e +pinctrl-samsung cee9395acd8043be0644b25c34bfa86623f2b935 +pinctrl-qcom 4d7c9430a26aea6a69521af3aa2d78dd5bc8d3ed +pwm e74b9a7ee50071aad25d3989bf985292f90b0c6e +ktest 932cdaf3e273a2727e77af97f79f12577174c5a0 +kselftest 30af56a227e27b892d34bc44c23298b5274913e3 +kunit cee9395acd8043be0644b25c34bfa86623f2b935 +kunit-next e38f53f0482468efd04397f66bda4648b70ac9fa +livepatching 5d791d3396ca4c9e5fd9d20b49386c20d8179eb8 +rtc cee9395acd8043be0644b25c34bfa86623f2b935 +nvdimm e99cb3ecd8334ca21e01ff9a79a916693f58f1fb +at24 dc59e4fea9d83f03bad6bddf3fa2e52491777482 +ntb dc59e4fea9d83f03bad6bddf3fa2e52491777482 +seccomp 832b9b176be06a20747a267db6bc0010a24b74f4 +slimbus 4350d705466992dce4a21c382985553bfb568f65 +nvmem 41f42ff4ae134eba8dbe3424cf2ca1824c652edd +hyperv be0cfab740e58b70047ef6e7e3d578f00ed5d258 +auxdisplay f63dc0eea92d07c5bb799a406571b555cd2bbd97 +kgdb fdbdd0ccb30af18d3b29e714ac8d5ab6163279e0 +hmm cee9395acd8043be0644b25c34bfa86623f2b935 +cfi dc59e4fea9d83f03bad6bddf3fa2e52491777482 +mhi 710bf7329abf86e37db529f19c8f394099238892 +cxl 71392a644e88c6bb921cdf12cb2b00adec070fe4 +zstd 65d1f5507ed2c78c64fce40e44e5574a9419eb09 +efi 7eef16311a234b5d77c1493e19c6538a3aa1a203 +unicode a511442085c140da9cdbe60e3f0fab1c71480801 +random 703749b069d53b55f92499e088b30b031e47b8f9 +landlock 02619311dbfc4351e78e5dc6652532cbe7603661 +sysctl 4991c8b72b564cd16cb48124f880617fd49f1013 +execve df2908090cda368b01ff43709f51890076c56157 +bitmap 452a6d5b5e0556848a8c28428c8f86874ea4ee02 +hte 30167fadbf87fa901a2fee49c30a5c5a506a43e8 +kspp 760f96b7f54beb8dc6f85d97ac7854fddba86835 +nolibc da27722051fdf0775d8c5b3e33dc21bcf605dbd4 +iommufd 54dadb030c7e2350957855d3995de05ae02c2e66 +turbostat ccdfcb7e7ab84573046061ece61bfd2368577f1e +pwrseq 09baa2f4caab013160eed975db47235e6c930af4 +capabilities-next 507adb6448378155ec89495fed8c3c09d8c83920 +ipe bcaa4d1d69368b6034b1ceb226ffd0a203a5fbc7 +kcsan a8488ecbd7ba44d65b912dfe88a73f438eba2447 +crc cee9395acd8043be0644b25c34bfa86623f2b935 +keys-next 965e9a2cf23b066d8bdeb690dff9cd7089c5f667 +fwctl cee9395acd8043be0644b25c34bfa86623f2b935 +devsec-tsm 3177779ae17db4c66c851f799505fb95c7530c03 +hisilicon a5db65458a911daa6f8927264366dab501715dc0 +device-id 995832b2cebe6969d1b42635db698803ee31294d +kthread fa39ec4f89f2637ed1cdbcde3656825951787668 +pagemap-headers e02cb91d4644dfff593146f89e70d2008cc6ac16 diff --git a/Next/Trees b/Next/Trees new file mode 100644 index 00000000000000..32b5b670b60a62 --- /dev/null +++ b/Next/Trees @@ -0,0 +1,433 @@ +Trees included into this release: + +Name Url +---- --- +origin https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git#master +ext4-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tytso/ext4.git#fixes +vfs-brauner-fixes https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git#vfs.fixes +fscrypt-current https://git.kernel.org/pub/scm/fs/fscrypt/linux.git#for-current +fsverity-current https://git.kernel.org/pub/scm/fs/fsverity/linux.git#for-current +btrfs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git#next-fixes +vfs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git#fixes +erofs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git#fixes +nfsd-fixes https://git.kernel.org/pub/scm/linux/kernel/git/cel/linux#nfsd-fixes +v9fs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/ericvh/v9fs.git#fixes/next +overlayfs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/overlayfs/vfs.git#ovl-fixes +fscrypt https://git.kernel.org/pub/scm/fs/fscrypt/linux.git#for-next +btrfs https://git.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git#for-next +ceph https://github.com/ceph/ceph-client.git#master +cifs https://git.manguebit.org/linux.git#cifs-next +configfs https://git.kernel.org/pub/scm/linux/kernel/git/leitao/linux.git#configfs-next +configfs-rust https://git.kernel.org/pub/scm/linux/kernel/git/a.hindborg/linux.git#configfs-next +ecryptfs https://git.kernel.org/pub/scm/linux/kernel/git/tyhicks/ecryptfs.git#next +dlm https://git.kernel.org/pub/scm/linux/kernel/git/teigland/linux-dlm.git#next +erofs https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git#dev +exfat https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/exfat.git#dev +ext3 https://git.kernel.org/pub/scm/linux/kernel/git/jack/linux-fs.git#for_next +ext4 https://git.kernel.org/pub/scm/linux/kernel/git/tytso/ext4.git#dev +f2fs https://git.kernel.org/pub/scm/linux/kernel/git/jaegeuk/f2fs.git#dev +fsverity https://git.kernel.org/pub/scm/fs/fsverity/linux.git#for-next +fuse https://git.kernel.org/pub/scm/linux/kernel/git/mszeredi/fuse.git#for-next +gfs2 https://git.kernel.org/pub/scm/linux/kernel/git/gfs2/linux-gfs2.git#for-next +jfs https://github.com/kleikamp/linux-shaggy.git#jfs-next +ksmbd https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/smb.git#ksmbd-for-next +nfs git://git.linux-nfs.org/projects/trondmy/nfs-2.6.git#linux-next +nfs-anna git://git.linux-nfs.org/projects/anna/linux-nfs.git#linux-next +nfsd https://git.kernel.org/pub/scm/linux/kernel/git/cel/linux#nfsd-next +ntfs https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/ntfs.git#ntfs-next +ntfs3 https://github.com/Paragon-Software-Group/linux-ntfs3.git#master +orangefs https://git.kernel.org/pub/scm/linux/kernel/git/hubcap/linux.git#for-next +overlayfs https://git.kernel.org/pub/scm/linux/kernel/git/overlayfs/vfs.git#overlayfs-next +ubifs https://git.kernel.org/pub/scm/linux/kernel/git/rw/ubifs.git#next +v9fs https://github.com/martinetd/linux#9p-next +v9fs-ericvh https://git.kernel.org/pub/scm/linux/kernel/git/ericvh/v9fs.git#ericvh/for-next +xfs https://git.kernel.org/pub/scm/fs/xfs/xfs-linux.git#for-next +zonefs https://git.kernel.org/pub/scm/linux/kernel/git/dlemoal/zonefs.git#for-next +vfs-brauner https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git#vfs.all +vfs https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git#for-next +mm-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mm/linux.git#for-next-fixes +kbuild-current https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git#kbuild-fixes-for-next +clang-fixes-current https://git.kernel.org/pub/scm/linux/kernel/git/nathan/linux.git#clang-fixes-for-current +arc-current https://git.kernel.org/pub/scm/linux/kernel/git/vgupta/arc.git#for-curr +arm-current https://git.kernel.org/pub/scm/linux/kernel/git/rmk/linux.git#fixes +arm64-fixes https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux#for-next/fixes +arm-soc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc.git#arm/fixes +davinci-current https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#davinci/for-current +realtek-fixes https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux.git#fixes +drivers-memory-fixes https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl.git#fixes +sophgo-fixes https://github.com/sophgo/linux.git#fixes +sophgo-soc-fixes https://github.com/sophgo/linux.git#soc-fixes +m68k-current https://git.kernel.org/pub/scm/linux/kernel/git/geert/linux-m68k.git#for-linus +powerpc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git#fixes +s390-fixes https://git.kernel.org/pub/scm/linux/kernel/git/s390/linux.git#fixes +net https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net.git#main +bpf https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf.git/#master +ipsec https://git.kernel.org/pub/scm/linux/kernel/git/klassert/ipsec.git#master +netfilter https://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf.git#main +ipvs https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs.git#main +bluetooth-fixes https://git.kernel.org/pub/scm/linux/kernel/git/bluetooth/bluetooth.git#master +wireless https://git.kernel.org/pub/scm/linux/kernel/git/wireless/wireless.git#for-next +ath https://git.kernel.org/pub/scm/linux/kernel/git/ath/ath.git#for-current +iwlwifi https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git#fixes +wpan https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan.git#master +rdma-fixes https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git#for-rc +sound-current https://git.kernel.org/pub/scm/linux/kernel/git/tiwai/sound.git#for-linus +sound-asoc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/broonie/sound.git#for-linus +regmap-fixes https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regmap.git#for-linus +regulator-fixes https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regulator.git#for-linus +spi-fixes https://git.kernel.org/pub/scm/linux/kernel/git/broonie/spi.git#for-linus +pci-current https://git.kernel.org/pub/scm/linux/kernel/git/pci/pci.git#for-linus +driver-core.current https://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git#driver-core-linus +tty.current https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git#tty-linus +usb.current https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/usb.git#usb-linus +usb-serial-fixes https://git.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial.git#usb-linus +phy https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git#fixes +staging.current https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/staging.git#staging-linus +iio-fixes https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git#fixes-togreg +watchdog-fixes https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git#watchdog +counter-current https://git.kernel.org/pub/scm/linux/kernel/git/wbg/counter.git#counter-current +char-misc.current https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git#char-misc-linus +soundwire-fixes https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/soundwire.git#fixes +thunderbolt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt.git#fixes +input-current https://git.kernel.org/pub/scm/linux/kernel/git/dtor/input.git#for-linus +crypto-current https://git.kernel.org/pub/scm/linux/kernel/git/herbert/crypto-2.6.git#master +libcrypto-fixes https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git#libcrypto-fixes +vfio-fixes https://github.com/awilliam/linux-vfio.git#for-linus +kselftest-fixes https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#fixes +dmaengine-fixes https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git#fixes +backlight-fixes https://git.kernel.org/pub/scm/linux/kernel/git/lee/backlight.git#for-backlight-fixes +mtd-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#mtd/fixes +mfd-fixes https://git.kernel.org/pub/scm/linux/kernel/git/lee/mfd.git#for-mfd-fixes +v4l-dvb-fixes git://linuxtv.org/media-ci/media-pending.git#fixes +reset-fixes https://git.kernel.org/pub/scm/linux/kernel/git/pza/linux#reset/fixes +mips-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mips/linux.git#mips-fixes +at91-fixes https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux.git#at91-fixes +omap-fixes https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git#fixes +tegra-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux.git#fixes +kvm-fixes git://git.kernel.org/pub/scm/virt/kvm/kvm.git#master +kvms390-fixes https://git.kernel.org/pub/scm/linux/kernel/git/kvms390/linux.git#master +kvm-arm-fixes https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git#fixes +hwmon-fixes https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git#hwmon +nvdimm-fixes https://git.kernel.org/pub/scm/linux/kernel/git/nvdimm/nvdimm.git#libnvdimm-fixes +cxl-fixes https://git.kernel.org/pub/scm/linux/kernel/git/cxl/cxl.git#fixes +dma-mapping-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mszyprowski/linux.git#dma-mapping-fixes +drivers-x86-fixes https://git.kernel.org/pub/scm/linux/kernel/git/pdx86/platform-drivers-x86.git#fixes +samsung-krzk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux.git#fixes +pinctrl-samsung-fixes https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/samsung.git#fixes +pinctrl-qcom-fixes https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#pinctrl-qcom/for-current +devicetree-fixes https://git.kernel.org/pub/scm/linux/kernel/git/robh/linux.git#dt/linus +dt-krzk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt.git#fixes +scsi-fixes https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git#fixes +drm-fixes https://gitlab.freedesktop.org/drm/kernel.git#drm-fixes +drm-intel-fixes https://gitlab.freedesktop.org/drm/i915/kernel.git#for-linux-next-fixes +mmc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/mmc.git#fixes +rtc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux.git#rtc-fixes +gnss-fixes https://git.kernel.org/pub/scm/linux/kernel/git/johan/gnss.git#gnss-linus +hyperv-fixes https://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux.git#hyperv-fixes +risc-v-fixes https://git.kernel.org/pub/scm/linux/kernel/git/riscv/linux.git#fixes +riscv-dt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git#riscv-dt-fixes +riscv-soc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git#riscv-soc-fixes +fpga-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fpga/linux-fpga.git#fixes +spdx https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/spdx.git#spdx-linus +gpio-brgl-fixes https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#gpio/for-current +gpio-intel-fixes https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-gpio-intel.git#fixes +pinctrl-intel-fixes https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/intel.git#fixes +auxdisplay-fixes https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-auxdisplay.git#fixes +kunit-fixes https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#kunit-fixes +renesas-fixes https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel.git#fixes +perf-current https://git.kernel.org/pub/scm/linux/kernel/git/perf/perf-tools.git#perf-tools +efi-fixes https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git#urgent +battery-fixes https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-power-supply.git#fixes +iommufd-fixes https://git.kernel.org/pub/scm/linux/kernel/git/jgg/iommufd.git#for-rc +rust-fixes https://github.com/Rust-for-Linux/linux.git#rust-fixes +w1-fixes https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-w1.git#fixes +pmdomain-fixes https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/linux-pm.git#fixes +i2c-andi-fixes https://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux.git#i2c/i2c-fixes +i2c-rust-fixes https://github.com/ikrtn/rust-for-linux#rust-i2c-fixes +sparc-fixes https://git.kernel.org/pub/scm/linux/kernel/git/alarsson/linux-sparc.git#for-linus +clk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/clk/linux.git#clk-fixes +thead-clk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#thead-clk-fixes +tenstorrent-clk-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git#tenstorrent-clk-fixes +fustini-config-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#riscv-config-fixes +pwrseq-fixes https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#pwrseq/for-current +thead-dt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#thead-dt-fixes +ftrace-fixes https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#ftrace/fixes +ring-buffer-fixes https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#ring-buffer/fixes +trace-fixes https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#trace/fixes +tracefs-fixes https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#tracefs/fixes +spacemit-fixes https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux#fixes +tip-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git#tip/urgent +kexec-fixes https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git#kexec-fixes +liveupdate-fixes https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git#fixes +drm-msm-fixes https://gitlab.freedesktop.org/drm/msm.git#msm-fixes +uml-fixes https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git#fixes +fwctl-fixes https://git.kernel.org/pub/scm/linux/kernel/git/fwctl/fwctl.git#for-rc +devsec-tsm-fixes https://git.kernel.org/pub/scm/linux/kernel/git/devsec/tsm.git#fixes +drm-rust-fixes https://gitlab.freedesktop.org/drm/rust/kernel.git#for-linux-next-fixes +tenstorrent-dt-fixes https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git#tenstorrent-dt-fixes +nfc-fixes https://codeberg.org/linux-nfc/linux.git#for-linus +mm-nonmm-hotfixes-stable https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm#mm-nonmm-hotfixes-stable +mm-nonmm-hotfixes-unstable https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm#mm-nonmm-hotfixes-unstable +drm-misc-fixes https://gitlab.freedesktop.org/drm/misc/kernel.git#for-linux-next-fixes +rust https://github.com/Rust-for-Linux/linux.git#rust-next +rust-interop https://github.com/Rust-for-Linux/linux.git#interop-next +rust-alloc https://github.com/Rust-for-Linux/linux.git#alloc-next +rust-io https://github.com/Rust-for-Linux/linux.git#io-next +rust-pin-init https://github.com/Rust-for-Linux/linux.git#pin-init-next +rust-timekeeping https://github.com/Rust-for-Linux/linux.git#timekeeping-next +rust-xarray https://github.com/Rust-for-Linux/linux.git#xarray-next +rust-analyzer https://github.com/Rust-for-Linux/linux.git#rust-analyzer-next +mm https://git.kernel.org/pub/scm/linux/kernel/git/mm/linux.git#for-next +mm-nonmm-stable https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm#mm-nonmm-stable +mm-nonmm-unstable https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm#mm-nonmm-unstable +kbuild https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git#kbuild-for-next +clang-fixes https://git.kernel.org/pub/scm/linux/kernel/git/nathan/linux.git#clang-fixes-for-next +clang-format https://github.com/ojeda/linux.git#clang-format +perf https://git.kernel.org/pub/scm/linux/kernel/git/perf/perf-tools-next.git#perf-tools-next +compiler-attributes https://github.com/ojeda/linux.git#compiler-attributes +dma-mapping https://git.kernel.org/pub/scm/linux/kernel/git/mszyprowski/linux.git#dma-mapping-for-next +asm-generic https://git.kernel.org/pub/scm/linux/kernel/git/arnd/asm-generic#master +alpha https://git.kernel.org/pub/scm/linux/kernel/git/mattst88/alpha.git#alpha-next +arm https://git.kernel.org/pub/scm/linux/kernel/git/rmk/linux.git#for-next +arm64 https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux#for-next/core +arm-perf https://git.kernel.org/pub/scm/linux/kernel/git/will/linux.git#for-next/perf +arm-soc https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc.git#for-next +amlogic https://git.kernel.org/pub/scm/linux/kernel/git/amlogic/linux.git#for-next +asahi-soc https://github.com/AsahiLinux/linux.git#asahi-soc/for-next +at91 https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux.git#at91-next +bmc https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux.git#for-next +broadcom https://github.com/Broadcom/stblinux.git#next +cix https://git.kernel.org/pub/scm/linux/kernel/git/peter.chen/cix.git#for-next +davinci https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#davinci/for-next +drivers-memory https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl.git#for-next +fsl https://git.kernel.org/pub/scm/linux/kernel/git/chleroy/linux.git#soc_fsl +imx-mxs https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux.git#for-next +mediatek https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux.git#for-next +mvebu https://git.kernel.org/pub/scm/linux/kernel/git/gclement/mvebu.git#for-next +omap https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git#for-next +qcom https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux.git#for-next +realtek https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux.git#for-next +renesas https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel.git#next +reset https://git.kernel.org/pub/scm/linux/kernel/git/pza/linux#reset/next +rockchip https://git.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip.git#for-next +samsung-krzk https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux.git#for-next +scmi https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux.git#for-linux-next +sophgo https://github.com/sophgo/linux.git#for-next +sophgo-soc https://github.com/sophgo/linux.git#soc-for-next +spacemit https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux#for-next +stm32 https://git.kernel.org/pub/scm/linux/kernel/git/atorgue/stm32.git#stm32-next +sunxi https://git.kernel.org/pub/scm/linux/kernel/git/sunxi/linux.git#sunxi/for-next +tee https://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee.git#next +tegra https://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux.git#for-next +tenstorrent-dt https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git#tenstorrent-dt-for-next +fustini-config https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#riscv-config-for-next +thead-dt https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#thead-dt-for-next +ti https://git.kernel.org/pub/scm/linux/kernel/git/ti/linux.git#ti-next +xilinx https://github.com/Xilinx/linux-xlnx.git#for-next +socfpga https://git.kernel.org/pub/scm/linux/kernel/git/dinguyen/linux.git#for-next +clk https://git.kernel.org/pub/scm/linux/kernel/git/clk/linux.git#clk-next +clk-imx https://git.kernel.org/pub/scm/linux/kernel/git/abelvesa/linux.git#for-next +clk-renesas https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-drivers.git#renesas-clk +thead-clk https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git#thead-clk-for-next +tenstorrent-clk https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git#tenstorrent-clk-for-next +csky https://github.com/c-sky/csky-linux.git#linux-next +loongarch https://git.kernel.org/pub/scm/linux/kernel/git/chenhuacai/linux-loongson.git#loongarch-next +m68k https://git.kernel.org/pub/scm/linux/kernel/git/geert/linux-m68k.git#for-next +m68knommu https://git.kernel.org/pub/scm/linux/kernel/git/gerg/m68knommu.git#for-next +microblaze git://git.monstr.eu/linux-2.6-microblaze.git#next +mips https://git.kernel.org/pub/scm/linux/kernel/git/mips/linux.git#mips-next +openrisc https://github.com/openrisc/linux.git#for-next +parisc-hd https://git.kernel.org/pub/scm/linux/kernel/git/deller/parisc-linux.git#for-next +powerpc https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git#next +risc-v https://git.kernel.org/pub/scm/linux/kernel/git/riscv/linux.git#for-next +riscv-dt https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git#riscv-dt-for-next +riscv-soc https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git#riscv-soc-for-next +s390 https://git.kernel.org/pub/scm/linux/kernel/git/s390/linux.git#for-next +sh https://git.kernel.org/pub/scm/linux/kernel/git/glaubitz/sh-linux.git#for-next +sparc https://git.kernel.org/pub/scm/linux/kernel/git/alarsson/linux-sparc.git#for-next +uml https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git#next +xtensa https://github.com/jcmvbkbc/linux-xtensa.git#xtensa-for-next +printk https://git.kernel.org/pub/scm/linux/kernel/git/printk/linux.git#for-next +pci https://git.kernel.org/pub/scm/linux/kernel/git/pci/pci.git#next +pstore https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git#for-next/pstore +hid https://git.kernel.org/pub/scm/linux/kernel/git/hid/hid.git#for-next +i2c https://git.kernel.org/pub/scm/linux/kernel/git/wsa/linux.git#i2c/for-next +i2c-andi https://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux.git#i2c/i2c-next +i2c-rust https://github.com/ikrtn/rust-for-linux#rust-i2c-next +i3c https://git.kernel.org/pub/scm/linux/kernel/git/i3c/linux.git#i3c/next +dmi https://git.kernel.org/pub/scm/linux/kernel/git/jdelvare/staging.git#dmi-for-next +hwmon-staging https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git#hwmon-next +jc_docs git://git.lwn.net/linux.git#docs-next +v4l-dvb git://linuxtv.org/media-ci/media-pending.git#next +v4l-dvb-next git://linuxtv.org/mchehab/media-next.git#master +pm https://git.kernel.org/pub/scm/linux/kernel/git/rafael/linux-pm.git#linux-next +cpufreq-arm https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git#cpufreq/arm/linux-next +cpupower https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux.git#cpupower +devfreq https://git.kernel.org/pub/scm/linux/kernel/git/chanwoo/linux.git#devfreq-next +pmdomain https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/linux-pm.git#next +opp https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git#opp/linux-next +thermal https://git.kernel.org/pub/scm/linux/kernel/git/thermal/linux.git#thermal/linux-next +rdma https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git#for-next +net-next https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net-next.git#main +bpf-next https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf-next.git#for-next +ipsec-next https://git.kernel.org/pub/scm/linux/kernel/git/klassert/ipsec-next.git#master +mlx5-next https://git.kernel.org/pub/scm/linux/kernel/git/mellanox/linux.git#mlx5-next +netfilter-next https://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf-next.git#main +ipvs-next https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs-next.git#main +bluetooth https://git.kernel.org/pub/scm/linux/kernel/git/bluetooth/bluetooth-next.git#master +wireless-next https://git.kernel.org/pub/scm/linux/kernel/git/wireless/wireless-next.git#for-next +ath-next https://git.kernel.org/pub/scm/linux/kernel/git/ath/ath.git#for-next +iwlwifi-next https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git#next +wpan-next https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan-next.git#master +wpan-staging https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan-next.git#staging +mtd https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#mtd/next +nand https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#nand/next +spi-nor https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#spi-nor/next +crypto https://git.kernel.org/pub/scm/linux/kernel/git/herbert/cryptodev-2.6.git#master +libcrypto https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git#libcrypto-next +drm https://gitlab.freedesktop.org/drm/kernel.git#drm-next +drm-exynos https://git.kernel.org/pub/scm/linux/kernel/git/daeinki/drm-exynos.git#for-linux-next +drm-misc https://gitlab.freedesktop.org/drm/misc/kernel.git#for-linux-next +amdgpu https://gitlab.freedesktop.org/agd5f/linux.git#drm-next +drm-intel https://gitlab.freedesktop.org/drm/i915/kernel.git#for-linux-next +drm-msm https://gitlab.freedesktop.org/drm/msm.git#msm-next +drm-msm-lumag https://gitlab.freedesktop.org/lumag/msm.git#msm-next-lumag +drm-xe https://gitlab.freedesktop.org/drm/xe/kernel.git#drm-xe-next +drm-rust https://gitlab.freedesktop.org/drm/rust/kernel.git#for-linux-next +drm-nova https://gitlab.freedesktop.org/drm/nova.git#nova-next +etnaviv https://git.pengutronix.de/git/lst/linux#etnaviv/next +fbdev https://git.kernel.org/pub/scm/linux/kernel/git/deller/linux-fbdev.git#for-next +regmap https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regmap.git#for-next +sound https://git.kernel.org/pub/scm/linux/kernel/git/tiwai/sound.git#for-next +ieee1394 https://git.kernel.org/pub/scm/linux/kernel/git/ieee1394/linux1394.git#for-next +sound-asoc https://git.kernel.org/pub/scm/linux/kernel/git/broonie/sound.git#for-next +modules https://git.kernel.org/pub/scm/linux/kernel/git/modules/linux.git#modules-next +input https://git.kernel.org/pub/scm/linux/kernel/git/dtor/input.git#next +block https://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux.git#for-next +device-mapper https://git.kernel.org/pub/scm/linux/kernel/git/device-mapper/linux-dm.git#for-next +libata https://git.kernel.org/pub/scm/linux/kernel/git/libata/linux#for-next +pcmcia https://git.kernel.org/pub/scm/linux/kernel/git/brodo/linux.git#pcmcia-next +mmc https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/mmc.git#next +mfd https://git.kernel.org/pub/scm/linux/kernel/git/lee/mfd.git#for-mfd-next +backlight https://git.kernel.org/pub/scm/linux/kernel/git/lee/backlight.git#for-backlight-next +battery https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-power-supply.git#for-next +regulator https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regulator.git#for-next +security https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git#next +apparmor https://git.kernel.org/pub/scm/linux/kernel/git/jj/linux-apparmor#apparmor-next +integrity https://git.kernel.org/pub/scm/linux/kernel/git/zohar/linux-integrity#next-integrity +selinux https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/selinux.git#next +smack https://github.com/cschaufler/smack-next#next +tomoyo git://git.code.sf.net/p/tomoyo/tomoyo.git#master +tpmdd-tpm https://git.kernel.org/pub/scm/linux/kernel/git/jarkko/linux-tpmdd.git#for-next-tpm +tpmdd-keys https://git.kernel.org/pub/scm/linux/kernel/git/jarkko/linux-tpmdd.git#for-next-keys +watchdog https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git#watchdog-next +iommu https://git.kernel.org/pub/scm/linux/kernel/git/iommu/linux.git#next +audit https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/audit.git#next +devicetree https://git.kernel.org/pub/scm/linux/kernel/git/robh/linux.git#for-next +dt-krzk https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt.git#for-next +mailbox https://git.kernel.org/pub/scm/linux/kernel/git/jassibrar/mailbox.git#for-next +spi https://git.kernel.org/pub/scm/linux/kernel/git/broonie/spi.git#for-next +tip https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git#master +kexec https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git#kexec-next +liveupdate https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git#next +clockevents https://git.kernel.org/pub/scm/linux/kernel/git/daniel.lezcano/linux.git#timers/drivers/next +edac https://git.kernel.org/pub/scm/linux/kernel/git/ras/ras.git#edac-for-next +ftrace https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git#for-next +rcu https://git.kernel.org/pub/scm/linux/kernel/git/rcu/linux#next +paulmck https://git.kernel.org/pub/scm/linux/kernel/git/paulmck/linux-rcu.git#non-rcu/next +kvm git://git.kernel.org/pub/scm/virt/kvm/kvm.git#next +kvm-arm https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git#next +kvms390 https://git.kernel.org/pub/scm/linux/kernel/git/kvms390/linux.git#next +kvm-ppc https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git#topic/ppc-kvm +kvm-riscv https://github.com/kvm-riscv/linux.git#riscv_kvm_next +kvm-x86 https://github.com/kvm-x86/linux.git#next +xen-tip https://git.kernel.org/pub/scm/linux/kernel/git/xen/tip.git#linux-next +percpu https://git.kernel.org/pub/scm/linux/kernel/git/dennis/percpu.git#for-next +workqueues https://git.kernel.org/pub/scm/linux/kernel/git/tj/wq.git#for-next +sched-ext https://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext.git#for-next +drivers-x86 https://git.kernel.org/pub/scm/linux/kernel/git/pdx86/platform-drivers-x86.git#for-next +chrome-platform https://git.kernel.org/pub/scm/linux/kernel/git/chrome-platform/linux.git#for-next +chrome-platform-firmware https://git.kernel.org/pub/scm/linux/kernel/git/chrome-platform/linux.git#for-firmware-next +hsi https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-hsi.git#for-next +leds-lj https://git.kernel.org/pub/scm/linux/kernel/git/lee/leds.git#for-leds-next +ipmi https://github.com/cminyard/linux-ipmi.git#for-next +driver-core https://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git#driver-core-next +usb https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/usb.git#usb-next +thunderbolt https://git.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt.git#next +usb-serial https://git.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial.git#usb-next +tty https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git#tty-next +char-misc https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git#char-misc-next +coresight https://git.kernel.org/pub/scm/linux/kernel/git/coresight/linux.git#next +fastrpc https://git.kernel.org/pub/scm/linux/kernel/git/srini/fastrpc.git#for-next +fpga https://git.kernel.org/pub/scm/linux/kernel/git/fpga/linux-fpga.git#for-next +icc https://git.kernel.org/pub/scm/linux/kernel/git/djakov/icc.git#icc-next +iio https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git#togreg +nfc https://codeberg.org/linux-nfc/linux.git#for-next +phy-next https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git#next +soundwire https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/soundwire.git#next +extcon https://git.kernel.org/pub/scm/linux/kernel/git/chanwoo/extcon.git#extcon-next +gnss https://git.kernel.org/pub/scm/linux/kernel/git/johan/gnss.git#gnss-next +vfio https://github.com/awilliam/linux-vfio.git#next +w1 https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-w1.git#for-next +spmi https://git.kernel.org/pub/scm/linux/kernel/git/sboyd/spmi.git#spmi-next +staging https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/staging.git#staging-next +counter-next https://git.kernel.org/pub/scm/linux/kernel/git/wbg/counter.git#counter-next +mux https://gitlab.com/peda-linux/mux.git#for-next +dmaengine https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git#next +cgroup https://git.kernel.org/pub/scm/linux/kernel/git/tj/cgroup.git#for-next +scsi https://git.kernel.org/pub/scm/linux/kernel/git/jejb/scsi.git#for-next +scsi-mkp https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git#for-next +vhost https://git.kernel.org/pub/scm/linux/kernel/git/mst/vhost.git#linux-next +rpmsg https://git.kernel.org/pub/scm/linux/kernel/git/remoteproc/linux.git#for-next +gpio-brgl https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#gpio/for-next +gpio-intel https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-gpio-intel.git#for-next +pinctrl https://git.kernel.org/pub/scm/linux/kernel/git/linusw/linux-pinctrl.git#for-next +pinctrl-intel https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/intel.git#for-next +pinctrl-renesas https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-drivers.git#renesas-pinctrl +pinctrl-samsung https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/samsung.git#for-next +pinctrl-qcom https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#pinctrl-qcom/for-next +pwm https://git.kernel.org/pub/scm/linux/kernel/git/ukleinek/linux.git#pwm/for-next +ktest https://git.kernel.org/pub/scm/linux/kernel/git/rostedt/linux-ktest.git#for-next +kselftest https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#next +kunit https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#test +kunit-next https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git#kunit +livepatching https://git.kernel.org/pub/scm/linux/kernel/git/livepatching/livepatching.git#for-next +rtc https://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux.git#rtc-next +nvdimm https://git.kernel.org/pub/scm/linux/kernel/git/nvdimm/nvdimm.git#libnvdimm-for-next +at24 https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#at24/for-next +ntb https://github.com/jonmason/ntb.git#ntb-next +seccomp https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git#for-next/seccomp +slimbus https://git.kernel.org/pub/scm/linux/kernel/git/srini/slimbus.git#for-next +nvmem https://git.kernel.org/pub/scm/linux/kernel/git/srini/nvmem.git#for-next +hyperv https://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux.git#hyperv-next +auxdisplay https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-auxdisplay.git#for-next +kgdb https://git.kernel.org/pub/scm/linux/kernel/git/danielt/linux.git#kgdb/for-next +hmm https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git#hmm +cfi https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git#cfi/next +mhi https://git.kernel.org/pub/scm/linux/kernel/git/mani/mhi.git#mhi-next +cxl https://git.kernel.org/pub/scm/linux/kernel/git/cxl/cxl.git#next +zstd https://github.com/terrelln/linux.git#zstd-next +efi https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git#next +unicode https://git.kernel.org/pub/scm/linux/kernel/git/krisman/unicode.git#for-next +random https://git.kernel.org/pub/scm/linux/kernel/git/crng/random.git#master +landlock https://git.kernel.org/pub/scm/linux/kernel/git/mic/linux.git#next +sysctl https://git.kernel.org/pub/scm/linux/kernel/git/sysctl/sysctl.git#sysctl-next +execve https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git#for-next/execve +bitmap https://github.com/norov/linux.git#bitmap-for-next +hte https://git.kernel.org/pub/scm/linux/kernel/git/pateldipen1984/linux.git#for-next +kspp https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git#for-next/kspp +nolibc https://git.kernel.org/pub/scm/linux/kernel/git/nolibc/linux-nolibc.git#for-next +iommufd https://git.kernel.org/pub/scm/linux/kernel/git/jgg/iommufd.git#for-next +turbostat https://git.kernel.org/pub/scm/linux/kernel/git/lenb/linux.git#next +pwrseq https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git#pwrseq/for-next +capabilities-next https://git.kernel.org/pub/scm/linux/kernel/git/sergeh/linux.git#caps-next +ipe https://git.kernel.org/pub/scm/linux/kernel/git/wufan/ipe.git#next +kcsan https://git.kernel.org/pub/scm/linux/kernel/git/melver/linux.git#next +crc https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git#crc-next +keys-next https://git.kernel.org/pub/scm/linux/kernel/git/dhowells/linux-fs.git#keys-next +fwctl https://git.kernel.org/pub/scm/linux/kernel/git/fwctl/fwctl.git#for-next +devsec-tsm https://git.kernel.org/pub/scm/linux/kernel/git/devsec/tsm.git#next +hisilicon https://github.com/hisilicon/linux-hisi.git#for-next +device-id https://git.kernel.org/pub/scm/linux/kernel/git/ukleinek/linux.git#device-id-rework +kthread https://git.kernel.org/pub/scm/linux/kernel/git/frederic/linux-dynticks.git#for-next +pagemap-headers git://git.infradead.org/users/willy/pagecache.git#headers diff --git a/Next/merge.log b/Next/merge.log new file mode 100644 index 00000000000000..8ca151c441a18b --- /dev/null +++ b/Next/merge.log @@ -0,0 +1,17408 @@ +$ date -R +Wed, 30 Sep 2026 11:42:57 +0100 +$ git checkout master +Already on 'master' +$ git reset --hard stable +Updating files: 41% (4753/11360) Updating files: 42% (4772/11360) Updating files: 43% (4885/11360) Updating files: 44% (4999/11360) Updating files: 45% (5112/11360) Updating files: 46% (5226/11360) Updating files: 47% (5340/11360) Updating files: 48% (5453/11360) Updating files: 49% (5567/11360) Updating files: 50% (5680/11360) Updating files: 51% (5794/11360) Updating files: 52% (5908/11360) Updating files: 53% (6021/11360) Updating files: 54% (6135/11360) Updating files: 55% (6248/11360) Updating files: 56% (6362/11360) Updating files: 57% (6476/11360) Updating files: 58% (6589/11360) Updating files: 59% (6703/11360) Updating files: 60% (6816/11360) Updating files: 61% (6930/11360) Updating files: 62% (7044/11360) Updating files: 63% (7157/11360) Updating files: 64% (7271/11360) Updating files: 65% (7384/11360) Updating files: 66% (7498/11360) Updating files: 67% (7612/11360) Updating files: 68% (7725/11360) Updating files: 69% (7839/11360) Updating files: 70% (7952/11360) Updating files: 71% (8066/11360) Updating files: 72% (8180/11360) Updating files: 73% (8293/11360) Updating files: 73% (8394/11360) Updating files: 74% (8407/11360) Updating files: 75% (8520/11360) Updating files: 76% (8634/11360) Updating files: 77% (8748/11360) Updating files: 78% (8861/11360) Updating files: 79% (8975/11360) Updating files: 80% (9088/11360) Updating files: 81% (9202/11360) Updating files: 82% (9316/11360) Updating files: 83% (9429/11360) Updating files: 84% (9543/11360) Updating files: 85% (9656/11360) Updating files: 86% (9770/11360) Updating files: 87% (9884/11360) Updating files: 88% (9997/11360) Updating files: 89% (10111/11360) Updating files: 90% (10224/11360) Updating files: 91% (10338/11360) Updating files: 92% (10452/11360) Updating files: 93% (10565/11360) Updating files: 94% (10679/11360) Updating files: 95% (10792/11360) Updating files: 96% (10906/11360) Updating files: 97% (11020/11360) Updating files: 98% (11133/11360) Updating files: 99% (11247/11360) Updating files: 100% (11360/11360) Updating files: 100% (11360/11360), done. +HEAD is now at 72d3fcf802c45 Linux 7.3-rc5 +Merging origin/master (551c722f40809 Merge tag 'rtc-7.3-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git origin/master +Updating 72d3fcf802c45..551c722f40809 +Fast-forward (no commit created; -m option ignored) + MAINTAINERS | 69 ++++++++++++--------- + arch/arc/Kconfig | 4 -- + arch/arc/include/asm/atomic64-arcv2.h | 6 ++ + arch/arc/include/asm/cmpxchg.h | 4 +- + arch/arc/kernel/setup.c | 1 + + arch/arc/kernel/smp.c | 1 - + drivers/mtd/chips/cfi_cmdset_0001.c | 42 ++++++++----- + drivers/mtd/devices/block2mtd.c | 2 +- + drivers/mtd/devices/mtd_intel_dg.c | 3 +- + drivers/mtd/mtdcore.c | 3 +- + drivers/mtd/nand/ecc.c | 7 +++ + drivers/mtd/nand/raw/cadence-nand-controller.c | 6 +- + drivers/mtd/nand/raw/vf610_nfc.c | 85 +++++++++++++++++++++----- + drivers/mtd/nand/spi/core.c | 53 +++++++++++----- + drivers/mtd/spi-nor/core.c | 2 +- + drivers/rtc/dev.c | 2 +- + drivers/rtc/rtc-ac100.c | 39 +++--------- + drivers/rtc/rtc-efi.c | 80 +++++++++++++++++++++++- + drivers/rtc/rtc-mpfs.c | 6 +- + drivers/rtc/rtc-spear.c | 11 ++-- + include/linux/mtd/nand.h | 2 + + include/uapi/linux/alloc_tag.h | 1 + + kernel/module/main.c | 8 ++- + mm/damon/core.c | 2 +- + mm/kasan/common.c | 13 +++- + mm/shmem.c | 7 ++- + mm/vma.c | 4 +- + 27 files changed, 327 insertions(+), 136 deletions(-) +Merging ext4-fixes/fixes (981fcc5674e67 jbd2: fix deadlock in jbd2_journal_cancel_revoke()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/tytso/ext4.git ext4-fixes/fixes +Already up to date. +Merging vfs-brauner-fixes/vfs.fixes (b78b728e21c32 netfs: Fix missing alloc tagging of direct mempool allocations) +$ git merge -m Merge branch 'vfs.fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git vfs-brauner-fixes/vfs.fixes +Already up to date. +Merging fscrypt-current/for-current (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-current' of https://git.kernel.org/pub/scm/fs/fscrypt/linux.git fscrypt-current/for-current +Already up to date. +Merging fsverity-current/for-current (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-current' of https://git.kernel.org/pub/scm/fs/fsverity/linux.git fsverity-current/for-current +Already up to date. +Merging btrfs-fixes/next-fixes (50a31f73ef35e Merge branch 'misc-7.3' into next-fixes) +$ git merge -m Merge branch 'next-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git btrfs-fixes/next-fixes +Merge made by the 'ort' strategy. +Merging vfs-fixes/fixes (49c5d168a3a8f udf: fix nls leak on udf_fill_super() failure) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git vfs-fixes/fixes +Auto-merging fs/udf/super.c +Merge made by the 'ort' strategy. +Merging erofs-fixes/fixes (135d84c66f854 erofs: add missing buf->off in erofs_bread()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git erofs-fixes/fixes +Already up to date. +Merging nfsd-fixes/nfsd-fixes (f76017a7663c4 nfsd: fix handling of NFSEXP_PNFS in the netlink codepath) +$ git merge -m Merge branch 'nfsd-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/cel/linux nfsd-fixes/nfsd-fixes +Already up to date. +Merging v9fs-fixes/fixes/next (028ef9c96e961 Linux 7.0) +$ git merge -m Merge branch 'fixes/next' of https://git.kernel.org/pub/scm/linux/kernel/git/ericvh/v9fs.git v9fs-fixes/fixes/next +Already up to date. +Merging overlayfs-fixes/ovl-fixes (4549871118cf6 Linux 7.1-rc7) +$ git merge -m Merge branch 'ovl-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/overlayfs/vfs.git overlayfs-fixes/ovl-fixes +Already up to date. +Merging fscrypt/for-next (a63883d2c3d47 MAINTAINERS: List the fscrypt branches) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/fs/fscrypt/linux.git fscrypt/for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 3 ++- + fs/crypto/Kconfig | 2 +- + fs/crypto/crypto.c | 5 +++++ + fs/crypto/fscrypt_private.h | 27 --------------------------- + fs/crypto/hooks.c | 9 +++++++++ + fs/crypto/keyring.c | 13 +++++++++++++ + fs/crypto/keysetup_v1.c | 5 ++--- + 7 files changed, 32 insertions(+), 32 deletions(-) +Merging btrfs/for-next (55d05b596923d Merge branch 'for-next-next-v7.3-20260917' into for-next-20260917) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/kdave/linux.git btrfs/for-next +Merge made by the 'ort' strategy. + fs/btrfs/Kconfig | 1 + + fs/btrfs/bio.c | 163 ++--- + fs/btrfs/bio.h | 9 +- + fs/btrfs/block-group.c | 498 ++------------ + fs/btrfs/block-group.h | 16 +- + fs/btrfs/btrfs_inode.h | 22 +- + fs/btrfs/compression.c | 61 +- + fs/btrfs/ctree.c | 4 +- + fs/btrfs/delalloc-space.c | 13 +- + fs/btrfs/delayed-inode.c | 115 +++- + fs/btrfs/delayed-inode.h | 22 +- + fs/btrfs/dev-replace.c | 10 +- + fs/btrfs/dir-item.c | 41 +- + fs/btrfs/dir-item.h | 5 +- + fs/btrfs/disk-io.c | 126 +--- + fs/btrfs/extent-io-tree.c | 2 +- + fs/btrfs/extent-tree.c | 10 + + fs/btrfs/extent_io.c | 247 ++++--- + fs/btrfs/extent_map.c | 8 + + fs/btrfs/file-item.c | 46 +- + fs/btrfs/free-space-cache.c | 1344 +------------------------------------- + fs/btrfs/free-space-cache.h | 29 +- + fs/btrfs/fs.c | 2 +- + fs/btrfs/fs.h | 4 +- + fs/btrfs/inode.c | 342 ++++------ + fs/btrfs/ioctl.c | 41 +- + fs/btrfs/ordered-data.c | 27 +- + fs/btrfs/qgroup.c | 126 ++-- + fs/btrfs/qgroup.h | 16 +- + fs/btrfs/raid56.c | 57 +- + fs/btrfs/relocation.c | 2 +- + fs/btrfs/send.c | 2 +- + fs/btrfs/space-info.c | 2 - + fs/btrfs/space-info.h | 4 - + fs/btrfs/super.c | 48 +- + fs/btrfs/sysfs.c | 129 +--- + fs/btrfs/tests/extent-io-tests.c | 129 ++-- + fs/btrfs/transaction.c | 64 +- + fs/btrfs/transaction.h | 31 +- + fs/btrfs/tree-checker.c | 83 ++- + fs/btrfs/tree-log.c | 11 +- + fs/btrfs/verity.c | 31 +- + fs/btrfs/volumes.c | 55 +- + fs/btrfs/volumes.h | 1 + + fs/btrfs/zlib.c | 10 +- + fs/btrfs/zoned.c | 21 +- + fs/btrfs/zstd.c | 73 ++- + include/uapi/linux/btrfs_tree.h | 25 +- + 48 files changed, 1091 insertions(+), 3037 deletions(-) +Merging ceph/master (dc173b37415e8 ceph: apply nearfull_sync option on remount) +$ git merge -m Merge branch 'master' of https://github.com/ceph/ceph-client.git ceph/master +Already up to date. +Merging cifs/cifs-next (f14572c203d57 Merge tag 'cifs-fixes-7.3-rc5' of https://git.manguebit.org/linux) +$ git merge -m Merge branch 'cifs-next' of https://git.manguebit.org/linux.git cifs/cifs-next +Already up to date. +Merging configfs/configfs-next (620938e7da894 samples: configfs: constify the configfs_attribute structures) +$ git merge -m Merge branch 'configfs-next' of https://git.kernel.org/pub/scm/linux/kernel/git/leitao/linux.git configfs/configfs-next +Auto-merging MAINTAINERS +Auto-merging tools/testing/selftests/Makefile +Merge made by the 'ort' strategy. + MAINTAINERS | 1 + + drivers/acpi/acpi_configfs.c | 2 +- + drivers/gpu/drm/xe/xe_configfs.c | 4 +- + drivers/virt/coco/guest/report.c | 6 +- + fs/configfs/configfs_internal.h | 10 +- + fs/configfs/dir.c | 10 +- + fs/configfs/file.c | 6 +- + fs/configfs/inode.c | 2 +- + include/linux/configfs.h | 73 ++-- + rust/kernel/configfs.rs | 20 +- + samples/configfs/configfs_sample.c | 137 +++++- + tools/testing/selftests/Makefile | 1 + + .../selftests/filesystems/configfs/.gitignore | 2 + + .../selftests/filesystems/configfs/Makefile | 8 + + .../testing/selftests/filesystems/configfs/config | 5 + + .../selftests/filesystems/configfs/configfs_test.c | 481 +++++++++++++++++++++ + 16 files changed, 697 insertions(+), 71 deletions(-) + create mode 100644 tools/testing/selftests/filesystems/configfs/.gitignore + create mode 100644 tools/testing/selftests/filesystems/configfs/Makefile + create mode 100644 tools/testing/selftests/filesystems/configfs/config + create mode 100644 tools/testing/selftests/filesystems/configfs/configfs_test.c +Merging configfs-rust/configfs-next (6dc9fb1516aa4 rust: configfs: fix data offset calculation for subsystem callbacks) +$ git merge -m Merge branch 'configfs-next' of https://git.kernel.org/pub/scm/linux/kernel/git/a.hindborg/linux.git configfs-rust/configfs-next +Auto-merging rust/kernel/configfs.rs +Merge made by the 'ort' strategy. + rust/kernel/configfs.rs | 107 ++++++++++++++++++++++++++++++++---------------- + 1 file changed, 71 insertions(+), 36 deletions(-) +Merging ecryptfs/next (f81cb44f9a4b8 ecryptfs: ecryptfs_kernel.h: clean up kernel-doc comments) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/tyhicks/ecryptfs.git ecryptfs/next +Already up to date. +Merging dlm/next (ed9b6a1296f10 dlm: wait for outstanding SRCU callbacks to complete in exit paths) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/teigland/linux-dlm.git dlm/next +Merge made by the 'ort' strategy. + fs/dlm/config.c | 30 ++++++++++++++++++++++++++++-- + fs/dlm/dlm_internal.h | 5 +++++ + fs/dlm/lock.c | 18 +++++++++++++----- + fs/dlm/lock.h | 4 ++-- + fs/dlm/lowcomms.c | 1 + + fs/dlm/midcomms.c | 1 + + fs/dlm/plock.c | 14 +++++++++++++- + fs/dlm/user.c | 23 +++++++++++++++++++++++ + 8 files changed, 86 insertions(+), 10 deletions(-) +Merging erofs/dev (a7d28aa0e9b2c erofs: simplify z_erofs_gbuf_growsize()) +$ git merge -m Merge branch 'dev' of https://git.kernel.org/pub/scm/linux/kernel/git/xiang/erofs.git erofs/dev +Already up to date. +Merging exfat/dev (216426aff8f69 MAINTAINERS: add exFAT documentation) +$ git merge -m Merge branch 'dev' of https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/exfat.git exfat/dev +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + Documentation/filesystems/exfat.rst | 117 ++++++++++++++++++++++++++++++++++++ + Documentation/filesystems/index.rst | 1 + + MAINTAINERS | 1 + + fs/exfat/dir.c | 52 ++++++++++++++-- + fs/exfat/exfat_fs.h | 1 + + fs/exfat/fatent.c | 19 +++--- + fs/exfat/iomap.c | 28 ++++++++- + fs/exfat/namei.c | 7 ++- + fs/exfat/super.c | 30 +++++++-- + 9 files changed, 234 insertions(+), 22 deletions(-) + create mode 100644 Documentation/filesystems/exfat.rst +Merging ext3/for_next (ba5855e74bcd7 Pull quota shrinker fix.) +$ git merge -m Merge branch 'for_next' of https://git.kernel.org/pub/scm/linux/kernel/git/jack/linux-fs.git ext3/for_next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 1 + + fs/ext2/Makefile | 2 + + fs/ext2/balloc.c | 4 ++ + fs/ext2/ext2.h | 19 ++++++---- + fs/ext2/file.c | 7 ++++ + fs/ext2/inode.c | 7 ++++ + fs/ext2/super.c | 33 +++++++++------- + fs/isofs/compress.c | 77 +++++++++++++++++++++----------------- + fs/isofs/inode.c | 46 ++++++++--------------- + fs/isofs/isofs.h | 2 + + fs/isofs/rock.c | 11 +++++- + fs/notify/fanotify/fanotify_user.c | 2 +- + fs/ocfs2/quota_local.c | 4 ++ + fs/quota/dquot.c | 16 ++++++++ + fs/udf/inode.c | 59 ++++++++++++++++++++++++++--- + fs/udf/misc.c | 15 ++++---- + fs/udf/super.c | 5 ++- + include/linux/once_lite.h | 8 ++-- + mm/shmem_quota.c | 5 +++ + 19 files changed, 217 insertions(+), 106 deletions(-) +Merging ext4/dev (9091c97be3408 ext4: fix estimate extent index blocks in ext4_ext_index_trans_blocks()) +$ git merge -m Merge branch 'dev' of https://git.kernel.org/pub/scm/linux/kernel/git/tytso/ext4.git ext4/dev +Already up to date. +Merging f2fs/dev (a4b9fb9e69116 f2fs: rename nr_pages_to_skip with nr_caches_to_skip) +$ git merge -m Merge branch 'dev' of https://git.kernel.org/pub/scm/linux/kernel/git/jaegeuk/f2fs.git f2fs/dev +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-fs-f2fs | 7 + + fs/f2fs/Makefile | 2 +- + fs/f2fs/acl.c | 26 +- + fs/f2fs/acl.h | 8 +- + fs/f2fs/cache.c | 720 +++++++++++++++++ + fs/f2fs/cache.h | 240 ++++++ + fs/f2fs/checkpoint.c | 558 +++++++------- + fs/f2fs/compress.c | 183 ++--- + fs/f2fs/data.c | 648 +++++++++++----- + fs/f2fs/debug.c | 94 ++- + fs/f2fs/dir.c | 221 +++--- + fs/f2fs/extent_cache.c | 25 +- + fs/f2fs/f2fs.h | 508 +++++++----- + fs/f2fs/file.c | 181 ++--- + fs/f2fs/gc.c | 222 +++--- + fs/f2fs/inline.c | 293 +++---- + fs/f2fs/inode.c | 242 +++--- + fs/f2fs/iostat.h | 11 + + fs/f2fs/namei.c | 114 +-- + fs/f2fs/node.c | 1277 +++++++++++++++---------------- + fs/f2fs/node.h | 191 +++-- + fs/f2fs/recovery.c | 255 +++--- + fs/f2fs/segment.c | 429 ++++++----- + fs/f2fs/segment.h | 87 ++- + fs/f2fs/shrinker.c | 16 +- + fs/f2fs/super.c | 280 +++---- + fs/f2fs/sysfs.c | 32 +- + fs/f2fs/verity.c | 6 +- + fs/f2fs/xattr.c | 133 ++-- + fs/f2fs/xattr.h | 23 +- + include/linux/f2fs_fs.h | 140 ++-- + include/trace/events/f2fs.h | 71 ++ + 32 files changed, 4345 insertions(+), 2898 deletions(-) + create mode 100644 fs/f2fs/cache.c + create mode 100644 fs/f2fs/cache.h +Merging fsverity/for-next (b08e4ee274917 MAINTAINERS: List the fsverity branches) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/fs/fsverity/linux.git fsverity/for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 3 ++- + 1 file changed, 2 insertions(+), 1 deletion(-) +Merging fuse/for-next (29b60c6c561f7 fuse: make fuse_backing_get() static) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mszeredi/fuse.git fuse/for-next +Auto-merging fs/fuse/file.c +Merge made by the 'ort' strategy. + fs/fuse/Kconfig | 8 +- + fs/fuse/Makefile | 2 +- + fs/fuse/backing.c | 2 +- + fs/fuse/dax.c | 215 +++++++++++---------- + fs/fuse/dev.c | 50 +++-- + fs/fuse/dev.h | 2 +- + fs/fuse/dev_uring.c | 6 + + fs/fuse/dir.c | 4 +- + fs/fuse/file.c | 129 ++++++++----- + fs/fuse/fuse_i.h | 96 +++++---- + fs/fuse/inode.c | 82 +++++--- + fs/fuse/ioctl.c | 3 - + fs/fuse/iomode.c | 4 +- + fs/fuse/notify.c | 4 +- + fs/fuse/passthrough.c | 15 +- + fs/fuse/req.c | 7 +- + fs/fuse/virtio_fs.c | 28 +-- + include/uapi/linux/fuse.h | 12 +- + .../testing/selftests/filesystems/fuse/.gitignore | 1 + + tools/testing/selftests/filesystems/fuse/Makefile | 2 +- + 20 files changed, 380 insertions(+), 292 deletions(-) +Merging gfs2/for-next (d0c4bce31a579 gfs2: gfs2_create_inode excl cleanup) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gfs2/linux-gfs2.git gfs2/for-next +Merge made by the 'ort' strategy. + fs/gfs2/acl.c | 11 +++-- + fs/gfs2/aops.c | 17 +++---- + fs/gfs2/bmap.c | 93 ++++++++++++++++++++---------------- + fs/gfs2/dentry.c | 6 +-- + fs/gfs2/dir.c | 61 +++++++++++++++--------- + fs/gfs2/export.c | 7 +-- + fs/gfs2/file.c | 72 ++++++++++++++++------------ + fs/gfs2/glock.c | 60 +++++++++++++++-------- + fs/gfs2/glops.c | 13 +++-- + fs/gfs2/incore.h | 8 ++-- + fs/gfs2/inode.c | 132 ++++++++++++++++++++++++++++----------------------- + fs/gfs2/lock_dlm.c | 4 +- + fs/gfs2/log.c | 3 +- + fs/gfs2/lops.c | 18 ++++--- + fs/gfs2/meta_io.c | 7 +-- + fs/gfs2/ops_fstype.c | 26 +++++----- + fs/gfs2/quota.c | 57 ++++++++++++---------- + fs/gfs2/recovery.c | 19 ++++---- + fs/gfs2/rgrp.c | 129 +++++++++++++++++++++++++------------------------ + fs/gfs2/super.c | 98 +++++++++++++++++++++++--------------- + fs/gfs2/trace_gfs2.h | 6 +-- + fs/gfs2/util.c | 8 ++-- + fs/gfs2/xattr.c | 69 ++++++++++++++++----------- + 23 files changed, 528 insertions(+), 396 deletions(-) +Merging jfs/jfs-next (dad98c5b2a05e jfs: avoid -Wtautological-constant-out-of-range-compare warning again) +$ git merge -m Merge branch 'jfs-next' of https://github.com/kleikamp/linux-shaggy.git jfs/jfs-next +Already up to date. +Merging ksmbd/ksmbd-for-next (d30f0c30c87d6 ksmbd: fix OOB read and cross-share confusion in ksmbd_validate_name_reconnect()) +$ git merge -m Merge branch 'ksmbd-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/smb.git ksmbd/ksmbd-for-next +Merge made by the 'ort' strategy. + Documentation/filesystems/smb/ksmbd.rst | 14 +- + fs/smb/common/compress/compress.c | 16 +- + fs/smb/common/fscc.h | 5 +- + fs/smb/common/smbglob.h | 1 + + fs/smb/server/Kconfig | 2 + + fs/smb/server/Makefile | 1 + + fs/smb/server/compress.c | 1 + + fs/smb/server/connection.c | 3 +- + fs/smb/server/connection.h | 3 + + fs/smb/server/ksmbd_work.c | 3 - + fs/smb/server/ksmbd_work.h | 5 +- + fs/smb/server/mgmt/user_session.c | 4 +- + fs/smb/server/smb2ops.c | 33 ++ + fs/smb/server/smb2pdu.c | 690 ++++++++++++++------------------ + fs/smb/server/smb2pdu.h | 22 +- + fs/smb/server/smb_common.c | 4 +- + fs/smb/server/smbacl.c | 7 +- + fs/smb/server/tests/Kconfig | 15 + + fs/smb/server/tests/Makefile | 4 + + fs/smb/server/tests/smbacl_kunit.c | 301 ++++++++++++++ + fs/smb/server/transport_tcp.c | 8 +- + fs/smb/server/vfs.c | 12 +- + fs/smb/server/vfs.h | 5 + + fs/smb/server/vfs_cache.c | 66 +-- + fs/smb/server/vfs_cache.h | 6 - + 25 files changed, 738 insertions(+), 493 deletions(-) + create mode 100644 fs/smb/server/tests/Kconfig + create mode 100644 fs/smb/server/tests/Makefile + create mode 100644 fs/smb/server/tests/smbacl_kunit.c +$ git am -3 ../patches/0001-ntfs3-Fix-up-merge-with-Linus.patch +Applying: ntfs3: Fix up merge with Linus +Using index info to reconstruct a base tree... +M fs/ntfs3/file.c +Falling back to patching base and 3-way merge... +Auto-merging fs/ntfs3/file.c +No changes -- Patch already applied. +Merging nfs/linux-next (fd73f4a665989 Linux 7.3-rc3) +$ git merge -m Merge branch 'linux-next' of git://git.linux-nfs.org/projects/trondmy/nfs-2.6.git nfs/linux-next +Already up to date. +Merging nfs-anna/linux-next (9bafc322b7fca nfs: split up block layout and SCSI layout support) +$ git merge -m Merge branch 'linux-next' of git://git.linux-nfs.org/projects/anna/linux-nfs.git nfs-anna/linux-next +Merge made by the 'ort' strategy. + fs/nfs/Kconfig | 17 +- + fs/nfs/blocklayout/Makefile | 3 +- + fs/nfs/blocklayout/blocklayout.c | 119 ++++++--- + fs/nfs/blocklayout/dev.c | 32 ++- + fs/nfs/callback_proc.c | 23 +- + fs/nfs/callback_xdr.c | 9 +- + fs/nfs/client.c | 6 +- + fs/nfs/direct.c | 139 +++++----- + fs/nfs/filelayout/filelayout.c | 17 +- + fs/nfs/filelayout/filelayoutdev.c | 19 +- + fs/nfs/flexfilelayout/flexfilelayout.c | 413 +++++++++++++++++++----------- + fs/nfs/flexfilelayout/flexfilelayout.h | 48 ++-- + fs/nfs/flexfilelayout/flexfilelayoutdev.c | 212 +++++++++------ + fs/nfs/internal.h | 3 +- + fs/nfs/netns.h | 5 +- + fs/nfs/nfs3client.c | 9 +- + fs/nfs/nfs4_fs.h | 3 + + fs/nfs/nfs4client.c | 8 +- + fs/nfs/nfs4proc.c | 136 ++++++++++ + fs/nfs/nfs4state.c | 7 +- + fs/nfs/nfs4xdr.c | 2 +- + fs/nfs/pagelist.c | 66 ++++- + fs/nfs/pnfs.c | 376 +++++++++++++++++++++++++-- + fs/nfs/pnfs.h | 102 +++++++- + fs/nfs/pnfs_dev.c | 30 ++- + fs/nfs/pnfs_nfs.c | 108 ++++++-- + fs/nfs/read.c | 2 +- + fs/nfs/write.c | 2 +- + include/linux/nfs_fs_sb.h | 8 + + include/linux/nfs_page.h | 8 +- + include/linux/nfs_xdr.h | 2 + + net/sunrpc/xprtsock.c | 69 ++--- + 32 files changed, 1513 insertions(+), 490 deletions(-) +Merging nfsd/nfsd-next (ac04dab23b5ff NFSD: Return NFSERR_ISDIR for NFSv2 READ and WRITE on a non-regular file) +$ git merge -m Merge branch 'nfsd-next' of https://git.kernel.org/pub/scm/linux/kernel/git/cel/linux nfsd/nfsd-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + Documentation/admin-guide/nfs/pnfs-scsi-server.rst | 2 +- + MAINTAINERS | 3 +- + fs/lockd/svc.c | 1 - + fs/lockd/svclock.c | 33 +- + fs/lockd/trace.h | 1 - + fs/lockd/xdr.h | 2 +- + fs/namei.c | 2 + + fs/nfs/nfs4file.c | 1 + + fs/nfs/super.c | 25 - + fs/nfs_common/nfs_ssc.c | 126 +++-- + fs/nfsd/blocklayout.c | 1 + + fs/nfsd/blocklayoutxdr.c | 11 + + fs/nfsd/export.c | 8 +- + fs/nfsd/export.h | 3 +- + fs/nfsd/filecache.c | 1 + + fs/nfsd/flexfilelayout.c | 24 +- + fs/nfsd/flexfilelayoutxdr.c | 9 +- + fs/nfsd/flexfilelayoutxdr.h | 8 +- + fs/nfsd/localio.c | 10 +- + fs/nfsd/lockd.c | 5 +- + fs/nfsd/netns.h | 17 +- + fs/nfsd/nfs2acl.c | 48 +- + fs/nfsd/nfs3acl.c | 1 + + fs/nfsd/nfs3proc.c | 79 ++- + fs/nfsd/nfs3xdr.c | 4 + + fs/nfsd/nfs4acl.c | 1 + + fs/nfsd/nfs4callback.c | 8 +- + fs/nfsd/nfs4ctl.h | 83 +++ + fs/nfsd/nfs4idmap.c | 1 + + fs/nfsd/nfs4layouts.c | 1 + + fs/nfsd/nfs4proc.c | 465 +++++++++------- + fs/nfsd/nfs4recover.c | 1 + + fs/nfsd/nfs4state.c | 587 ++++++++++++++++++--- + fs/nfsd/nfs4xdr.c | 188 +++++-- + fs/nfsd/nfscache.c | 1 + + fs/nfsd/nfsctl.c | 36 +- + fs/nfsd/nfsd.h | 237 +-------- + fs/nfsd/nfserr.h | 159 ++++++ + fs/nfsd/nfsfh.c | 73 +-- + fs/nfsd/nfsfh.h | 28 +- + fs/nfsd/nfsproc.c | 39 +- + fs/nfsd/nfssvc.c | 8 + + fs/nfsd/nfsxdr.c | 1 + + fs/nfsd/state.h | 46 +- + fs/nfsd/trace.h | 45 +- + fs/nfsd/vfs.c | 298 +++++------ + fs/nfsd/vfs.h | 39 +- + fs/nfsd/xdr.h | 6 +- + fs/nfsd/xdr3.h | 2 +- + fs/nfsd/xdr4.h | 156 +----- + fs/nfsd/xdr4cb.h | 20 +- + include/linux/nfs.h | 55 +- + include/linux/nfs3.h | 43 ++ + include/linux/nfs4.h | 6 + + include/linux/nfs_fh.h | 63 +++ + include/linux/nfs_ssc.h | 69 +-- + include/linux/nfsd_ssc.h | 38 ++ + include/linux/nfslocalio.h | 11 +- + include/linux/sunrpc/svc_xprt.h | 5 +- + include/trace/misc/nfs.h | 13 +- + net/sunrpc/svc_xprt.c | 36 +- + net/sunrpc/svcauth_unix.c | 9 + + net/sunrpc/svcsock.c | 490 ++++++++++------- + net/sunrpc/xprtrdma/svc_rdma_recvfrom.c | 18 +- + net/sunrpc/xprtrdma/svc_rdma_transport.c | 3 +- + 65 files changed, 2431 insertions(+), 1382 deletions(-) + create mode 100644 fs/nfsd/nfs4ctl.h + create mode 100644 fs/nfsd/nfserr.h + create mode 100644 include/linux/nfs_fh.h + create mode 100644 include/linux/nfsd_ssc.h +$ git am -3 ../patches/0001-Revert-smb-client-implement-fileattr_get-to-support-.patch +Applying: Revert "smb: client: implement fileattr_get to support FS_IOC_GETFLAGS" +Using index info to reconstruct a base tree... +M fs/smb/client/cifsfs.c +M fs/smb/client/cifsfs.h +M fs/smb/client/inode.c +Falling back to patching base and 3-way merge... +Auto-merging fs/smb/client/inode.c +Auto-merging fs/smb/client/cifsfs.h +Auto-merging fs/smb/client/cifsfs.c +No changes -- Patch already applied. +Merging ntfs/ntfs-next (708f9d56cacae ntfs: reject non-resident attributes whose sizes exceed their allocation) +$ git merge -m Merge branch 'ntfs-next' of https://git.kernel.org/pub/scm/linux/kernel/git/linkinjeon/ntfs.git ntfs/ntfs-next +Merge made by the 'ort' strategy. + fs/ntfs/attrib.c | 76 ++++- + fs/ntfs/attrlist.c | 14 +- + fs/ntfs/bitmap.c | 19 +- + fs/ntfs/collate.c | 8 +- + fs/ntfs/dir.c | 23 +- + fs/ntfs/file.c | 56 ++-- + fs/ntfs/index.c | 2 +- + fs/ntfs/index.h | 2 +- + fs/ntfs/inode.c | 131 ++++++-- + fs/ntfs/iomap.c | 16 +- + fs/ntfs/layout.h | 19 +- + fs/ntfs/mft.c | 921 +++++++++++++++++++++++++++++++++++------------------ + fs/ntfs/mft.h | 19 +- + fs/ntfs/mst.c | 4 +- + fs/ntfs/namei.c | 31 +- + fs/ntfs/ntfs.h | 7 +- + fs/ntfs/reparse.c | 2 +- + fs/ntfs/runlist.c | 5 +- + fs/ntfs/super.c | 391 +++++++++++++++++++---- + fs/ntfs/unistr.c | 29 +- + fs/ntfs/volume.h | 15 + + 21 files changed, 1266 insertions(+), 524 deletions(-) +Merging ntfs3/master (f3b8ee6c05bed fs/ntfs3: return -ERANGE for short xattr buffers) +$ git merge -m Merge branch 'master' of https://github.com/Paragon-Software-Group/linux-ntfs3.git ntfs3/master +Auto-merging fs/ntfs3/inode.c +Merge made by the 'ort' strategy. + Documentation/filesystems/ntfs3.rst | 23 ++++++++++++++++++ + fs/ntfs3/attrib.c | 47 ++++++++++++++++++++++++++++++------- + fs/ntfs3/dir.c | 3 +++ + fs/ntfs3/file.c | 3 +++ + fs/ntfs3/frecord.c | 5 ++++ + fs/ntfs3/fslog.c | 4 +++- + fs/ntfs3/fsntfs.c | 3 +++ + fs/ntfs3/index.c | 13 ++++++++-- + fs/ntfs3/inode.c | 19 ++++++++++----- + fs/ntfs3/record.c | 16 ++++++++++++- + fs/ntfs3/super.c | 2 +- + fs/ntfs3/xattr.c | 14 +++++++---- + 12 files changed, 127 insertions(+), 25 deletions(-) +Merging orangefs/for-next (2bc09cb0e9e44 orangefs: don't continue on to gpf if client dies on write.) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/hubcap/linux.git orangefs/for-next +Merge made by the 'ort' strategy. + fs/orangefs/inode.c | 21 ++++++++++++++++++++- + fs/orangefs/orangefs-debugfs.c | 10 ++++++---- + fs/orangefs/super.c | 18 ++++++++++++++++++ + 3 files changed, 44 insertions(+), 5 deletions(-) +Merging overlayfs/overlayfs-next (1f6ee9be92f8d ovl: make fsync after metadata copy-up opt-in mount option) +$ git merge -m Merge branch 'overlayfs-next' of https://git.kernel.org/pub/scm/linux/kernel/git/overlayfs/vfs.git overlayfs/overlayfs-next +Already up to date. +Merging ubifs/next (a5e0055eac837 UBI: support per-device wear-leveling threshold) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/rw/ubifs.git ubifs/next +Already up to date. +Merging v9fs/9p-next (c60ae98c5aa64 9p: Fix v9fs_issue_write() to update i_size and remote_i_size) +$ git merge -m Merge branch '9p-next' of https://github.com/martinetd/linux v9fs/9p-next +Already up to date. +Merging v9fs-ericvh/ericvh/for-next (028ef9c96e961 Linux 7.0) +$ git merge -m Merge branch 'ericvh/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ericvh/v9fs.git v9fs-ericvh/ericvh/for-next +Already up to date. +Merging xfs/for-next (b4787e7b9730d Merge remote-tracking branch 'xfs-linux/xfs-7.3-fixes' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/fs/xfs/xfs-linux.git xfs/for-next +Merge made by the 'ort' strategy. + .../filesystems/xfs/xfs-online-fsck-design.rst | 13 +- + fs/xfs/libxfs/xfs_attr.c | 14 +- + fs/xfs/libxfs/xfs_bmap.c | 5 +- + fs/xfs/libxfs/xfs_bmap.h | 2 +- + fs/xfs/libxfs/xfs_btree.c | 10 +- + fs/xfs/libxfs/xfs_btree_mem.c | 6 +- + fs/xfs/libxfs/xfs_dquot_buf.c | 13 +- + fs/xfs/libxfs/xfs_exchmaps.c | 1 - + fs/xfs/libxfs/xfs_exchmaps.h | 3 +- + fs/xfs/libxfs/xfs_ialloc.c | 2 +- + fs/xfs/libxfs/xfs_ialloc_btree.c | 1 - + fs/xfs/libxfs/xfs_ialloc_btree.h | 5 +- + fs/xfs/libxfs/xfs_inode_fork.c | 4 +- + fs/xfs/libxfs/xfs_metadir.c | 32 +- + fs/xfs/libxfs/xfs_metadir.h | 5 +- + fs/xfs/libxfs/xfs_parent.h | 1 - + fs/xfs/libxfs/xfs_refcount_btree.c | 16 +- + fs/xfs/libxfs/xfs_refcount_btree.h | 26 ++ + fs/xfs/libxfs/xfs_rmap.c | 13 +- + fs/xfs/libxfs/xfs_rmap_btree.c | 16 +- + fs/xfs/libxfs/xfs_rmap_btree.h | 26 ++ + fs/xfs/libxfs/xfs_rtrefcount_btree.c | 121 +----- + fs/xfs/libxfs/xfs_rtrefcount_btree.h | 5 +- + fs/xfs/libxfs/xfs_rtrmap_btree.c | 236 +----------- + fs/xfs/libxfs/xfs_rtrmap_btree.h | 3 +- + fs/xfs/libxfs/xfs_sb.c | 40 -- + fs/xfs/libxfs/xfs_sb.h | 1 - + fs/xfs/libxfs/xfs_symlink_remote.c | 8 +- + fs/xfs/libxfs/xfs_symlink_remote.h | 2 +- + fs/xfs/libxfs/xfs_trans_resv.c | 45 +-- + fs/xfs/libxfs/xfs_trans_space.c | 6 +- + fs/xfs/libxfs/xfs_types.c | 7 +- + fs/xfs/libxfs/xfs_types.h | 7 +- + fs/xfs/scrub/alloc_repair.c | 15 +- + fs/xfs/scrub/bmap.c | 11 +- + fs/xfs/scrub/bmap_repair.c | 4 +- + fs/xfs/scrub/inode_repair.c | 32 +- + fs/xfs/scrub/orphanage.c | 2 +- + fs/xfs/scrub/quota.c | 2 +- + fs/xfs/scrub/quota_repair.c | 9 +- + fs/xfs/scrub/quotacheck_repair.c | 54 ++- + fs/xfs/scrub/rtrefcount_repair.c | 3 +- + fs/xfs/scrub/rtrmap_repair.c | 2 +- + fs/xfs/scrub/xfarray.c | 201 ++++------ + fs/xfs/scrub/xfarray.h | 9 +- + fs/xfs/xfs_bmap_item.c | 2 +- + fs/xfs/xfs_dquot.c | 20 +- + fs/xfs/xfs_dquot.h | 2 +- + fs/xfs/xfs_exchmaps_item.c | 4 +- + fs/xfs/xfs_exchrange.c | 2 +- + fs/xfs/xfs_file.c | 22 +- + fs/xfs/xfs_handle.c | 2 +- + fs/xfs/xfs_inode.c | 20 +- + fs/xfs/xfs_ioctl.c | 421 +++++++++++++-------- + fs/xfs/xfs_ioctl.h | 4 +- + fs/xfs/xfs_ioctl32.c | 188 +++++---- + fs/xfs/xfs_qm.c | 6 +- + fs/xfs/xfs_rmap_item.c | 2 +- + fs/xfs/xfs_rtalloc.h | 49 ++- + fs/xfs/xfs_super.c | 2 +- + fs/xfs/xfs_symlink.c | 4 +- + fs/xfs/xfs_trace.h | 28 +- + fs/xfs/xfs_trans_dquot.c | 6 +- + 63 files changed, 881 insertions(+), 942 deletions(-) +Merging zonefs/for-next (3a8389d42bdf4 zonefs: handle integer overflow in zonefs_fname_to_fno) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/dlemoal/zonefs.git zonefs/for-next +Already up to date. +Merging vfs-brauner/vfs.all (84086827932b5 Merge branch 'vfs-7.4.sync_close' into vfs.all) +$ git merge -m Merge branch 'vfs.all' of https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git vfs-brauner/vfs.all +Auto-merging Documentation/filesystems/index.rst +Auto-merging MAINTAINERS +Auto-merging drivers/gpio/gpiolib-cdev.c +Auto-merging fs/9p/vfs_addr.c +Auto-merging fs/btrfs/btrfs_inode.h +Auto-merging fs/btrfs/inode.c +Auto-merging fs/btrfs/ioctl.c +Auto-merging fs/configfs/configfs_internal.h +Auto-merging fs/configfs/dir.c +Auto-merging fs/configfs/inode.c +Auto-merging fs/dax.c +Auto-merging fs/debugfs/inode.c +Auto-merging fs/exec.c +Auto-merging fs/exfat/exfat_fs.h +Auto-merging fs/exfat/namei.c +Auto-merging fs/ext2/ext2.h +Auto-merging fs/ext2/inode.c +Auto-merging fs/f2fs/acl.c +Auto-merging fs/f2fs/acl.h +Auto-merging fs/f2fs/f2fs.h +Auto-merging fs/f2fs/file.c +Auto-merging fs/f2fs/namei.c +Auto-merging fs/f2fs/xattr.c +Auto-merging fs/fuse/dir.c +Auto-merging fs/fuse/file.c +Auto-merging fs/fuse/fuse_i.h +Auto-merging fs/fuse/ioctl.c +Auto-merging fs/fuse/req.c +Auto-merging fs/gfs2/acl.c +Auto-merging fs/gfs2/file.c +Auto-merging fs/gfs2/inode.c +Auto-merging fs/gfs2/log.c +Auto-merging fs/gfs2/lops.c +Auto-merging fs/gfs2/xattr.c +Auto-merging fs/namei.c +Auto-merging fs/nfs/Kconfig +Auto-merging fs/nfs/internal.h +Auto-merging fs/nfs/nfs4proc.c +Auto-merging fs/nfsd/vfs.c +Auto-merging fs/ntfs/file.c +Auto-merging fs/ntfs/namei.c +Auto-merging fs/ntfs3/file.c +Auto-merging fs/ntfs3/inode.c +Auto-merging fs/ntfs3/xattr.c +Auto-merging fs/ocfs2/namei.c +Auto-merging fs/ocfs2/xattr.c +Auto-merging fs/orangefs/inode.c +Auto-merging fs/quota/dquot.c +Auto-merging fs/smb/client/dir.c +Auto-merging fs/smb/client/transport.c +Auto-merging fs/smb/server/smb2pdu.c +CONFLICT (content): Merge conflict in fs/smb/server/smb2pdu.c +Auto-merging fs/smb/server/smb_common.c +Auto-merging fs/smb/server/smbacl.c +Auto-merging fs/smb/server/vfs.c +CONFLICT (content): Merge conflict in fs/smb/server/vfs.c +Auto-merging fs/smb/server/vfs.h +CONFLICT (content): Merge conflict in fs/smb/server/vfs.h +Auto-merging fs/super.c +Auto-merging fs/xfs/libxfs/xfs_errortag.h +Auto-merging fs/xfs/xfs_file.c +Auto-merging fs/xfs/xfs_handle.c +Auto-merging fs/xfs/xfs_inode.c +Auto-merging fs/xfs/xfs_ioctl.c +Auto-merging fs/xfs/xfs_ioctl.h +Auto-merging fs/xfs/xfs_super.c +Auto-merging fs/xfs/xfs_symlink.c +Auto-merging fs/xfs/xfs_trace.h +Auto-merging fs/xfs/xfs_zone_alloc.c +Auto-merging include/linux/sched.h +Auto-merging kernel/exit.c +Auto-merging kernel/fork.c +Auto-merging kernel/signal.c +Auto-merging mm/shmem.c +Auto-merging security/selinux/hooks.c +Auto-merging tools/testing/selftests/Makefile +Resolved 'fs/smb/server/smb2pdu.c' using previous resolution. +Resolved 'fs/smb/server/vfs.c' using previous resolution. +Resolved 'fs/smb/server/vfs.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[fs-next 5eb3fdca6ca76] Merge branch 'vfs.all' of https://git.kernel.org/pub/scm/linux/kernel/git/vfs/vfs.git +$ git diff -M --stat --summary HEAD^.. + CREDITS | 2 +- + Documentation/admin-guide/binfmt-misc.rst | 3 + + Documentation/filesystems/befs.rst | 4 +- + Documentation/filesystems/bfs.rst | 60 - + Documentation/filesystems/index.rst | 1 - + Documentation/filesystems/locking.rst | 24 +- + Documentation/filesystems/netfs_library.rst | 7 +- + Documentation/filesystems/porting.rst | 25 +- + Documentation/filesystems/proc.rst | 4 + + Documentation/filesystems/sharedsubtree.rst | 20 +- + Documentation/filesystems/squashfs.rst | 6 +- + Documentation/filesystems/vfs.rst | 30 +- + MAINTAINERS | 8 +- + arch/mips/configs/malta_defconfig | 1 - + arch/mips/configs/malta_kvm_defconfig | 1 - + arch/mips/configs/maltaup_xpa_defconfig | 1 - + arch/mips/configs/rm200_defconfig | 1 - + arch/powerpc/Kconfig | 1 - + arch/powerpc/configs/fsl-emb-nonhw.config | 1 - + arch/powerpc/configs/ppc6xx_defconfig | 1 - + arch/powerpc/include/asm/elf.h | 6 - + arch/powerpc/include/asm/spu.h | 3 - + arch/powerpc/platforms/cell/Kconfig | 1 - + arch/powerpc/platforms/cell/spu_syscalls.c | 20 - + arch/powerpc/platforms/cell/spufs/Makefile | 1 - + arch/powerpc/platforms/cell/spufs/coredump.c | 183 -- + arch/powerpc/platforms/cell/spufs/file.c | 114 -- + arch/powerpc/platforms/cell/spufs/inode.c | 14 +- + arch/powerpc/platforms/cell/spufs/spufs.h | 12 - + arch/powerpc/platforms/cell/spufs/syscalls.c | 4 - + block/bio-integrity-fs.c | 15 +- + block/bio-integrity.c | 1 + + block/bio.c | 219 +-- + block/blk-map.c | 2 +- + block/blk-settings.c | 6 + + block/fops.c | 3 +- + block/partitions/efi.h | 8 +- + drivers/android/binder/rust_binderfs.c | 2 +- + drivers/android/binderfs.c | 2 +- + drivers/base/devtmpfs.c | 108 +- + drivers/gpio/gpiolib-cdev.c | 18 +- + drivers/gpu/drm/msm/msm_perfcntr.c | 4 +- + drivers/media/mc/mc-request.c | 8 +- + drivers/misc/ntsync.c | 6 +- + fs/9p/acl.c | 4 +- + fs/9p/acl.h | 4 +- + fs/9p/v9fs.h | 2 +- + fs/9p/v9fs_vfs.h | 2 +- + fs/9p/vfs_addr.c | 1 - + fs/9p/vfs_inode.c | 14 +- + fs/9p/vfs_inode_dotl.c | 14 +- + fs/9p/xattr.c | 2 +- + fs/Kconfig | 9 +- + fs/Makefile | 1 - + fs/adfs/adfs.h | 2 +- + fs/adfs/dir.c | 2 +- + fs/adfs/inode.c | 2 +- + fs/affs/affs.h | 10 +- + fs/affs/inode.c | 2 +- + fs/affs/namei.c | 8 +- + fs/afs/dir.c | 16 +- + fs/afs/file.c | 8 +- + fs/afs/inode.c | 4 +- + fs/afs/internal.h | 8 +- + fs/afs/security.c | 2 +- + fs/afs/xattr.c | 4 +- + fs/aio.c | 11 +- + fs/anon_inodes.c | 4 +- + fs/attr.c | 16 +- + fs/autofs/root.c | 12 +- + fs/backing-file.c | 2 +- + fs/bad_inode.c | 20 +- + fs/bfs/Kconfig | 21 - + fs/bfs/Makefile | 8 - + fs/bfs/bfs.h | 69 - + fs/bfs/dir.c | 4 +- + fs/bfs/file.c | 203 -- + fs/bfs/inode.c | 538 ------ + fs/binfmt_elf.c | 18 +- + fs/binfmt_elf_fdpic.c | 14 +- + fs/binfmt_misc.c | 12 +- + fs/bpf_fs_kfuncs.c | 4 +- + fs/btrfs/acl.c | 2 +- + fs/btrfs/acl.h | 2 +- + fs/btrfs/btrfs_inode.h | 2 +- + fs/btrfs/inode.c | 24 +- + fs/btrfs/ioctl.c | 16 +- + fs/btrfs/ioctl.h | 2 +- + fs/btrfs/xattr.c | 6 +- + fs/buffer.c | 131 +- + fs/cachefiles/Kconfig | 2 +- + fs/cachefiles/interface.c | 96 +- + fs/cachefiles/internal.h | 18 +- + fs/cachefiles/io.c | 472 +++-- + fs/cachefiles/namei.c | 34 +- + fs/cachefiles/xattr.c | 82 +- + fs/ceph/Kconfig | 1 + + fs/ceph/acl.c | 2 +- + fs/ceph/addr.c | 4 +- + fs/ceph/dir.c | 10 +- + fs/ceph/file.c | 2 +- + fs/ceph/inode.c | 10 +- + fs/ceph/mds_client.h | 2 +- + fs/ceph/super.h | 10 +- + fs/ceph/xattr.c | 2 +- + fs/char_dev.c | 4 +- + fs/coda/coda_linux.h | 6 +- + fs/coda/dir.c | 10 +- + fs/coda/inode.c | 4 +- + fs/coda/pioctl.c | 4 +- + fs/configfs/configfs_internal.h | 4 +- + fs/configfs/dir.c | 2 +- + fs/configfs/inode.c | 2 +- + fs/configfs/symlink.c | 2 +- + fs/coredump.c | 558 ++++-- + fs/dax.c | 26 +- + fs/dcache.c | 206 +- + fs/debugfs/inode.c | 2 +- + fs/devpts/inode.c | 2 + + fs/ecryptfs/inode.c | 26 +- + fs/efivarfs/inode.c | 6 +- + fs/erofs/inode.c | 2 +- + fs/erofs/internal.h | 2 +- + fs/eventfd.c | 4 +- + fs/eventpoll.c | 6 +- + fs/exec.c | 28 +- + fs/exfat/exfat_fs.h | 4 +- + fs/exfat/file.c | 6 +- + fs/exfat/misc.c | 2 +- + fs/exfat/namei.c | 6 +- + fs/ext2/acl.c | 2 +- + fs/ext2/acl.h | 2 +- + fs/ext2/ext2.h | 6 +- + fs/ext2/inode.c | 4 +- + fs/ext2/ioctl.c | 2 +- + fs/ext2/namei.c | 12 +- + fs/ext2/xattr.c | 2 +- + fs/ext2/xattr_security.c | 2 +- + fs/ext2/xattr_trusted.c | 2 +- + fs/ext2/xattr_user.c | 2 +- + fs/ext4/acl.c | 2 +- + fs/ext4/acl.h | 2 +- + fs/ext4/ext4.h | 10 +- + fs/ext4/ext4_jbd2.c | 2 +- + fs/ext4/ialloc.c | 2 +- + fs/ext4/inode.c | 6 +- + fs/ext4/ioctl.c | 6 +- + fs/ext4/mmp.c | 2 +- + fs/ext4/namei.c | 16 +- + fs/ext4/symlink.c | 2 +- + fs/ext4/xattr_hurd.c | 2 +- + fs/ext4/xattr_security.c | 2 +- + fs/ext4/xattr_trusted.c | 2 +- + fs/ext4/xattr_user.c | 2 +- + fs/f2fs/acl.c | 6 +- + fs/f2fs/acl.h | 2 +- + fs/f2fs/f2fs.h | 8 +- + fs/f2fs/file.c | 14 +- + fs/f2fs/namei.c | 24 +- + fs/f2fs/xattr.c | 4 +- + fs/failfs.c | 4 +- + fs/fat/fat.h | 4 +- + fs/fat/file.c | 6 +- + fs/fat/misc.c | 2 +- + fs/fat/namei_msdos.c | 6 +- + fs/fat/namei_vfat.c | 6 +- + fs/fhandle.c | 2 +- + fs/file.c | 321 +++- + fs/file_attr.c | 6 +- + fs/fs-writeback.c | 23 +- + fs/fs_pin.c | 9 +- + fs/fuse/acl.c | 4 +- + fs/fuse/dir.c | 51 +- + fs/fuse/file.c | 2 +- + fs/fuse/fuse_i.h | 12 +- + fs/fuse/ioctl.c | 2 +- + fs/fuse/req.c | 8 +- + fs/fuse/xattr.c | 2 +- + fs/gfs2/acl.c | 2 +- + fs/gfs2/acl.h | 2 +- + fs/gfs2/file.c | 2 +- + fs/gfs2/inode.c | 16 +- + fs/gfs2/inode.h | 4 +- + fs/gfs2/log.c | 4 +- + fs/gfs2/lops.c | 4 +- + fs/gfs2/xattr.c | 2 +- + fs/hfs/attr.c | 2 +- + fs/hfs/dir.c | 6 +- + fs/hfs/hfs_fs.h | 2 +- + fs/hfs/inode.c | 2 +- + fs/hfsplus/dir.c | 10 +- + fs/hfsplus/hfsplus_fs.h | 4 +- + fs/hfsplus/inode.c | 6 +- + fs/hfsplus/xattr.c | 2 +- + fs/hfsplus/xattr_security.c | 2 +- + fs/hfsplus/xattr_trusted.c | 2 +- + fs/hfsplus/xattr_user.c | 2 +- + fs/hostfs/hostfs_kern.c | 14 +- + fs/hpfs/hpfs_fn.h | 2 +- + fs/hpfs/inode.c | 2 +- + fs/hpfs/namei.c | 10 +- + fs/hugetlbfs/inode.c | 14 +- + fs/inode.c | 14 +- + fs/internal.h | 30 +- + fs/iomap/bio.c | 4 +- + fs/iomap/buffered-io.c | 8 +- + fs/iomap/direct-io.c | 46 +- + fs/iomap/ioend.c | 122 +- + fs/jbd2/commit.c | 22 +- + fs/jbd2/journal.c | 31 +- + fs/jbd2/transaction.c | 2 +- + fs/jffs2/acl.c | 2 +- + fs/jffs2/acl.h | 2 +- + fs/jffs2/dir.c | 20 +- + fs/jffs2/fs.c | 2 +- + fs/jffs2/os-linux.h | 2 +- + fs/jffs2/security.c | 2 +- + fs/jffs2/xattr_trusted.c | 2 +- + fs/jffs2/xattr_user.c | 2 +- + fs/jfs/acl.c | 2 +- + fs/jfs/file.c | 2 +- + fs/jfs/ioctl.c | 2 +- + fs/jfs/jfs_acl.h | 2 +- + fs/jfs/jfs_inode.h | 4 +- + fs/jfs/namei.c | 10 +- + fs/jfs/xattr.c | 4 +- + fs/kernfs/dir.c | 231 ++- + fs/kernfs/file.c | 47 +- + fs/kernfs/inode.c | 10 +- + fs/kernfs/kernfs-internal.h | 24 +- + fs/kernfs/mount.c | 34 +- + fs/kernfs/symlink.c | 17 +- + fs/libfs.c | 8 +- + fs/minix/file.c | 2 +- + fs/minix/inode.c | 2 +- + fs/minix/minix.h | 2 +- + fs/minix/namei.c | 12 +- + fs/mnt_idmapping.c | 33 +- + fs/mount.h | 3 +- + fs/namei.c | 153 +- + fs/namespace.c | 75 +- + fs/netfs/Kconfig | 3 + + fs/netfs/Makefile | 2 +- + fs/netfs/buffered_read.c | 217 ++- + fs/netfs/buffered_write.c | 69 +- + fs/netfs/direct_read.c | 14 +- + fs/netfs/direct_write.c | 8 +- + fs/netfs/fscache_cookie.c | 8 +- + fs/netfs/fscache_internal.h | 14 - + fs/netfs/fscache_io.c | 10 +- + fs/netfs/internal.h | 70 +- + fs/netfs/iterator.c | 2 +- + fs/netfs/main.c | 1 - + fs/netfs/misc.c | 10 +- + fs/netfs/objects.c | 7 +- + fs/netfs/read_collect.c | 14 +- + fs/netfs/read_pgpriv2.c | 13 +- + fs/netfs/read_retry.c | 7 +- + fs/netfs/read_single.c | 50 +- + fs/netfs/stats.c | 4 +- + fs/netfs/write_collect.c | 157 +- + fs/netfs/write_issue.c | 141 +- + fs/netfs/write_retry.c | 8 +- + fs/nfs/Kconfig | 1 + + fs/nfs/dir.c | 12 +- + fs/nfs/inode.c | 4 +- + fs/nfs/internal.h | 10 +- + fs/nfs/namespace.c | 4 +- + fs/nfs/nfs3_fs.h | 2 +- + fs/nfs/nfs3acl.c | 2 +- + fs/nfs/nfs4proc.c | 10 +- + fs/nfs/unlink.c | 3 + + fs/nfsd/vfs.c | 13 +- + fs/nilfs2/inode.c | 4 +- + fs/nilfs2/ioctl.c | 2 +- + fs/nilfs2/namei.c | 10 +- + fs/nilfs2/nilfs.h | 6 +- + fs/nls/nls_iso8859-14.c | 26 +- + fs/nsfs.c | 4 +- + fs/ntfs/ea.c | 10 +- + fs/ntfs/ea.h | 6 +- + fs/ntfs/file.c | 4 +- + fs/ntfs/inode.h | 4 +- + fs/ntfs/namei.c | 12 +- + fs/ntfs3/file.c | 6 +- + fs/ntfs3/inode.c | 2 +- + fs/ntfs3/namei.c | 10 +- + fs/ntfs3/ntfs_fs.h | 16 +- + fs/ntfs3/xattr.c | 12 +- + fs/ocfs2/acl.c | 2 +- + fs/ocfs2/acl.h | 2 +- + fs/ocfs2/buffer_head_io.c | 12 +- + fs/ocfs2/dlmfs/dlmfs.c | 6 +- + fs/ocfs2/file.c | 6 +- + fs/ocfs2/file.h | 6 +- + fs/ocfs2/ioctl.c | 2 +- + fs/ocfs2/ioctl.h | 2 +- + fs/ocfs2/journal.c | 25 +- + fs/ocfs2/namei.c | 10 +- + fs/ocfs2/xattr.c | 6 +- + fs/omfs/dir.c | 6 +- + fs/omfs/file.c | 2 +- + fs/omfs/inode.c | 4 +- + fs/open.c | 50 +- + fs/orangefs/acl.c | 2 +- + fs/orangefs/inode.c | 8 +- + fs/orangefs/namei.c | 8 +- + fs/orangefs/orangefs-kernel.h | 8 +- + fs/orangefs/xattr.c | 2 +- + fs/overlayfs/dir.c | 14 +- + fs/overlayfs/file.c | 2 +- + fs/overlayfs/inode.c | 16 +- + fs/overlayfs/overlayfs.h | 14 +- + fs/overlayfs/ovl_entry.h | 2 +- + fs/overlayfs/util.c | 4 +- + fs/overlayfs/xattrs.c | 4 +- + fs/pidfs.c | 6 +- + fs/pipe.c | 2 +- + fs/pnode.c | 108 +- + fs/posix_acl.c | 26 +- + fs/proc/base.c | 53 +- + fs/proc/fd.c | 6 +- + fs/proc/fd.h | 2 +- + fs/proc/generic.c | 4 +- + fs/proc/internal.h | 4 +- + fs/proc/proc_net.c | 2 +- + fs/proc/proc_sysctl.c | 6 +- + fs/proc/root.c | 2 +- + fs/proc/vmcore.c | 20 + + fs/quota/dquot.c | 2 +- + fs/ramfs/file-nommu.c | 4 +- + fs/ramfs/inode.c | 10 +- + fs/read_write.c | 8 +- + fs/remap_range.c | 2 +- + fs/smb/client/cifsacl.c | 4 +- + fs/smb/client/cifsfs.c | 2 +- + fs/smb/client/cifsfs.h | 16 +- + fs/smb/client/cifsproto.h | 4 +- + fs/smb/client/dir.c | 6 +- + fs/smb/client/inode.c | 8 +- + fs/smb/client/link.c | 2 +- + fs/smb/client/transport.c | 13 +- + fs/smb/client/xattr.c | 2 +- + fs/smb/server/ndr.c | 2 +- + fs/smb/server/ndr.h | 2 +- + fs/smb/server/oplock.c | 2 +- + fs/smb/server/smb2pdu.c | 22 +- + fs/smb/server/smb_common.c | 2 +- + fs/smb/server/smbacl.c | 20 +- + fs/smb/server/smbacl.h | 8 +- + fs/smb/server/vfs.c | 44 +- + fs/smb/server/vfs.h | 32 +- + fs/splice.c | 70 +- + fs/stat.c | 4 +- + fs/super.c | 19 +- + fs/tests/.kunitconfig | 2 + + fs/tests/fdtable_kunit.c | 72 + + fs/tracefs/event_inode.c | 2 +- + fs/tracefs/inode.c | 8 +- + fs/ubifs/dir.c | 14 +- + fs/ubifs/file.c | 4 +- + fs/ubifs/ioctl.c | 2 +- + fs/ubifs/ubifs.h | 6 +- + fs/ubifs/xattr.c | 2 +- + fs/udf/file.c | 2 +- + fs/udf/namei.c | 12 +- + fs/udf/symlink.c | 2 +- + fs/ufs/dir.c | 2 +- + fs/ufs/inode.c | 2 +- + fs/ufs/namei.c | 10 +- + fs/ufs/ufs.h | 2 +- + fs/vboxsf/dir.c | 8 +- + fs/vboxsf/utils.c | 4 +- + fs/vboxsf/vfsmod.h | 4 +- + fs/xattr.c | 30 +- + fs/xfs/libxfs/xfs_errortag.h | 6 +- + fs/xfs/libxfs/xfs_inode_util.h | 2 +- + fs/xfs/xfs_acl.c | 2 +- + fs/xfs/xfs_acl.h | 2 +- + fs/xfs/xfs_aops.c | 13 +- + fs/xfs/xfs_buf.c | 11 + + fs/xfs/xfs_file.c | 11 +- + fs/xfs/xfs_handle.c | 6 +- + fs/xfs/xfs_inode.c | 4 +- + fs/xfs/xfs_inode.h | 2 +- + fs/xfs/xfs_ioctl.c | 2 +- + fs/xfs/xfs_ioctl.h | 2 +- + fs/xfs/xfs_ioend.c | 137 +- + fs/xfs/xfs_ioend.h | 2 + + fs/xfs/xfs_iops.c | 24 +- + fs/xfs/xfs_iops.h | 2 +- + fs/xfs/xfs_itable.c | 2 +- + fs/xfs/xfs_itable.h | 2 +- + fs/xfs/xfs_mount.h | 8 + + fs/xfs/xfs_super.c | 1 + + fs/xfs/xfs_symlink.c | 2 +- + fs/xfs/xfs_symlink.h | 2 +- + fs/xfs/xfs_sysfs.c | 78 +- + fs/xfs/xfs_trace.h | 1 + + fs/xfs/xfs_xattr.c | 2 +- + fs/xfs/xfs_zone_alloc.c | 4 + + fs/zonefs/super.c | 2 +- + include/linux/binfmts.h | 3 +- + include/linux/bio-integrity.h | 3 +- + include/linux/bio.h | 11 +- + include/linux/blkdev.h | 8 +- + include/linux/buffer_head.h | 86 +- + include/linux/capability.h | 8 +- + include/linux/cleanup.h | 7 - + include/linux/coredump.h | 37 +- + include/linux/dax.h | 12 - + include/linux/dcache.h | 38 +- + include/linux/fdtable.h | 15 +- + include/linux/file.h | 130 +- + include/linux/fileattr.h | 2 +- + include/linux/fs.h | 127 +- + include/linux/fs_context.h | 4 + + include/linux/fscache-cache.h | 2 +- + include/linux/fscache.h | 53 +- + include/linux/iomap.h | 38 +- + include/linux/lsm_hook_defs.h | 25 +- + include/linux/mnt_idmapping.h | 24 +- + include/linux/mount.h | 4 +- + include/linux/namei.h | 19 +- + include/linux/netfs.h | 112 +- + include/linux/nfs_fs.h | 6 +- + include/linux/posix_acl.h | 24 +- + include/linux/quotaops.h | 6 +- + include/linux/sched.h | 2 +- + include/linux/sched/signal.h | 33 +- + include/linux/security.h | 61 +- + include/linux/splice.h | 4 +- + include/linux/uidgid.h | 12 +- + include/linux/user_namespace.h | 11 +- + include/linux/wait_bit.h | 26 + + include/linux/xattr.h | 20 +- + include/trace/events/cachefiles.h | 81 +- + include/trace/events/fscache.h | 10 +- + include/trace/events/netfs.h | 145 +- + include/uapi/linux/close_range.h | 31 +- + include/uapi/linux/coredump.h | 149 +- + include/uapi/linux/fs.h | 2 +- + init/Kconfig | 11 + + init/initramfs.c | 11 +- + io_uring/io-wq.c | 2 + + io_uring/mock_file.c | 8 +- + ipc/mqueue.c | 9 +- + kernel/Makefile | 1 + + kernel/bpf/bpf_iter.c | 6 +- + kernel/bpf/inode.c | 6 +- + kernel/bpf/token.c | 6 +- + kernel/capability.c | 4 +- + kernel/exit.c | 15 +- + kernel/fork.c | 68 +- + kernel/kthread.c | 2 +- + kernel/pid_namespace.c | 3 +- + kernel/ptrace.c | 6 + + kernel/signal.c | 14 + + kernel/tests/.kunitconfig | 4 + + kernel/tests/user_ns_map_kunit.c | 98 + + kernel/user_namespace.c | 60 +- + kernel/utsname.c | 1 - + mm/secretmem.c | 2 +- + mm/shmem.c | 30 +- + mm/userfaultfd.c | 6 +- + net/core/scm.c | 8 +- + net/handshake/netlink.c | 8 +- + net/kcm/kcmsock.c | 6 +- + net/socket.c | 6 +- + net/unix/af_unix.c | 2 +- + security/apparmor/apparmorfs.c | 2 +- + security/apparmor/lsm.c | 4 +- + security/commoncap.c | 10 +- + security/integrity/evm/evm_main.c | 26 +- + security/integrity/ima/ima.h | 10 +- + security/integrity/ima/ima_api.c | 2 +- + security/integrity/ima/ima_appraise.c | 12 +- + security/integrity/ima/ima_main.c | 6 +- + security/integrity/ima/ima_policy.c | 4 +- + security/security.c | 49 +- + security/selinux/hooks.c | 45 +- + security/selinux/selinuxfs.c | 2 +- + security/smack/smack_lsm.c | 14 +- + tools/include/uapi/linux/coredump.h | 149 +- + tools/testing/selftests/Makefile | 3 + + .../selftests/clone3/clone3_clear_sighand.c | 6 +- + tools/testing/selftests/core/close_range_test.c | 951 ++++++++++ + tools/testing/selftests/coredump/.gitignore | 2 + + tools/testing/selftests/coredump/Makefile | 11 +- + .../selftests/coredump/coredump_notify_signal.h | 29 + + .../coredump/coredump_notify_signal_helper.c | 46 + + .../coredump/coredump_notify_signal_test.c | 245 +++ + .../selftests/coredump/coredump_signal_test.c | 238 +++ + .../coredump/coredump_socket_protocol_test.c | 1983 +++++++++++++++----- + tools/testing/selftests/coredump/coredump_test.h | 32 +- + .../selftests/coredump/coredump_test_helpers.c | 1742 ++++++++++++++++- + .../selftests/coredump/coredump_test_helpers.h | 79 + + .../selftests/coredump/coredump_worker_test.c | 447 +++++ + tools/testing/selftests/exec/Makefile | 4 + + tools/testing/selftests/exec/binfmt_misc_delim.c | 127 ++ + tools/testing/selftests/filesystems/.gitignore | 1 - + tools/testing/selftests/filesystems/Makefile | 2 +- + tools/testing/selftests/filesystems/config | 8 + + .../selftests/filesystems/file_stressor/.gitignore | 2 + + .../selftests/filesystems/file_stressor/Makefile | 6 + + .../{ => file_stressor}/file_stressor.c | 0 + .../selftests/filesystems/file_stressor/settings | 3 + + .../selftests/filesystems/fscontext_ns/.gitignore | 2 + + tools/testing/selftests/filesystems/kernfs_test.c | 1203 +++++++++++- + .../filesystems/mntns_unbindable/Makefile | 6 + + .../mntns_unbindable/mntns_unbindable_test.c | 227 +++ + .../selftests/filesystems/openat2/openat2_test.c | 10 +- + .../selftests/filesystems/openat2/resolve_test.c | 7 +- + .../filesystems/statmount/statmount_test.c | 2 +- + .../filesystems/umount_propagation/Makefile | 6 + + .../umount_propagation/umount_propagation_test.c | 226 +++ + .../move_mount_set_group_test.c | 74 +- + tools/testing/selftests/pidfd/pidfd_open_test.c | 2 +- + virt/kvm/guest_memfd.c | 2 +- + 519 files changed, 12646 insertions(+), 5165 deletions(-) + delete mode 100644 Documentation/filesystems/bfs.rst + delete mode 100644 arch/powerpc/platforms/cell/spufs/coredump.c + delete mode 100644 fs/bfs/Kconfig + delete mode 100644 fs/bfs/Makefile + delete mode 100644 fs/bfs/bfs.h + delete mode 100644 fs/bfs/file.c + delete mode 100644 fs/bfs/inode.c + delete mode 100644 fs/netfs/fscache_internal.h + create mode 100644 fs/tests/.kunitconfig + create mode 100644 fs/tests/fdtable_kunit.c + create mode 100644 kernel/tests/.kunitconfig + create mode 100644 kernel/tests/user_ns_map_kunit.c + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal.h + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal_helper.c + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal_test.c + create mode 100644 tools/testing/selftests/coredump/coredump_signal_test.c + create mode 100644 tools/testing/selftests/coredump/coredump_test_helpers.h + create mode 100644 tools/testing/selftests/coredump/coredump_worker_test.c + create mode 100644 tools/testing/selftests/exec/binfmt_misc_delim.c + create mode 100644 tools/testing/selftests/filesystems/config + create mode 100644 tools/testing/selftests/filesystems/file_stressor/.gitignore + create mode 100644 tools/testing/selftests/filesystems/file_stressor/Makefile + rename tools/testing/selftests/filesystems/{ => file_stressor}/file_stressor.c (100%) + create mode 100644 tools/testing/selftests/filesystems/file_stressor/settings + create mode 100644 tools/testing/selftests/filesystems/fscontext_ns/.gitignore + create mode 100644 tools/testing/selftests/filesystems/mntns_unbindable/Makefile + create mode 100644 tools/testing/selftests/filesystems/mntns_unbindable/mntns_unbindable_test.c + create mode 100644 tools/testing/selftests/filesystems/umount_propagation/Makefile + create mode 100644 tools/testing/selftests/filesystems/umount_propagation/umount_propagation_test.c +$ git am -3 ../patches/0001-ksmbd-Fix-removal-of-type-parameter-from-vfs_path_pa.patch +Applying: ksmbd: Fix removal of type parameter from vfs_path_parent_lookup() +Using index info to reconstruct a base tree... +M fs/smb/server/vfs.c +Falling back to patching base and 3-way merge... +Auto-merging fs/smb/server/vfs.c +No changes -- Patch already applied. +Merging vfs/for-next (4dda01b67c866 Merge branches 'work.dcache', 'work.dcache-d_add' and 'work.configfs' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git vfs/for-next +Merge made by the 'ort' strategy. +Merging mm-fixes/for-next-fixes (ffd79f732828e Merge https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm.git mm-hotfixes-unstable into for-next-fixes) +$ git merge -m Merge branch 'for-next-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mm/linux.git mm-fixes/for-next-fixes +Updating 551c722f40809..ffd79f732828e +Fast-forward (no commit created; -m option ignored) + .mailmap | 3 +- + MAINTAINERS | 2 +- + drivers/char/mem.c | 2 +- + lib/test_xarray.c | 70 ++++++++++++++++++ + lib/xarray.c | 8 ++- + mm/hugetlb.c | 25 +++++-- + mm/mremap.c | 87 +++++++++++++--------- + mm/page_alloc.c | 103 ++++++++++++++++++--------- + mm/pgtable-generic.c | 20 +++++- + mm/slub.c | 7 +- + mm/userfaultfd.c | 1 + + mm/vma.c | 19 ++++- + mm/vma.h | 7 +- + mm/vmalloc.c | 79 +++++++++++++------- + tools/testing/selftests/mm/hugetlb-mmap.c | 4 +- + tools/testing/selftests/mm/uffd-unit-tests.c | 84 ++++++++++++++++++++++ + tools/testing/vma/tests/vma.c | 10 +-- + 17 files changed, 413 insertions(+), 118 deletions(-) +Merging fs-current (5a9410e0a34d2 Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git) +$ git merge -m Merge branch 'fs-current' of linux-next fs-current +Merge made by the 'ort' strategy. +Merging kbuild-current/kbuild-fixes-for-next (fd73f4a665989 Linux 7.3-rc3) +$ git merge -m Merge branch 'kbuild-fixes-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git kbuild-current/kbuild-fixes-for-next +Already up to date. +Merging clang-fixes-current/clang-fixes-for-current (df2908090cda3 Linux 7.3-rc2) +$ git merge -m Merge branch 'clang-fixes-for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/nathan/linux.git clang-fixes-current/clang-fixes-for-current +Already up to date. +Merging arc-current/for-curr (f050c3e61d2a1 ARC: arch_cmpxchg_relaxed to use size of pointed type not pointer) +$ git merge -m Merge branch 'for-curr' of https://git.kernel.org/pub/scm/linux/kernel/git/vgupta/arc.git arc-current/for-curr +Already up to date. +Merging arm-current/fixes (1039bffd6ae9c ARM: 9485/1: mm: acquire mmap write lock around show_pte() for user faults) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/rmk/linux.git arm-current/fixes +Already up to date. +Merging arm64-fixes/for-next/fixes (3872cc6b92940 arm64: topology: fix arch_freq_get_on_cpu() overflow above 4.19 GHz) +$ git merge -m Merge branch 'for-next/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux arm64-fixes/for-next/fixes +Merge made by the 'ort' strategy. + arch/arm64/kernel/topology.c | 6 ++---- + arch/arm64/mm/mmu.c | 2 +- + 2 files changed, 3 insertions(+), 5 deletions(-) +Merging arm-soc-fixes/arm/fixes (4dd1999783d7d soc: samsung: exynos-pmu: fix use-after-free of interrupt generator node) +$ git merge -m Merge branch 'arm/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc.git arm-soc-fixes/arm/fixes +Already up to date. +Merging davinci-current/davinci/for-current (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'davinci/for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git davinci-current/davinci/for-current +Already up to date. +Merging realtek-fixes/fixes (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux.git realtek-fixes/fixes +Already up to date. +Merging drivers-memory-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl.git drivers-memory-fixes/fixes +Already up to date. +Merging sophgo-fixes/fixes (19272b37aa4f8 Linux 6.16-rc1) +$ git merge -m Merge branch 'fixes' of https://github.com/sophgo/linux.git sophgo-fixes/fixes +Already up to date. +Merging sophgo-soc-fixes/soc-fixes (0af2f6be1b428 Linux 6.15-rc1) +$ git merge -m Merge branch 'soc-fixes' of https://github.com/sophgo/linux.git sophgo-soc-fixes/soc-fixes +Already up to date. +Merging m68k-current/for-linus (2f8e3cad53b5c m68k: nfcon: Do not call console_is_registered() in nfcon_device()) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/linux-m68k.git m68k-current/for-linus +Already up to date. +Merging powerpc-fixes/fixes (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git powerpc-fixes/fixes +Already up to date. +Merging s390-fixes/fixes (5b76268dac968 s390/cio: Fix NULL pointer dereference in ccw_device_get_util_str()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/s390/linux.git s390-fixes/fixes +Already up to date. +Merging net/main (99b43ede9e355 net: microchip: vcap: stop scanning after deleting key field) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net.git net/main +Auto-merging .mailmap +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .mailmap | 2 + + MAINTAINERS | 9 +- + drivers/bluetooth/btintel.c | 15 +- + drivers/bluetooth/btintel_pcie.c | 12 +- + drivers/net/ethernet/broadcom/bcmsysport.c | 8 +- + drivers/net/ethernet/broadcom/genet/bcmgenet.c | 34 +- + drivers/net/ethernet/broadcom/genet/bcmgenet.h | 2 + + drivers/net/ethernet/google/gve/gve_tx_dqo.c | 14 +- + drivers/net/ethernet/marvell/octeontx2/af/mcs.c | 4 +- + drivers/net/ethernet/marvell/octeontx2/nic/cn20k.c | 11 +- + .../ethernet/marvell/octeontx2/nic/otx2_common.c | 16 +- + .../ethernet/marvell/octeontx2/nic/otx2_common.h | 9 + + drivers/net/ethernet/mediatek/mtk_eth_soc.c | 38 ++- + drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c | 10 +- + drivers/net/ethernet/microchip/sparx5/sparx5_ptp.c | 3 + + drivers/net/ethernet/microchip/vcap/vcap_api.c | 2 +- + drivers/net/ethernet/stmicro/stmmac/stmmac_main.c | 83 +++-- + drivers/net/pcs/pcs-rzn1-miic.c | 3 + + drivers/net/pfcp.c | 3 + + drivers/net/phy/qcom/at803x.c | 8 +- + include/linux/netfilter_netdev.h | 2 +- + include/linux/rtnetlink.h | 7 +- + include/linux/skbuff.h | 2 +- + include/net/bluetooth/hci_core.h | 5 +- + include/net/gso.h | 8 + + include/net/sch_generic.h | 9 + + net/bluetooth/hci_conn.c | 18 +- + net/bluetooth/hci_core.c | 54 ++- + net/bluetooth/hci_debugfs.c | 2 +- + net/bluetooth/hci_sync.c | 19 +- + net/bluetooth/rfcomm/tty.c | 7 +- + net/bluetooth/smp.c | 2 + + net/core/dev.c | 78 +++-- + net/core/dev_ioctl.c | 3 +- + net/core/gso.c | 1 + + net/core/page_pool.c | 2 +- + net/core/skbuff.c | 8 +- + net/core/sock.c | 4 +- + net/ethtool/tsconfig.c | 24 +- + net/ipv4/af_inet.c | 3 + + net/ipv4/tcp_input.c | 6 + + net/ipv6/addrconf.c | 2 +- + net/ipv6/ip6_offload.c | 4 + + net/ipv6/seg6_local.c | 140 ++++++-- + net/mctp/device.c | 6 +- + net/packet/af_packet.c | 3 + + net/sched/act_skbedit.c | 5 +- + net/sched/cls_api.c | 15 +- + net/sched/sch_codel.c | 2 +- + net/sched/sch_fq_codel.c | 4 +- + net/tipc/crypto.c | 20 +- + rust/kernel/net/netlink.rs | 9 + + tools/testing/selftests/net/.gitignore | 1 + + tools/testing/selftests/net/Makefile | 1 + + .../selftests/net/packetdrill/tcp_old_ack_ts.pkt | 22 ++ + tools/testing/selftests/net/udp_splice_checksum.c | 378 +++++++++++++++++++++ + 56 files changed, 940 insertions(+), 222 deletions(-) + create mode 100644 tools/testing/selftests/net/packetdrill/tcp_old_ack_ts.pkt + create mode 100644 tools/testing/selftests/net/udp_splice_checksum.c +Merging bpf/master (564b7e8e09ec9 Merge branch 'bpf-fix-use-after-free-of-progs-detached-from-busy-trampolines') +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf.git/ bpf/master +Merge made by the 'ort' strategy. + arch/arm64/net/bpf_jit_comp.c | 37 ++-- + arch/loongarch/net/bpf_jit.c | 60 ++++--- + arch/powerpc/net/bpf_jit_comp.c | 45 ++--- + arch/riscv/net/bpf_jit_comp64.c | 56 ++++-- + arch/s390/net/bpf_jit_comp.c | 46 +++-- + arch/x86/net/bpf_jit_comp.c | 26 +-- + include/linux/bpf.h | 45 ++++- + kernel/bpf/core.c | 10 +- + kernel/bpf/hashtab.c | 17 +- + kernel/bpf/syscall.c | 19 ++- + kernel/bpf/trampoline.c | 85 ++++++++-- + .../selftests/bpf/prog_tests/bpf_mod_race.c | 34 +--- + .../selftests/bpf/prog_tests/tramp_prog_detach.c | 188 +++++++++++++++++++++ + .../selftests/bpf/progs/tramp_prog_detach.c | 56 ++++++ + tools/testing/selftests/bpf/testing_helpers.c | 28 +++ + tools/testing/selftests/bpf/testing_helpers.h | 2 + + 16 files changed, 596 insertions(+), 158 deletions(-) + create mode 100644 tools/testing/selftests/bpf/prog_tests/tramp_prog_detach.c + create mode 100644 tools/testing/selftests/bpf/progs/tramp_prog_detach.c +Merging ipsec/master (868f63c8bfafa xfrm: retry inexact policy lookup after node reinsertion) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/klassert/ipsec.git ipsec/master +Merge made by the 'ort' strategy. + net/xfrm/xfrm_policy.c | 11 +++++++++-- + 1 file changed, 9 insertions(+), 2 deletions(-) +Merging netfilter/main (9c572a83037a7 net/sched: fix potential stack infoleak in em_text_dump()) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf.git netfilter/main +Already up to date. +Merging ipvs/main (a401a9d547c50 Merge branch 'net-macb-fix-two-probe-path-leaks') +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs.git ipvs/main +Already up to date. +Merging bluetooth-fixes/master (86ef0f58bdecd Bluetooth: MGMT: Fix status of pending commands flushed on power off) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/bluetooth/bluetooth.git bluetooth-fixes/master +Merge made by the 'ort' strategy. + drivers/bluetooth/btintel_pcie.c | 9 ++- + include/net/bluetooth/rfcomm.h | 1 + + net/bluetooth/mgmt.c | 2 +- + net/bluetooth/rfcomm/core.c | 134 +++++++++++++++++++++++++++------------ + net/bluetooth/rfcomm/sock.c | 5 +- + net/bluetooth/sco.c | 87 ++++++++++++++++++------- + 6 files changed, 174 insertions(+), 64 deletions(-) +Merging wireless/for-next (6f63e919fe1e3 wifi: mac80211: fix slab-out-of-bounds read in ieee80211_monitor_select_queue()) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/wireless/wireless.git wireless/for-next +Merge made by the 'ort' strategy. + drivers/net/wireless/ath/ath11k/mac.c | 1 + + drivers/net/wireless/ath/ath9k/hif_usb.c | 21 +++--- + drivers/net/wireless/ath/ath9k/htc.h | 1 - + drivers/net/wireless/ath/ath9k/htc_drv_init.c | 7 +- + drivers/net/wireless/ath/ath9k/htc_drv_txrx.c | 8 +-- + drivers/net/wireless/ath/ath9k/wmi.c | 28 +++++--- + drivers/net/wireless/ath/ath9k/wmi.h | 1 + + drivers/net/wireless/intel/iwlegacy/3945-mac.c | 1 + + drivers/net/wireless/intersil/p54/fwio.c | 42 ++++++++++-- + drivers/net/wireless/silabs/wfx/hif_tx.c | 4 +- + drivers/net/wireless/st/cw1200/bh.c | 8 ++- + drivers/net/wireless/st/cw1200/txrx.c | 14 ++++ + drivers/net/wireless/ti/wlcore/main.c | 4 +- + net/mac80211/agg-rx.c | 2 +- + net/mac80211/chan.c | 50 ++++++++++++--- + net/mac80211/drop.h | 1 + + net/mac80211/fils_aead.c | 8 +++ + net/mac80211/ieee80211_i.h | 2 +- + net/mac80211/iface.c | 4 +- + net/mac80211/mesh_pathtbl.c | 30 +++++++-- + net/mac80211/mlme.c | 4 +- + net/mac80211/rc80211_minstrel_ht.c | 89 +++++++++++++++++++++++--- + net/mac80211/rx.c | 6 ++ + net/mac80211/spectmgmt.c | 4 +- + net/mac80211/sta_info.c | 6 +- + net/mac80211/status.c | 70 ++++++++++++++++++++ + net/mac80211/tx.c | 45 +++++++++---- + net/wireless/nl80211.c | 11 ++-- + net/wireless/scan.c | 22 ++++--- + 29 files changed, 395 insertions(+), 99 deletions(-) +Merging ath/for-current (6f63e919fe1e3 wifi: mac80211: fix slab-out-of-bounds read in ieee80211_monitor_select_queue()) +$ git merge -m Merge branch 'for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/ath/ath.git ath/for-current +Already up to date. +Merging iwlwifi/fixes (6f63e919fe1e3 wifi: mac80211: fix slab-out-of-bounds read in ieee80211_monitor_select_queue()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git iwlwifi/fixes +Already up to date. +Merging wpan/master (2b4707a149a55 net: mctp: i3c: serialize probe with bus removal) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan.git wpan/master +Already up to date. +Merging rdma-fixes/for-rc (72d3fcf802c45 Linux 7.3-rc5) +$ git merge -m Merge branch 'for-rc' of https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git rdma-fixes/for-rc +Already up to date. +Merging sound-current/for-linus (a154f7b9f197b ALSA: usb-audio: Apply IGNORE_CTL_ERROR quirk to all Audient devices) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/tiwai/sound.git sound-current/for-linus +Merge made by the 'ort' strategy. + sound/core/seq/seq_compat.c | 4 ++- + sound/hda/codecs/realtek/alc269.c | 58 +++++++++++++++++++++++++++++++++++++-- + sound/hda/controllers/tegra.c | 3 +- + sound/usb/caiaq/device.c | 3 ++ + sound/usb/line6/playback.c | 7 +++++ + sound/usb/misc/ua101.c | 9 ++++++ + sound/usb/mixer_maps.c | 4 +++ + sound/usb/quirks.c | 16 ++++++++--- + 8 files changed, 95 insertions(+), 9 deletions(-) +Merging sound-asoc-fixes/for-linus (b2047b8cadadc ASoC: codecs: lpass-wsa-macro: rewrite the interpolator volume after enabling clocks) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/sound.git sound-asoc-fixes/for-linus +Merge made by the 'ort' strategy. + sound/soc/amd/acp-config.c | 14 +++++++++++ + sound/soc/amd/acp/Kconfig | 1 + + sound/soc/amd/acp/amd-acp70-acpi-match.c | 20 ++++++++++++++++ + sound/soc/amd/yc/acp6x-mach.c | 14 +++++++++++ + sound/soc/codecs/Kconfig | 4 +--- + sound/soc/codecs/lpass-wsa-macro.c | 3 +++ + sound/soc/codecs/max98363.c | 2 +- + sound/soc/codecs/rt1017-sdca-sdw.c | 27 +++++++--------------- + sound/soc/codecs/rt274.c | 4 +++- + sound/soc/codecs/rt286.c | 2 +- + sound/soc/codecs/wm8903.c | 2 +- + sound/soc/intel/boards/bytcr_rt5651.c | 17 ++++++++++++++ + sound/soc/intel/common/soc-acpi-intel-lnl-match.c | 1 + + .../soc/intel/common/soc-acpi-intel-sdca-quirks.c | 16 +++++++++++++ + .../soc/intel/common/soc-acpi-intel-sdca-quirks.h | 1 + + sound/soc/qcom/lpass-cpu.c | 4 ++-- + sound/soc/sdca/sdca_class_function.c | 19 ++++++++++----- + sound/soc/tegra/tegra186_asrc.c | 2 +- + 18 files changed, 118 insertions(+), 35 deletions(-) +Merging regmap-fixes/for-linus (2e42cade8ff1f regmap: irq: Free the irqdomain we create) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regmap.git regmap-fixes/for-linus +Merge made by the 'ort' strategy. + drivers/base/regmap/regmap-irq.c | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) +Merging regulator-fixes/for-linus (f3e6ef13e24c9 regulator: pf1550: fix which regulator is notified) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regulator.git regulator-fixes/for-linus +Already up to date. +Merging spi-fixes/for-linus (3d743adf090cd spi: fsl-qspi: Reprogram the clock rate when the operation frequency changes) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/spi.git spi-fixes/for-linus +Already up to date. +Merging pci-current/for-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/pci/pci.git pci-current/for-linus +Already up to date. +Merging driver-core.current/driver-core-linus (72d3fcf802c45 Linux 7.3-rc5) +$ git merge -m Merge branch 'driver-core-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git driver-core.current/driver-core-linus +Already up to date. +Merging tty.current/tty-linus (5dad87615c986 tty: add break_wait kernel-doc) +$ git merge -m Merge branch 'tty-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git tty.current/tty-linus +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 5 + + drivers/tty/amiserial.c | 9 +- + drivers/tty/n_gsm.c | 47 ++----- + drivers/tty/serial/8250/8250_bcm7271.c | 2 +- + drivers/tty/serial/8250/8250_omap.c | 15 ++- + drivers/tty/serial/8250/8250_port.c | 2 +- + drivers/tty/serial/kgdboc.c | 9 ++ + drivers/tty/serial/ma35d1_serial.c | 6 +- + drivers/tty/serial/qcom_geni_serial.c | 229 ++++++++++++++++++--------------- + drivers/tty/serial/serial_core.c | 76 ++++++++--- + drivers/tty/serial/vt8500_serial.c | 1 + + drivers/tty/tty_io.c | 36 ++++-- + drivers/tty/tty_port.c | 91 ++++++++----- + drivers/tty/vcc.c | 5 +- + drivers/tty/vt/vt.c | 2 + + include/linux/soc/qcom/geni-se.h | 15 +-- + include/linux/tty.h | 2 + + include/linux/tty_port.h | 22 +--- + 18 files changed, 333 insertions(+), 241 deletions(-) +Merging usb.current/usb-linus (abc36cbda29d8 Merge tag 'usb-serial-7.3-rc4' of ssh://gitolite.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial into usb-linus) +$ git merge -m Merge branch 'usb-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/usb.git usb.current/usb-linus +Merge made by the 'ort' strategy. + drivers/thunderbolt/ctl.c | 63 +++++++++-------- + drivers/thunderbolt/nhi.c | 5 ++ + drivers/thunderbolt/switch.c | 9 ++- + drivers/thunderbolt/tb.c | 37 ++++++---- + drivers/thunderbolt/test.c | 58 +++++++++++++--- + drivers/thunderbolt/tunnel.c | 73 +++++++++++++------- + drivers/thunderbolt/tunnel.h | 8 ++- + drivers/thunderbolt/xdomain.c | 1 - + drivers/usb/cdns3/cdns3-gadget.c | 4 ++ + drivers/usb/chipidea/ci_hdrc_tegra.c | 4 +- + drivers/usb/class/cdc-acm.c | 36 +++++----- + drivers/usb/class/cdc-acm.h | 1 - + drivers/usb/core/hub.c | 7 +- + drivers/usb/dwc2/hcd.c | 3 +- + drivers/usb/dwc3/dwc3-am62.c | 1 + + drivers/usb/gadget/function/f_midi2.c | 4 +- + drivers/usb/host/octeon-hcd.c | 54 +++++++++++++-- + drivers/usb/host/ohci-da8xx.c | 2 + + drivers/usb/host/ohci-s3c2410.c | 1 + + drivers/usb/host/ohci-spear.c | 1 + + drivers/usb/host/ohci-st.c | 1 + + drivers/usb/serial/bus.c | 6 +- + drivers/usb/serial/cp210x.c | 1 + + drivers/usb/serial/generic.c | 10 ++- + drivers/usb/serial/option.c | 14 ++++ + drivers/usb/serial/quatech2.c | 19 ++++- + drivers/usb/serial/ssu100.c | 18 ++++- + drivers/usb/serial/usb-serial.c | 126 +++++++++++++++++++++++++++------- + drivers/usb/serial/xr_serial.c | 10 +-- + drivers/usb/storage/sierra_ms.c | 7 ++ + drivers/usb/typec/anx7411.c | 5 +- + drivers/usb/typec/tcpm/tcpm.c | 6 +- + drivers/usb/typec/tipd/core.c | 11 +-- + drivers/usb/typec/ucsi/displayport.c | 9 ++- + drivers/usb/typec/ucsi/ucsi_acpi.c | 34 +++++++++ + drivers/usb/typec/ucsi/ucsi_glink.c | 15 +++- + include/linux/thunderbolt.h | 3 + + include/linux/usb/serial.h | 1 + + 38 files changed, 499 insertions(+), 169 deletions(-) +Merging usb-serial-fixes/usb-linus (ec06546b43b61 USB: serial: quatech2: fix baud rate overflow) +$ git merge -m Merge branch 'usb-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial.git usb-serial-fixes/usb-linus +Already up to date. +Merging phy/fixes (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git phy/fixes +Already up to date. +Merging staging.current/staging-linus (df2908090cda3 Linux 7.3-rc2) +$ git merge -m Merge branch 'staging-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/staging.git staging.current/staging-linus +Already up to date. +Merging iio-fixes/fixes-togreg (d2f2868d2f487 iio: adc: ad_sigma_delta: fix use-after-free on unbind) +$ git merge -m Merge branch 'fixes-togreg' of https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git iio-fixes/fixes-togreg +Merge made by the 'ort' strategy. + .../bindings/iio/adc/rockchip-saradc.yaml | 18 +++---- + drivers/android/binder.c | 3 +- + drivers/android/binder/node.rs | 20 +++++-- + drivers/android/binder/node/wrapper.rs | 34 +++++++++++- + drivers/android/binder/rust_binderfs.c | 8 +++ + drivers/android/binder/thread.rs | 8 ++- + drivers/android/binderfs.c | 3 +- + drivers/hid/hid-sensor-hub.c | 6 ++- + drivers/iio/accel/kionix-kx022a.c | 30 ++++++++--- + drivers/iio/accel/kxcjk-1013.c | 2 +- + drivers/iio/accel/sca3000.c | 2 +- + drivers/iio/adc/ad4030.c | 11 +++- + drivers/iio/adc/ad7173.c | 4 +- + drivers/iio/adc/ad_sigma_delta.c | 31 ++++++----- + drivers/iio/adc/ade9000.c | 36 +++++++------ + drivers/iio/adc/adi-axi-adc.c | 4 ++ + drivers/iio/adc/aspeed_adc.c | 4 +- + drivers/iio/adc/axp288_adc.c | 8 +++ + drivers/iio/adc/max1363.c | 8 +++ + drivers/iio/adc/pac1934.c | 4 ++ + drivers/iio/adc/rohm-bd79124.c | 20 ++++--- + drivers/iio/adc/stm32-adc.c | 47 +++++++++------- + drivers/iio/adc/sun4i-gpadc-iio.c | 10 ++-- + drivers/iio/adc/xilinx-xadc-core.c | 9 ++-- + drivers/iio/buffer/industrialio-buffer-dmaengine.c | 16 ++++-- + drivers/iio/cdc/ad7150.c | 8 +-- + .../iio/common/inv_sensors/inv_sensors_timestamp.c | 11 ++-- + drivers/iio/dac/mcp47a1.c | 4 +- + drivers/iio/dac/rohm-bd79703.c | 3 ++ + drivers/iio/frequency/adf4377.c | 4 +- + drivers/iio/frequency/admv1013.c | 9 ++-- + drivers/iio/gyro/adis16136.c | 2 +- + drivers/iio/health/max30102.c | 12 +++-- + drivers/iio/imu/adis16400.c | 2 +- + drivers/iio/imu/adis16480.c | 4 +- + drivers/iio/imu/inv_icm42607/inv_icm42607_core.c | 38 +++++++++---- + drivers/iio/industrialio-buffer.c | 7 ++- + drivers/iio/industrialio-trigger.c | 2 + + drivers/iio/light/gp2ap020a00f.c | 2 + + drivers/iio/light/rohm-bu27034.c | 62 +++++++++++++--------- + drivers/iio/pressure/bmp280-core.c | 2 +- + drivers/iio/pressure/rohm-bm1390.c | 2 +- + drivers/iio/proximity/aw96103.c | 30 ++++++++--- + drivers/iio/proximity/isl29501.c | 2 +- + drivers/iio/proximity/pulsedlight-lidar-lite-v2.c | 4 +- + drivers/iio/proximity/sx9324.c | 2 +- + drivers/iio/proximity/vcnl3020.c | 24 ++++++--- + drivers/iio/proximity/vl53l0x-i2c.c | 3 ++ + drivers/virt/nitro_enclaves/ne_misc_dev.c | 1 + + include/linux/hid-sensor-hub.h | 2 + + 50 files changed, 414 insertions(+), 174 deletions(-) +Merging watchdog-fixes/watchdog (2686e13649a9e watchdog: mediatek: Apply the driver's mode to a watchdog left running) +$ git merge -m Merge branch 'watchdog' of https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git watchdog-fixes/watchdog +Merge made by the 'ort' strategy. + drivers/watchdog/mtk_wdt.c | 43 ++++++++++++++++++++++++------------------- + 1 file changed, 24 insertions(+), 19 deletions(-) +Merging counter-current/counter-current (fd73f4a665989 Linux 7.3-rc3) +$ git merge -m Merge branch 'counter-current' of https://git.kernel.org/pub/scm/linux/kernel/git/wbg/counter.git counter-current/counter-current +Already up to date. +Merging char-misc.current/char-misc-linus (540f55de8c9f7 Merge tag 'icc-7.3-rc5' of ssh://gitolite.kernel.org/pub/scm/linux/kernel/git/djakov/icc into char-misc-linus) +$ git merge -m Merge branch 'char-misc-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git char-misc.current/char-misc-linus +Merge made by the 'ort' strategy. + drivers/interconnect/qcom/x1e80100.c | 485 ----------------------------------- + 1 file changed, 485 deletions(-) +Merging soundwire-fixes/fixes (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/soundwire.git soundwire-fixes/fixes +Already up to date. +Merging thunderbolt-fixes/fixes (395e9f2967a7a thunderbolt: Disable CL states for the Anker Prime TB5 dock) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt.git thunderbolt-fixes/fixes +Auto-merging drivers/thunderbolt/stream.c +Merge made by the 'ort' strategy. + drivers/thunderbolt/domain.c | 26 +------------------------- + drivers/thunderbolt/nhi.c | 26 -------------------------- + drivers/thunderbolt/nhi.h | 21 ++------------------- + drivers/thunderbolt/nhi_regs.h | 4 ---- + drivers/thunderbolt/pci.c | 23 ----------------------- + drivers/thunderbolt/quirks.c | 3 +++ + drivers/thunderbolt/stream.c | 3 +++ + drivers/thunderbolt/xdomain.c | 4 ++++ + 8 files changed, 13 insertions(+), 97 deletions(-) +Merging input-current/for-linus (309731e959171 Input: hp_sdc - shut down kicker timer on module exit) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/dtor/input.git input-current/for-linus +Already up to date. +Merging crypto-current/master (10396a2d6d41d crypto: s390/hmac - Generate intermediate CV for API partial block handling) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/herbert/crypto-2.6.git crypto-current/master +Already up to date. +Merging libcrypto-fixes/libcrypto-fixes (6d996c3b974c5 crypto: aes - Fix undesired override of some optimized AES modes) +$ git merge -m Merge branch 'libcrypto-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git libcrypto-fixes/libcrypto-fixes +Merge made by the 'ort' strategy. + arch/x86/crypto/aesni-intel_glue.c | 2 +- + crypto/aes.c | 51 +++++++++++++++++++++++++++++++++++--- + 2 files changed, 48 insertions(+), 5 deletions(-) +Merging vfio-fixes/for-linus (e242e974e812e vfio: selftests: Add luuid to libvfio.mk's list of libraries, not to the Makefile) +$ git merge -m Merge branch 'for-linus' of https://github.com/awilliam/linux-vfio.git vfio-fixes/for-linus +Already up to date. +Merging kselftest-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kselftest-fixes/fixes +Already up to date. +Merging dmaengine-fixes/fixes (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git dmaengine-fixes/fixes +Already up to date. +Merging backlight-fixes/for-backlight-fixes (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'for-backlight-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/backlight.git backlight-fixes/for-backlight-fixes +Already up to date. +Merging mtd-fixes/mtd/fixes (1c1a342aceec7 mtd: spinand: Do not update the QE bit on devices without one) +$ git merge -m Merge branch 'mtd/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git mtd-fixes/mtd/fixes +Already up to date. +Merging mfd-fixes/for-mfd-fixes (d5d2d7a8d8be1 MAINTAINERS: Add a mailing list entry to MFD) +$ git merge -m Merge branch 'for-mfd-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/mfd.git mfd-fixes/for-mfd-fixes +Already up to date. +Merging v4l-dvb-fixes/fixes (2579cbe68005f media: em28xx: use video_unregister_device for radio_dev) +$ git merge -m Merge branch 'fixes' of git://linuxtv.org/media-ci/media-pending.git v4l-dvb-fixes/fixes +Merge made by the 'ort' strategy. + drivers/media/usb/em28xx/em28xx-video.c | 4 ++-- + 1 file changed, 2 insertions(+), 2 deletions(-) +Merging reset-fixes/reset/fixes (71827776667f4 reset: imx7: Correct polarity of MIPI CSI resets on i.MX8MQ) +$ git merge -m Merge branch 'reset/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/pza/linux reset-fixes/reset/fixes +Already up to date. +Merging mips-fixes/mips-fixes (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'mips-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mips/linux.git mips-fixes/mips-fixes +Already up to date. +Merging at91-fixes/at91-fixes (afb1ecfeda3fc ARM: configs: sama5: enable current Microchip KSZ DSA symbols) +$ git merge -m Merge branch 'at91-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux.git at91-fixes/at91-fixes +Merge made by the 'ort' strategy. + arch/arm/configs/sama5_defconfig | 4 ++-- + 1 file changed, 2 insertions(+), 2 deletions(-) +Merging omap-fixes/fixes (2fabd2f406d0c ARM: dts: ti/omap: dra7: fix PCIe PHY clock divider definition) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git omap-fixes/fixes +Merge made by the 'ort' strategy. + arch/arm/boot/dts/ti/omap/dra7xx-clocks.dtsi | 2 +- + 1 file changed, 1 insertion(+), 1 deletion(-) +Merging tegra-fixes/fixes (5257486664997 Merge branch for-7.3/arm64/dt into fixes) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux.git tegra-fixes/fixes +Merge made by the 'ort' strategy. + .../boot/dts/nvidia/tegra264-p4071-0000+p3834.dtsi | 13 ++ + arch/arm64/boot/dts/nvidia/tegra264.dtsi | 180 ++++++++++++++++++--- + drivers/soc/tegra/fuse/tegra-apbmisc.c | 2 +- + drivers/soc/tegra/pmc.c | 1 + + 4 files changed, 171 insertions(+), 25 deletions(-) +Merging kvm-fixes/master (973ea70393e88 KVM: SEV: Do cache maintenance on the source VM *before* clearing SEV state) +$ git merge -m Merge branch 'master' of git://git.kernel.org/pub/scm/virt/kvm/kvm.git kvm-fixes/master +Merge made by the 'ort' strategy. + arch/arm64/kvm/vgic/vgic-init.c | 12 +- + arch/powerpc/kvm/book3s_hv.c | 2 - + arch/riscv/kvm/aia_device.c | 2 +- + arch/s390/kvm/s390/s390.c | 5 +- + arch/x86/kvm/mmu/mmu.c | 6 + + arch/x86/kvm/svm/sev.c | 57 +++--- + arch/x86/kvm/svm/svm.c | 43 ++-- + arch/x86/kvm/svm/svm.h | 1 + + arch/x86/kvm/vmx/nested.c | 6 + + arch/x86/kvm/vmx/tdx.c | 5 - + include/linux/kvm_host.h | 1 - + tools/testing/selftests/kvm/Makefile.kvm | 1 + + .../testing/selftests/kvm/x86/nested_x2apic_test.c | 222 +++++++++++++++++++++ + virt/kvm/kvm_main.c | 35 +--- + 14 files changed, 302 insertions(+), 96 deletions(-) + create mode 100644 tools/testing/selftests/kvm/x86/nested_x2apic_test.c +Merging kvms390-fixes/master (f47190b08b71e s390/uv: Prevent potential out-of-bounds read) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/kvms390/linux.git kvms390-fixes/master +Already up to date. +Merging kvm-arm-fixes/fixes (afb4334fb52c7 KVM: arm64: selftests: Check a feature hidden in an ID register is UNDEF) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git kvm-arm-fixes/fixes +Auto-merging tools/testing/selftests/kvm/Makefile.kvm +Merge made by the 'ort' strategy. + arch/arm64/kvm/emulate-nested.c | 3 - + arch/arm64/kvm/hyp/include/nvhe/pkvm.h | 9 + + arch/arm64/kvm/hyp/nvhe/hyp-main.c | 7 +- + arch/arm64/kvm/hyp/nvhe/pkvm.c | 25 ++- + arch/arm64/kvm/vgic/vgic-its.c | 12 +- + arch/arm64/kvm/vgic/vgic-v2.c | 6 +- + arch/arm64/kvm/vgic/vgic-v3.c | 6 +- + arch/arm64/kvm/vgic/vgic.c | 21 ++- + tools/testing/selftests/kvm/Makefile.kvm | 1 + + .../testing/selftests/kvm/arm64/hidden_features.c | 184 +++++++++++++++++++++ + 10 files changed, 246 insertions(+), 28 deletions(-) + create mode 100644 tools/testing/selftests/kvm/arm64/hidden_features.c +Merging hwmon-fixes/hwmon (61406e9cac695 hwmon: (sht4x) Fix jiffies wraparound in heater-ready check) +$ git merge -m Merge branch 'hwmon' of https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git hwmon-fixes/hwmon +Merge made by the 'ort' strategy. + Documentation/hwmon/k10temp.rst | 8 +++----- + drivers/hwmon/coretemp.c | 5 +++++ + drivers/hwmon/k10temp.c | 5 ++--- + drivers/hwmon/lm70.c | 8 ++++---- + drivers/hwmon/lm95245.c | 6 +++--- + drivers/hwmon/sht4x.c | 18 +++++++++--------- + drivers/hwmon/tmp102.c | 8 ++++---- + drivers/hwmon/tmp108.c | 8 ++++---- + 8 files changed, 34 insertions(+), 32 deletions(-) +Merging nvdimm-fixes/libnvdimm-fixes (a8aec14230322 nvdimm/bus: Fix potential use after free in asynchronous initialization) +$ git merge -m Merge branch 'libnvdimm-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/nvdimm/nvdimm.git nvdimm-fixes/libnvdimm-fixes +Already up to date. +Merging cxl-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/cxl/cxl.git cxl-fixes/fixes +Already up to date. +Merging dma-mapping-fixes/dma-mapping-fixes (057e5e07420c7 iommu/dma: skip swiotlb bounce for DMA_ATTR_MMIO in iommu_dma_map_phys) +$ git merge -m Merge branch 'dma-mapping-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mszyprowski/linux.git dma-mapping-fixes/dma-mapping-fixes +Merge made by the 'ort' strategy. + drivers/iommu/dma-iommu.c | 4 ++-- + 1 file changed, 2 insertions(+), 2 deletions(-) +Merging drivers-x86-fixes/fixes (d144a494d81fc MAINTAINERS: fix sysfs-platform-ayaneo-ec documentation path) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/pdx86/platform-drivers-x86.git drivers-x86-fixes/fixes +Already up to date. +Merging samsung-krzk-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux.git samsung-krzk-fixes/fixes +Already up to date. +Merging pinctrl-samsung-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/samsung.git pinctrl-samsung-fixes/fixes +Already up to date. +Merging pinctrl-qcom-fixes/pinctrl-qcom/for-current (19fc4240358be pinctrl: qcom: spmi-gpio: make direction changes exclusive) +$ git merge -m Merge branch 'pinctrl-qcom/for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git pinctrl-qcom-fixes/pinctrl-qcom/for-current +Merge made by the 'ort' strategy. + drivers/pinctrl/qcom/pinctrl-hawi.c | 1 + + drivers/pinctrl/qcom/pinctrl-ipq5018.c | 4 ++-- + drivers/pinctrl/qcom/pinctrl-maili.c | 1 + + drivers/pinctrl/qcom/pinctrl-nord.c | 1 + + drivers/pinctrl/qcom/pinctrl-qcs8300.c | 1 + + drivers/pinctrl/qcom/pinctrl-spmi-gpio.c | 37 +++++++++++++++++++++++++------- + 6 files changed, 35 insertions(+), 10 deletions(-) +Merging devicetree-fixes/dt/linus (21c89ff1fc86f of/overlay: don't create "//" paths for fragments targeting the root) +$ git merge -m Merge branch 'dt/linus' of https://git.kernel.org/pub/scm/linux/kernel/git/robh/linux.git devicetree-fixes/dt/linus +Merge made by the 'ort' strategy. + drivers/of/overlay.c | 17 +++++++++++++---- + 1 file changed, 13 insertions(+), 4 deletions(-) +Merging dt-krzk-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt.git dt-krzk-fixes/fixes +Already up to date. +Merging scsi-fixes/fixes (42d1221d321e5 scsi: megaraid_sas: Protect megasas_get_ctrl_info() in megasas_resume()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git scsi-fixes/fixes +Already up to date. +Merging drm-fixes/drm-fixes (a9ed3aa9b87ee Merge tag 'drm-misc-fixes-2026-09-24' of https://gitlab.freedesktop.org/drm/misc/kernel into drm-fixes) +$ git merge -m Merge branch 'drm-fixes' of https://gitlab.freedesktop.org/drm/kernel.git drm-fixes/drm-fixes +Already up to date. +Merging drm-intel-fixes/for-linux-next-fixes (c034e8a46e4cb drm/i915/vrr: Disable DC balance by default) +$ git merge -m Merge branch 'for-linux-next-fixes' of https://gitlab.freedesktop.org/drm/i915/kernel.git drm-intel-fixes/for-linux-next-fixes +Merge made by the 'ort' strategy. + drivers/gpu/drm/i915/display/intel_display_params.c | 4 ++++ + drivers/gpu/drm/i915/display/intel_display_params.h | 1 + + drivers/gpu/drm/i915/display/intel_vrr.c | 4 +++- + 3 files changed, 8 insertions(+), 1 deletion(-) +Merging mmc-fixes/fixes (aa0a37b5024d9 memstick: core: wait for request completion before freeing card) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/mmc.git mmc-fixes/fixes +Merge made by the 'ort' strategy. + drivers/memstick/core/memstick.c | 8 ++------ + drivers/mmc/host/cavium-octeon.c | 5 ++++- + drivers/mmc/host/cavium-thunderx.c | 5 ++++- + drivers/mmc/host/mtk-sd.c | 1 + + drivers/mmc/host/sdhci-sprd.c | 4 ++++ + 5 files changed, 15 insertions(+), 8 deletions(-) +Merging rtc-fixes/rtc-fixes (055ef5ce9f67a rtc: spear: initialize IRQ state before requesting alarm IRQ) +$ git merge -m Merge branch 'rtc-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux.git rtc-fixes/rtc-fixes +Already up to date. +Merging gnss-fixes/gnss-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'gnss-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/johan/gnss.git gnss-fixes/gnss-linus +Already up to date. +Merging hyperv-fixes/hyperv-fixes (ca039df94983c PCI: hv: Warn when wait_for_response() waits indefinitely) +$ git merge -m Merge branch 'hyperv-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux.git hyperv-fixes/hyperv-fixes +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 2 -- + arch/x86/hyperv/hv_init.c | 7 +++--- + drivers/hv/channel.c | 2 +- + drivers/hv/channel_mgmt.c | 12 ++++++--- + drivers/hv/connection.c | 1 - + drivers/hv/mshv_vtl_main.c | 3 +++ + drivers/hv/vmbus_drv.c | 19 ++++++-------- + drivers/pci/controller/pci-hyperv.c | 49 +++++++++++++++++++++++++++++-------- + tools/hv/hv_get_dns_info.sh | 2 +- + 9 files changed, 64 insertions(+), 33 deletions(-) +Merging risc-v-fixes/fixes (e111609483333 riscv: vector: Fix output operands in context save) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/riscv/linux.git risc-v-fixes/fixes +Merge made by the 'ort' strategy. + arch/riscv/Kconfig.errata | 2 +- + arch/riscv/Kconfig.socs | 1 + + arch/riscv/include/asm/vector.h | 8 ++++---- + arch/riscv/include/uapi/asm/vendor/mips.h | 4 +++- + 4 files changed, 9 insertions(+), 6 deletions(-) +Merging riscv-dt-fixes/riscv-dt-fixes (0f70fd6f4c1d6 riscv: dts: microchip: beaglev-fire: remove double definition of gpio interrupts) +$ git merge -m Merge branch 'riscv-dt-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git riscv-dt-fixes/riscv-dt-fixes +Merge made by the 'ort' strategy. + arch/riscv/boot/dts/microchip/mpfs-beaglev-fire.dts | 18 ------------------ + 1 file changed, 18 deletions(-) +Merging riscv-soc-fixes/riscv-soc-fixes (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'riscv-soc-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git riscv-soc-fixes/riscv-soc-fixes +Already up to date. +Merging fpga-fixes/fixes (19272b37aa4f8 Linux 6.16-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/fpga/linux-fpga.git fpga-fixes/fixes +Already up to date. +Merging spdx/spdx-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'spdx-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/spdx.git spdx/spdx-linus +Already up to date. +Merging gpio-brgl-fixes/gpio/for-current (ff82fc3a4a1d4 gpio: xilinx: fix runtime PM leak on request error path) +$ git merge -m Merge branch 'gpio/for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git gpio-brgl-fixes/gpio/for-current +Merge made by the 'ort' strategy. + drivers/gpio/gpio-xilinx.c | 9 +-------- + 1 file changed, 1 insertion(+), 8 deletions(-) +Merging gpio-intel-fixes/fixes (8d37f20173a27 gpiolib: acpi: Add quirk for Lenovo IdeaPad Slim 3 15ABR8) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-gpio-intel.git gpio-intel-fixes/fixes +Merge made by the 'ort' strategy. + drivers/gpio/gpio-crystalcove.c | 1 + + drivers/gpio/gpiolib-acpi-quirks.c | 13 +++++++++++++ + 2 files changed, 14 insertions(+) +Merging pinctrl-intel-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/intel.git pinctrl-intel-fixes/fixes +Already up to date. +Merging auxdisplay-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-auxdisplay.git auxdisplay-fixes/fixes +Already up to date. +Merging kunit-fixes/kunit-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'kunit-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kunit-fixes/kunit-fixes +Already up to date. +Merging renesas-fixes/fixes (8dc2615d57020 arm64: dts: renesas: r8a779f0: Set UFS lane count) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel.git renesas-fixes/fixes +Already up to date. +Merging perf-current/perf-tools (aadea57f53288 perf powerpc-vpadtl: Fix raw_size of DTL samples) +$ git merge -m Merge branch 'perf-tools' of https://git.kernel.org/pub/scm/linux/kernel/git/perf/perf-tools.git perf-current/perf-tools +Already up to date. +Merging efi-fixes/urgent (d8809f6931065 efi: sysfb_efi: Extend quirk to cover IdeaPad Duet 3 10IGL5-LTE) +$ git merge -m Merge branch 'urgent' of https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git efi-fixes/urgent +Already up to date. +Merging battery-fixes/fixes (a58cbc8b36ec0 power: supply: ds2780: Fix ds2780_get_capacity() returning raw value) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-power-supply.git battery-fixes/fixes +Merge made by the 'ort' strategy. + drivers/power/supply/ds2780_battery.c | 2 +- + drivers/power/supply/max17042_battery.c | 2 +- + include/linux/power_supply.h | 6 +++--- + 3 files changed, 5 insertions(+), 5 deletions(-) +Merging iommufd-fixes/for-rc (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-rc' of https://git.kernel.org/pub/scm/linux/kernel/git/jgg/iommufd.git iommufd-fixes/for-rc +Already up to date. +Merging rust-fixes/rust-fixes (551c722f40809 Merge tag 'rtc-7.3-fixes' of git://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux) +$ git merge -m Merge branch 'rust-fixes' of https://github.com/Rust-for-Linux/linux.git rust-fixes/rust-fixes +Already up to date. +Merging w1-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-w1.git w1-fixes/fixes +Already up to date. +Merging pmdomain-fixes/fixes (4c66c423593e2 pmdomain: rockchip: don't ignore clock lookup errors on attach) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/linux-pm.git pmdomain-fixes/fixes +Merge made by the 'ort' strategy. + drivers/pmdomain/imx/imx8m-blk-ctrl.c | 15 +++++++++++++++ + drivers/pmdomain/rockchip/pm-domains.c | 20 +++++++++++++++++--- + 2 files changed, 32 insertions(+), 3 deletions(-) +Merging i2c-andi-fixes/i2c/i2c-fixes (840d8acc87925 i2c: xiic: don't clobber msg->len to signal block-read completion) +$ git merge -m Merge branch 'i2c/i2c-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux.git i2c-andi-fixes/i2c/i2c-fixes +Merge made by the 'ort' strategy. + drivers/i2c/busses/i2c-xiic.c | 61 ++++++++++++++++++++++++++++++++----------- + 1 file changed, 46 insertions(+), 15 deletions(-) +Merging i2c-rust-fixes/rust-i2c-fixes (4eb422482ca5d rust: i2c: fix I2cAdapter refcounts double increment) +$ git merge -m Merge branch 'rust-i2c-fixes' of https://github.com/ikrtn/rust-for-linux i2c-rust-fixes/rust-i2c-fixes +Already up to date. +Merging sparc-fixes/for-linus (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-linus' of https://git.kernel.org/pub/scm/linux/kernel/git/alarsson/linux-sparc.git sparc-fixes/for-linus +Already up to date. +Merging clk-fixes/clk-fixes (81493c1dd1b1e Merge tag 'tags/spacemit-clk-fixes-for-7.3-1' into clk-fixes) +$ git merge -m Merge branch 'clk-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/clk/linux.git clk-fixes/clk-fixes +Merge made by the 'ort' strategy. + drivers/clk/spacemit/ccu-k3.c | 92 +++++++++++++++++++++++++++++++++++++++++++ + drivers/clk/ti/composite.c | 26 ++++++------ + 2 files changed, 104 insertions(+), 14 deletions(-) +Merging thead-clk-fixes/thead-clk-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'thead-clk-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git thead-clk-fixes/thead-clk-fixes +Already up to date. +Merging tenstorrent-clk-fixes/tenstorrent-clk-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'tenstorrent-clk-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git tenstorrent-clk-fixes/tenstorrent-clk-fixes +Already up to date. +Merging fustini-config-fixes/riscv-config-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'riscv-config-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git fustini-config-fixes/riscv-config-fixes +Already up to date. +Merging pwrseq-fixes/pwrseq/for-current (58a0033024815 power: sequencing: pcie-m2: Add Lenovo ThinkPad T14s gen6 WCN7850 subsystem PCI ids) +$ git merge -m Merge branch 'pwrseq/for-current' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git pwrseq-fixes/pwrseq/for-current +Merge made by the 'ort' strategy. + drivers/power/sequencing/pwrseq-pcie-m2.c | 2 ++ + 1 file changed, 2 insertions(+) +Merging thead-dt-fixes/thead-dt-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'thead-dt-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git thead-dt-fixes/thead-dt-fixes +Already up to date. +Merging ftrace-fixes/ftrace/fixes (1650a1b6cb1ae fgraph: Check ftrace_pids_enabled on registration for early filtering) +$ git merge -m Merge branch 'ftrace/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git ftrace-fixes/ftrace/fixes +Already up to date. +Merging ring-buffer-fixes/ring-buffer/fixes (057caace5214d tracing: Create output file from cmd_check_undefined) +$ git merge -m Merge branch 'ring-buffer/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git ring-buffer-fixes/ring-buffer/fixes +Already up to date. +Merging trace-fixes/trace/fixes (d860c67c05168 ring-buffer: Check resize_disabled before publishing the new subbuf order) +$ git merge -m Merge branch 'trace/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git trace-fixes/trace/fixes +Already up to date. +Merging tracefs-fixes/tracefs/fixes (07004a8c4b572 eventfs: Hold eventfs_mutex and SRCU when remount walks events) +$ git merge -m Merge branch 'tracefs/fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git tracefs-fixes/tracefs/fixes +Already up to date. +Merging spacemit-fixes/fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux spacemit-fixes/fixes +Already up to date. +Merging tip-fixes/tip/urgent (3ebb3531c6d0d Merge branch into tip/master: 'x86/mm') +$ git merge -m Merge branch 'tip/urgent' of https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git tip-fixes/tip/urgent +Merge made by the 'ort' strategy. + arch/x86/include/asm/pgtable.h | 6 +++++- + arch/x86/include/uapi/asm/ptrace-abi.h | 4 ++-- + arch/x86/kernel/sys_x86_64.c | 15 ++++++++++---- + arch/x86/mm/pat/set_memory.c | 23 ++++++++++++++++---- + arch/x86/mm/pgtable.c | 38 +++++++++++++--------------------- + arch/x86/um/asm/ptrace.h | 4 +--- + arch/x86/um/ptrace_32.c | 1 + + include/linux/hrtimer_rearm.h | 2 +- + 8 files changed, 54 insertions(+), 39 deletions(-) +Merging kexec-fixes/kexec-fixes (a901b0778ae82 Merge patch series "kexec: fix probe error codes and error propagation") +$ git merge -m Merge branch 'kexec-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git kexec-fixes/kexec-fixes +Merge made by the 'ort' strategy. + arch/arm64/kernel/kexec_image.c | 4 ++-- + arch/loongarch/kernel/kexec_efi.c | 4 ++-- + arch/riscv/kernel/kexec_image.c | 4 ++-- + kernel/kexec_elf.c | 4 ++-- + kernel/kexec_file.c | 12 +++++++----- + 5 files changed, 15 insertions(+), 13 deletions(-) +Merging liveupdate-fixes/fixes (3a0b8fa2eb36a kho: fix size calculation in kho_preserved_memory_reserve()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git liveupdate-fixes/fixes +Already up to date. +Merging drm-msm-fixes/msm-fixes (a15fac810c763 dt-bindings: display/msm: Use consistent indentation in the example) +$ git merge -m Merge branch 'msm-fixes' of https://gitlab.freedesktop.org/drm/msm.git drm-msm-fixes/msm-fixes +Already up to date. +Merging uml-fixes/fixes (af421e9aed392 um: vector: fix use-after-free in vector_mmsg_rx()) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git uml-fixes/fixes +Already up to date. +Merging fwctl-fixes/for-rc (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-rc' of https://git.kernel.org/pub/scm/linux/kernel/git/fwctl/fwctl.git fwctl-fixes/for-rc +Already up to date. +Merging devsec-tsm-fixes/fixes (c3fd16c3b98ed virt: tdx-guest: Fix handling of host controlled 'quote' buffer length) +$ git merge -m Merge branch 'fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/devsec/tsm.git devsec-tsm-fixes/fixes +Already up to date. +Merging drm-rust-fixes/for-linux-next-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-linux-next-fixes' of https://gitlab.freedesktop.org/drm/rust/kernel.git drm-rust-fixes/for-linux-next-fixes +Already up to date. +Merging tenstorrent-dt-fixes/tenstorrent-dt-fixes (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'tenstorrent-dt-fixes' of https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git tenstorrent-dt-fixes/tenstorrent-dt-fixes +Already up to date. +Merging nfc-fixes/for-linus (b61732f47316d nfc: pn533: fix OOB read in pn533_acr122_is_rx_frame_valid()) +$ git merge -m Merge branch 'for-linus' of https://codeberg.org/linux-nfc/linux.git nfc-fixes/for-linus +Already up to date. +Merging mm-nonmm-hotfixes-stable/mm-nonmm-hotfixes-stable (a243ede718463 Merge tag 'mtd/fixes-for-7.3-rc6' of git://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux) +$ git merge -m Merge branch 'mm-nonmm-hotfixes-stable' of https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm mm-nonmm-hotfixes-stable/mm-nonmm-hotfixes-stable +Already up to date. +Merging mm-nonmm-hotfixes-unstable/mm-nonmm-hotfixes-unstable (844c67369f72d fault-inject: fix dentry leak) +$ git merge -m Merge branch 'mm-nonmm-hotfixes-unstable' of https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm mm-nonmm-hotfixes-unstable/mm-nonmm-hotfixes-unstable +Merge made by the 'ort' strategy. + drivers/infiniband/hw/hfi1/fault.c | 1 - + fs/fat/fat.h | 2 +- + fs/fat/fatent.c | 21 ++++++++++++++++++--- + fs/fat/file.c | 3 ++- + fs/fat/misc.c | 6 ++---- + include/linux/fault-inject.h | 10 ++++++++-- + kernel/resource.c | 2 +- + lib/fault-inject.c | 7 +++++-- + 8 files changed, 37 insertions(+), 15 deletions(-) +Merging drm-misc-fixes/for-linux-next-fixes (e78a9fb7a40c5 drm/vmwgfx: Add blend mode property) +$ git merge -m Merge branch 'for-linux-next-fixes' of https://gitlab.freedesktop.org/drm/misc/kernel.git drm-misc-fixes/for-linux-next-fixes +Merge made by the 'ort' strategy. + drivers/gpu/drm/bridge/th1520-dw-hdmi.c | 8 ++++---- + drivers/gpu/drm/vc4/vc4_drv.c | 5 ++++- + drivers/gpu/drm/vc4/vc4_v3d.c | 8 ++++++++ + drivers/gpu/drm/vmwgfx/vmwgfx_cursor_plane.c | 26 +++++++++++++++++++++----- + drivers/gpu/drm/vmwgfx/vmwgfx_kms.c | 8 ++++++++ + 5 files changed, 45 insertions(+), 10 deletions(-) +Merging rust/rust-next (72d3fcf802c45 Linux 7.3-rc5) +$ git merge -m Merge branch 'rust-next' of https://github.com/Rust-for-Linux/linux.git rust/rust-next +Already up to date. +Merging rust-interop/interop-next (05f7e89ab9731 Linux 6.19) +$ git merge -m Merge branch 'interop-next' of https://github.com/Rust-for-Linux/linux.git rust-interop/interop-next +Already up to date. +Merging rust-alloc/alloc-next (6e8339118040f rust: alloc: add `NumaNode::id()` accessor) +$ git merge -m Merge branch 'alloc-next' of https://github.com/Rust-for-Linux/linux.git rust-alloc/alloc-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 1 + + rust/kernel/alloc.rs | 6 ++++++ + rust/kernel/alloc/kbox.rs | 7 ++++--- + 3 files changed, 11 insertions(+), 3 deletions(-) +Merging rust-io/io-next (86731a2a651e5 Linux 6.16-rc3) +$ git merge -m Merge branch 'io-next' of https://github.com/Rust-for-Linux/linux.git rust-io/io-next +Already up to date. +Merging rust-pin-init/pin-init-next (95593ed361619 rust: pin-init: internal: improve diagnostics robustness against panicking) +$ git merge -m Merge branch 'pin-init-next' of https://github.com/Rust-for-Linux/linux.git rust-pin-init/pin-init-next +Merge made by the 'ort' strategy. + rust/pin-init/internal/src/diagnostics.rs | 169 +++++++-- + rust/pin-init/internal/src/init.rs | 546 ++++++++++++++++++++++++------ + rust/pin-init/internal/src/lib.rs | 40 ++- + rust/pin-init/internal/src/pin_data.rs | 164 +++++---- + rust/pin-init/internal/src/util.rs | 54 +++ + rust/pin-init/src/__internal.rs | 64 ++-- + rust/pin-init/src/lib.rs | 82 ++++- + 7 files changed, 850 insertions(+), 269 deletions(-) + create mode 100644 rust/pin-init/internal/src/util.rs +Merging rust-timekeeping/timekeeping-next (2ea0119f72dba rust: hrtimer: document handle based design rationale) +$ git merge -m Merge branch 'timekeeping-next' of https://github.com/Rust-for-Linux/linux.git rust-timekeeping/timekeeping-next +Auto-merging rust/bindings/lib.rs +Merge made by the 'ort' strategy. + rust/bindings/lib.rs | 5 ++ + rust/helpers/helpers.c | 1 + + rust/helpers/math.c | 8 +++ + rust/kernel/Kconfig.test | 10 ++++ + rust/kernel/time.rs | 139 ++++++++++++++++++++++++++++++++++++++------ + rust/kernel/time/delay.rs | 17 ++++-- + rust/kernel/time/hrtimer.rs | 93 ++++++++++++++++------------- + 7 files changed, 210 insertions(+), 63 deletions(-) + create mode 100644 rust/helpers/math.c +Merging rust-xarray/xarray-next (c455f19bbe610 rust: xarray: add __rust_helper to helpers) +$ git merge -m Merge branch 'xarray-next' of https://github.com/Rust-for-Linux/linux.git rust-xarray/xarray-next +Already up to date. +Merging rust-analyzer/rust-analyzer-next (5f45afb8ab04d scripts: generate_rust_analyzer.py: pass cfg to macros crate) +$ git merge -m Merge branch 'rust-analyzer-next' of https://github.com/Rust-for-Linux/linux.git rust-analyzer/rust-analyzer-next +Auto-merging scripts/generate_rust_analyzer.py +Merge made by the 'ort' strategy. + scripts/generate_rust_analyzer.py | 1 + + 1 file changed, 1 insertion(+) +Merging mm/for-next (7a6c9825d917f Merge https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm.git mm-unstable into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mm/linux.git mm/for-next +Auto-merging MAINTAINERS +Auto-merging arch/arm64/mm/mmu.c +Auto-merging arch/x86/include/asm/pgtable.h +Auto-merging arch/x86/mm/pat/set_memory.c +Auto-merging drivers/usb/cdns3/cdns3-gadget.c +Merge made by the 'ort' strategy. + .clang-format | 2 +- + Documentation/ABI/testing/sysfs-kernel-mm-damon | 29 +- + Documentation/admin-guide/cgroup-v2.rst | 4 + + Documentation/admin-guide/kdump/vmcoreinfo.rst | 2 +- + Documentation/admin-guide/kernel-parameters.txt | 3 + + Documentation/admin-guide/mm/damon/usage.rst | 26 +- + Documentation/admin-guide/mm/ksm.rst | 8 +- + Documentation/arch/powerpc/vmemmap_dedup.rst | 88 +- + Documentation/core-api/memory-allocation.rst | 32 +- + Documentation/core-api/mm-api.rst | 9 + + Documentation/core-api/pin_user_pages.rst | 9 - + Documentation/dev-tools/kmemleak.rst | 24 +- + Documentation/filesystems/mmap_prepare.rst | 81 ++ + Documentation/filesystems/vfs.rst | 6 +- + Documentation/mm/damon/design.rst | 49 +- + Documentation/mm/damon/maintainer-profile.rst | 19 +- + Documentation/mm/index.rst | 1 + + Documentation/mm/kernel-page-tables.rst | 410 ++++++++++ + Documentation/mm/ksm.rst | 2 +- + Documentation/mm/page_owner.rst | 8 +- + Documentation/mm/physical_memory.rst | 6 + + Documentation/mm/process_addrs.rst | 11 + + Documentation/mm/vmemmap_dedup.rst | 32 +- + MAINTAINERS | 12 +- + arch/Kconfig | 8 - + arch/alpha/Kconfig | 1 - + arch/alpha/include/asm/pgtable.h | 7 - + arch/arc/include/asm/pgalloc.h | 6 +- + arch/arc/include/asm/pgtable-levels.h | 11 - + arch/arm/Kconfig | 2 - + arch/arm/Kconfig.debug | 2 +- + arch/arm/configs/aspeed_g4_defconfig | 2 +- + arch/arm/configs/aspeed_g5_defconfig | 2 +- + arch/arm/configs/shmobile_defconfig | 2 +- + arch/arm/include/asm/pgtable.h | 7 - + arch/arm/include/asm/ptdump.h | 6 +- + arch/arm/kernel/traps.c | 17 - + arch/arm/mm/init.c | 2 +- + arch/arm64/Kconfig | 4 +- + arch/arm64/include/asm/mte.h | 2 - + arch/arm64/include/asm/pgtable.h | 64 +- + arch/arm64/include/asm/set_memory.h | 5 +- + arch/arm64/kvm/mmu.c | 4 +- + arch/arm64/mm/fault.c | 16 +- + arch/arm64/mm/mmu.c | 48 ++ + arch/arm64/mm/mteswap.c | 32 +- + arch/arm64/mm/pageattr.c | 24 +- + arch/csky/include/asm/pgtable.h | 4 - + arch/hexagon/include/asm/pgtable.h | 3 - + arch/loongarch/Kconfig | 2 - + arch/loongarch/include/asm/pgtable.h | 14 +- + arch/loongarch/include/asm/set_memory.h | 5 +- + arch/loongarch/mm/init.c | 4 +- + arch/loongarch/mm/pageattr.c | 27 +- + arch/m68k/Kconfig | 1 + + arch/m68k/include/asm/mcf_pgalloc.h | 5 +- + arch/m68k/include/asm/mcf_pgtable.h | 6 - + arch/m68k/include/asm/motorola_pgalloc.h | 9 +- + arch/m68k/include/asm/motorola_pgtable.h | 8 - + arch/m68k/include/asm/sun3_pgtable.h | 7 - + arch/m68k/mm/motorola.c | 119 ++- + arch/microblaze/include/asm/pgalloc.h | 2 +- + arch/microblaze/include/asm/pgtable.h | 7 - + arch/mips/Kconfig | 1 - + arch/mips/include/asm/pgtable-32.h | 10 - + arch/mips/include/asm/pgtable-64.h | 13 - + arch/nios2/include/asm/pgtable.h | 7 - + arch/openrisc/include/asm/pgtable.h | 7 - + arch/parisc/Kconfig | 1 - + arch/parisc/include/asm/pgtable.h | 9 - + arch/parisc/kernel/pci-dma.c | 6 +- + arch/powerpc/Kconfig | 3 +- + arch/powerpc/configs/ppc64_defconfig | 2 +- + arch/powerpc/include/asm/book3s/32/pgtable.h | 2 - + arch/powerpc/include/asm/book3s/64/pgtable.h | 7 - + arch/powerpc/include/asm/nohash/32/pgtable.h | 2 - + arch/powerpc/include/asm/nohash/64/pgtable-4k.h | 3 - + arch/powerpc/include/asm/nohash/64/pgtable.h | 5 - + arch/powerpc/mm/book3s64/radix_pgtable.c | 124 +-- + arch/powerpc/mm/book3s64/radix_tlb.c | 6 +- + arch/powerpc/mm/nohash/e500_hugetlbpage.c | 2 +- + arch/powerpc/mm/nohash/tlb.c | 2 +- + arch/powerpc/mm/ptdump/ptdump.c | 2 +- + arch/powerpc/platforms/powernv/Kconfig | 1 - + arch/powerpc/platforms/pseries/Kconfig | 1 - + arch/riscv/Kconfig | 4 +- + arch/riscv/include/asm/page.h | 6 - + arch/riscv/include/asm/pgtable-64.h | 9 - + arch/riscv/include/asm/pgtable.h | 4 - + arch/riscv/include/asm/set_memory.h | 5 +- + arch/riscv/kvm/mmu.c | 2 +- + arch/riscv/mm/init.c | 1 + + arch/riscv/mm/pageattr.c | 23 +- + arch/riscv/mm/tlbflush.c | 2 +- + arch/s390/Kconfig | 4 +- + arch/s390/configs/debug_defconfig | 2 +- + arch/s390/configs/defconfig | 2 +- + arch/s390/include/asm/pgtable.h | 11 - + arch/s390/include/asm/set_memory.h | 5 +- + arch/s390/mm/dump_pagetables.c | 2 +- + arch/s390/mm/gmap_helpers.c | 6 +- + arch/s390/mm/pageattr.c | 20 +- + arch/sh/Kconfig | 1 + + arch/sh/include/asm/pgalloc.h | 6 +- + arch/sh/include/asm/pgtable-3level.h | 3 - + arch/sh/include/asm/pgtable_32.h | 13 - + arch/sh/mm/init.c | 16 +- + arch/sh/mm/pgtable.c | 20 + + arch/sparc/Kconfig | 4 +- + arch/sparc/include/asm/pgalloc_32.h | 7 +- + arch/sparc/include/asm/pgalloc_64.h | 8 - + arch/sparc/include/asm/pgtable_32.h | 3 - + arch/sparc/include/asm/pgtable_64.h | 10 - + arch/sparc/include/asm/tlb_64.h | 2 - + arch/sparc/lib/bitext.c | 14 +- + arch/sparc/mm/init_64.c | 2 +- + arch/sparc/mm/srmmu.c | 32 +- + arch/um/Kconfig | 1 - + arch/um/include/asm/pgtable-2level.h | 7 - + arch/um/include/asm/pgtable-4level.h | 13 - + arch/x86/Kconfig | 6 +- + arch/x86/configs/x86_64_defconfig | 2 +- + arch/x86/entry/vdso/vdso32/fake_32bit_build.h | 2 +- + arch/x86/events/intel/bts.c | 3 - + arch/x86/events/intel/pt.c | 6 +- + arch/x86/include/asm/pgtable-2level.h | 5 - + arch/x86/include/asm/pgtable-3level.h | 11 - + arch/x86/include/asm/pgtable.h | 6 +- + arch/x86/include/asm/pgtable_64.h | 18 - + arch/x86/include/asm/set_memory.h | 5 +- + arch/x86/include/asm/string_64.h | 85 +- + arch/x86/kernel/uprobes.c | 2 +- + arch/x86/mm/pat/set_memory.c | 20 +- + arch/x86/mm/pti.c | 2 +- + arch/xtensa/include/asm/pgtable.h | 4 - + arch/xtensa/include/asm/tlb.h | 2 +- + drivers/android/binder/page_range.rs | 19 +- + drivers/android/binder_alloc.c | 63 +- + drivers/base/memory.c | 9 +- + drivers/block/zram/zcomp.c | 23 +- + drivers/block/zram/zcomp.h | 4 +- + drivers/block/zram/zram_drv.c | 82 +- + drivers/char/Makefile | 2 +- + drivers/gpu/drm/drm_gpusvm.c | 5 +- + drivers/gpu/drm/sti/sti_cursor.c | 4 +- + drivers/gpu/drm/sti/sti_hqvdp.c | 2 +- + drivers/gpu/ipu-v3/ipu-image-convert.c | 2 +- + drivers/hid/hid-core.c | 4 +- + drivers/hsi/clients/cmt_speech.c | 35 +- + drivers/infiniband/hw/hfi1/file_ops.c | 84 +- + drivers/md/md-bitmap.c | 18 +- + drivers/media/platform/nxp/imx7-media-csi.c | 2 +- + .../media/platform/nxp/imx8-isi/imx8-isi-video.c | 2 +- + drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c | 2 +- + drivers/nvdimm/pmem.c | 2 +- + drivers/nvdimm/pmem.h | 12 - + drivers/scsi/sg.c | 115 ++- + drivers/spi/spi-atmel.c | 4 +- + drivers/spi/spi-ti-qspi.c | 2 +- + drivers/staging/media/imx/imx-media-utils.c | 2 +- + drivers/usb/cdns3/cdns3-gadget.c | 2 +- + drivers/usb/gadget/udc/cdns2/cdns2-gadget.c | 2 +- + drivers/usb/gadget/udc/lpc32xx_udc.c | 2 +- + drivers/usb/mon/mon_bin.c | 98 ++- + drivers/video/fbdev/core/fb_defio.c | 6 +- + drivers/video/fbdev/fsl-diu-fb.c | 2 +- + drivers/video/fbdev/ssd1307fb.c | 2 + + drivers/xen/grant-table.c | 11 +- + fs/Kconfig | 2 +- + fs/buffer.c | 8 - + fs/ceph/addr.c | 8 +- + fs/coredump.c | 6 +- + fs/crypto/crypto.c | 2 - + fs/erofs/data.c | 16 +- + fs/erofs/zdata.c | 13 +- + fs/f2fs/compress.c | 35 +- + fs/f2fs/data.c | 2 +- + fs/f2fs/f2fs.h | 99 +-- + fs/f2fs/segment.c | 2 +- + fs/fuse/dax.c | 2 +- + fs/hugetlbfs/inode.c | 10 +- + fs/iomap/buffered-io.c | 3 +- + fs/nfs/file.c | 4 +- + fs/nfs/write.c | 2 - + fs/proc/base.c | 3 +- + fs/proc/internal.h | 2 - + fs/proc/page.c | 8 +- + fs/proc/task_mmu.c | 448 ++++------- + fs/proc/vmcore.c | 27 +- + fs/ubifs/file.c | 8 +- + fs/xfs/libxfs/xfs_btree.c | 18 +- + fs/xfs/xfs_platform.h | 4 - + include/asm-generic/pgtable-nop4d.h | 1 - + include/asm-generic/pgtable-nopmd.h | 1 - + include/asm-generic/pgtable-nopud.h | 1 - + include/asm-generic/tlb.h | 70 +- + include/linux/buffer_head.h | 8 +- + include/linux/cache.h | 1 + + include/linux/cgroup.h | 3 + + include/linux/compaction.h | 11 +- + include/linux/compiler.h | 5 + + include/linux/damon.h | 98 ++- + include/linux/gfp_types.h | 2 +- + include/linux/huge_mm.h | 9 - + include/linux/hugetlb.h | 50 +- + include/linux/hugetlb_inline.h | 28 - + include/linux/idr.h | 5 +- + include/linux/kernel-page-flags.h | 1 - + include/linux/maple_tree.h | 4 +- + include/linux/memblock.h | 1 - + include/linux/memcontrol.h | 464 ++++++----- + include/linux/mm.h | 369 +++++++-- + include/linux/mm_inline.h | 115 ++- + include/linux/mm_types.h | 66 +- + include/linux/mmap_lock.h | 106 ++- + include/linux/mmzone.h | 224 +++--- + include/linux/page-flags.h | 64 +- + include/linux/page_counter.h | 3 +- + include/linux/pagemap.h | 68 +- + include/linux/pgtable.h | 24 +- + include/linux/ptdump.h | 4 +- + include/linux/rmap.h | 2 +- + include/linux/sched.h | 4 +- + include/linux/set_memory.h | 120 ++- + include/linux/shmem_fs.h | 12 +- + include/linux/string.h | 13 + + include/linux/swap.h | 59 +- + include/linux/swapops.h | 9 - + include/linux/userfaultfd_k.h | 1 - + include/linux/vmalloc.h | 4 + + include/linux/vmemmap-optimization.h | 115 +++ + include/linux/zsmalloc.h | 4 - + include/linux/zswap.h | 8 +- + include/net/mana/mana.h | 4 +- + include/trace/events/huge_memory.h | 18 +- + include/trace/events/mmflags.h | 45 +- + include/trace/events/pagemap.h | 2 +- + include/trace/events/vmscan.h | 14 - + init/main.c | 2 +- + kernel/bpf/arena.c | 3 +- + kernel/bpf/stackmap.c | 17 +- + kernel/bpf/task_iter.c | 2 +- + kernel/cgroup/cgroup-internal.h | 1 - + kernel/configs/debug.config | 2 +- + kernel/events/core.c | 2 +- + kernel/events/ring_buffer.c | 7 +- + kernel/events/uprobes.c | 4 +- + kernel/fork.c | 2 - + kernel/power/snapshot.c | 4 +- + kernel/sched/fair.c | 3 +- + kernel/vmcore_info.c | 1 - + lib/interval_tree_test.c | 2 +- + lib/region_alloc_benchmark.c | 8 +- + lib/test_hmm.c | 2 + + mm/Kconfig | 82 +- + mm/Kconfig.debug | 40 - + mm/Makefile | 3 +- + mm/alloc_tag.c | 2 +- + drivers/char/mem.c => mm/char-mem.c | 21 +- + mm/cma.h | 23 +- + mm/collapse.h | 162 ++++ + mm/compaction.c | 11 +- + mm/damon/core.c | 392 ++++++--- + mm/damon/lru_sort.c | 12 +- + mm/damon/ops-common.c | 195 ++++- + mm/damon/ops-common.h | 12 + + mm/damon/paddr.c | 83 +- + mm/damon/reclaim.c | 8 - + mm/damon/sysfs-schemes.c | 51 +- + mm/damon/sysfs.c | 364 ++++++++- + mm/damon/tests/core-kunit.h | 883 ++++++++++++++++++++- + mm/damon/vaddr.c | 299 +++++-- + mm/debug.c | 4 - + mm/execmem.c | 148 ++-- + mm/filemap.c | 34 +- + mm/folio.c | 21 +- + mm/gup.c | 222 +++--- + mm/gup_test.c | 11 +- + mm/hmm.c | 3 +- + mm/huge_memory.c | 757 +++++++++--------- + mm/hugetlb.c | 303 ++++--- + mm/hugetlb_cgroup.c | 33 + + mm/hugetlb_sysfs.c | 10 +- + mm/hugetlb_vmemmap.c | 130 +-- + mm/hugetlb_vmemmap.h | 15 +- + mm/init-mm.c | 2 - + mm/internal.h | 140 ++-- + mm/interval_tree.c | 33 + + mm/kfence/core.c | 2 +- + mm/khugepaged.c | 690 ++++++++-------- + mm/kmemleak.c | 72 +- + mm/ksm.c | 27 +- + mm/list_lru.c | 13 +- + mm/madvise.c | 231 +++++- + mm/memblock.c | 27 +- + mm/memcontrol-v1.c | 413 +--------- + mm/memcontrol-v1.h | 31 +- + mm/memcontrol.c | 648 ++++++++++----- + mm/memfd.c | 31 +- + mm/memory-failure.c | 31 +- + mm/memory-tiers.c | 7 +- + mm/memory.c | 266 ++++--- + mm/memory_hotplug.c | 149 ++-- + mm/mempolicy.c | 191 +++-- + mm/migrate.c | 21 +- + mm/migrate_device.c | 18 +- + mm/mincore.c | 31 +- + mm/mlock.c | 61 +- + mm/mm_init.c | 243 +++--- + mm/mm_init.h | 7 +- + mm/mmap.c | 8 +- + mm/mmap_lock.c | 61 +- + mm/mmu_gather.c | 32 +- + mm/mprotect.c | 11 +- + mm/mremap.c | 21 +- + mm/nommu.c | 6 +- + mm/oom_kill.c | 15 +- + mm/page-writeback.c | 2 +- + mm/page_alloc.c | 121 ++- + mm/page_alloc.h | 2 +- + mm/page_counter.c | 41 +- + mm/page_io.c | 59 +- + mm/page_isolation.c | 40 +- + mm/page_owner.c | 6 +- + mm/page_table_check.c | 6 +- + mm/page_vma_mapped.c | 11 +- + mm/pagewalk.c | 4 +- + mm/percpu.c | 9 +- + mm/pgtable-generic.c | 37 +- + mm/rmap.c | 37 +- + mm/secretmem.c | 8 +- + mm/shmem.c | 418 +++++++--- + mm/slab.h | 28 +- + mm/slab_common.c | 2 +- + mm/slub.c | 330 +++++--- + mm/sparse-vmemmap.c | 399 +++++----- + mm/sparse.c | 236 ++---- + mm/sparse.h | 32 +- + mm/swap.h | 17 +- + mm/swap_state.c | 82 +- + mm/swapfile.c | 151 ++-- + mm/truncate.c | 131 +-- + mm/userfaultfd.c | 120 +-- + mm/util.c | 31 +- + mm/vma.c | 440 +++++++--- + mm/vma.h | 65 +- + mm/vma_internal.h | 1 - + mm/vmalloc.c | 231 +++--- + mm/vmalloc.h | 2 +- + mm/vmpressure.c | 3 - + mm/vmscan.c | 629 +++++++++------ + mm/vmstat.c | 17 +- + mm/workingset.c | 7 +- + mm/zpdesc.h | 2 +- + mm/zsmalloc.c | 101 +-- + mm/zswap.c | 363 +++++---- + net/ipv4/tcp.c | 31 +- + rust/kernel/mm.rs | 58 +- + samples/damon/mtier.c | 8 +- + samples/damon/prcl.c | 5 +- + samples/damon/wsse.c | 5 +- + scripts/gdb/linux/mm.py | 22 +- + security/selinux/selinuxfs.c | 11 +- + sound/core/pcm_native.c | 48 +- + sound/mips/snd-n64.c | 2 +- + tools/include/linux/compiler.h | 5 + + tools/include/linux/mm.h | 6 +- + tools/lib/mm/file_utils.c | 106 +++ + tools/lib/mm/file_utils.h | 13 + + .../selftests => lib}/mm/hugepage_settings.c | 186 ++++- + .../selftests => lib}/mm/hugepage_settings.h | 12 + + tools/mm/.gitignore | 1 + + tools/mm/Makefile | 11 +- + .../selftests/mm/gup_test.c => mm/gup_bench.c} | 192 ++--- + tools/mm/page-types.c | 2 - + tools/mm/page_owner_sort.c | 134 +++- + tools/sched_ext/include/scx/common.bpf.h | 2 - + tools/testing/memblock/Makefile | 3 +- + tools/testing/memblock/README | 12 +- + tools/testing/memblock/TODO | 5 - + tools/testing/memblock/asm/dma.h | 6 + + tools/testing/memblock/main.c | 2 + + tools/testing/memblock/tests/alloc_low_api.c | 148 ++++ + tools/testing/memblock/tests/alloc_low_api.h | 9 + + tools/testing/memblock/tests/basic_api.c | 20 +- + tools/testing/memblock/tests/common.c | 12 +- + tools/testing/radix-tree/idr-test.c | 20 +- + tools/testing/radix-tree/maple.c | 12 +- + tools/testing/selftests/cgroup/test_zswap.c | 24 +- + tools/testing/selftests/damon/.gitignore | 1 + + tools/testing/selftests/damon/_damon_sysfs.py | 161 +++- + tools/testing/selftests/damon/damon_nr_regions.py | 3 + + .../selftests/damon/damos_apply_interval.py | 3 + + tools/testing/selftests/damon/damos_quota.py | 3 + + tools/testing/selftests/damon/damos_quota_goal.py | 3 + + .../testing/selftests/damon/damos_tried_regions.py | 3 + + .../selftests/damon/drgn_dump_damon_status.py | 32 + + tools/testing/selftests/damon/sysfs.py | 39 +- + tools/testing/selftests/damon/sysfs.sh | 28 + + .../selftests/damon/sysfs_memcg_path_leak.sh | 7 + + .../selftests/damon/sysfs_no_op_commit_break.py | 35 +- + .../sysfs_update_schemes_tried_regions_hang.py | 3 + + ..._update_schemes_tried_regions_wss_estimation.py | 3 + + tools/testing/selftests/mm/.gitignore | 2 + + tools/testing/selftests/mm/Makefile | 21 +- + tools/testing/selftests/mm/compaction_test.c | 2 +- + tools/testing/selftests/mm/cow.c | 1 - + tools/testing/selftests/mm/folio_order_check.c | 122 +++ + tools/testing/selftests/mm/folio_split_race_test.c | 5 +- + tools/testing/selftests/mm/guard-regions.c | 15 +- + tools/testing/selftests/mm/gup.c | 262 ++++++ + tools/testing/selftests/mm/gup_longterm.c | 1 - + tools/testing/selftests/mm/hmm-tests.c | 7 +- + tools/testing/selftests/mm/hugetlb-madvise.c | 1 - + tools/testing/selftests/mm/hugetlb-mmap.c | 1 - + tools/testing/selftests/mm/hugetlb-mremap.c | 1 - + tools/testing/selftests/mm/hugetlb-shm.c | 1 - + tools/testing/selftests/mm/hugetlb-soft-offline.c | 59 +- + tools/testing/selftests/mm/hugetlb_dio.c | 1 - + .../selftests/mm/hugetlb_fault_after_madv.c | 1 - + tools/testing/selftests/mm/hugetlb_madv_vs_map.c | 142 +++- + tools/testing/selftests/mm/khugepaged.c | 580 ++++++++++++-- + tools/testing/selftests/mm/khugepaged_race.c | 538 +++++++++++++ + tools/testing/selftests/mm/khugepaged_sync_check.c | 179 +++++ + tools/testing/selftests/mm/ksm_tests.c | 1 - + tools/testing/selftests/mm/memfd_secret.c | 3 +- + tools/testing/selftests/mm/memory-failure.c | 114 ++- + tools/testing/selftests/mm/merge.c | 95 +++ + tools/testing/selftests/mm/migration.c | 3 +- + tools/testing/selftests/mm/mlock-random-test.c | 1 - + tools/testing/selftests/mm/mlock2-tests.c | 2 +- + tools/testing/selftests/mm/mlock2.h | 8 +- + tools/testing/selftests/mm/mremap_test.c | 44 +- + tools/testing/selftests/mm/pagemap_ioctl.c | 95 +-- + tools/testing/selftests/mm/pkey-helpers.h | 7 - + tools/testing/selftests/mm/pkey_sighandler_tests.c | 4 +- + tools/testing/selftests/mm/prctl_thp_disable.c | 1 - + tools/testing/selftests/mm/protection_keys.c | 2 +- + tools/testing/selftests/mm/run_vmtests.sh | 90 +-- + tools/testing/selftests/mm/soft-dirty.c | 1 - + tools/testing/selftests/mm/split_huge_page_test.c | 90 +-- + tools/testing/selftests/mm/thuge-gen.c | 1 - + tools/testing/selftests/mm/transhuge-stress.c | 1 - + tools/testing/selftests/mm/uffd-common.c | 4 +- + tools/testing/selftests/mm/uffd-common.h | 1 - + tools/testing/selftests/mm/uffd-wp-mremap.c | 4 +- + tools/testing/selftests/mm/va_high_addr_switch.c | 1 - + tools/testing/selftests/mm/vm_util.c | 408 ++++++---- + tools/testing/selftests/mm/vm_util.h | 19 +- + tools/testing/selftests/proc/proc-maps-race.c | 186 ++++- + .../selftests/proc/proc-self-map-files-001.c | 2 +- + .../selftests/proc/proc-self-map-files-002.c | 2 +- + tools/testing/vma/include/dup.h | 97 ++- + tools/testing/vma/include/stubs.h | 14 +- + tools/testing/vma/shared.c | 9 + + tools/testing/vma/tests/merge.c | 10 +- + tools/testing/vma/tests/mmap.c | 105 +++ + tools/testing/vma/vma_internal.h | 1 - + 458 files changed, 14599 insertions(+), 8013 deletions(-) + create mode 100644 Documentation/mm/kernel-page-tables.rst + delete mode 100644 include/linux/hugetlb_inline.h + create mode 100644 include/linux/vmemmap-optimization.h + rename drivers/char/mem.c => mm/char-mem.c (97%) + create mode 100644 mm/collapse.h + create mode 100644 tools/lib/mm/file_utils.c + create mode 100644 tools/lib/mm/file_utils.h + rename tools/{testing/selftests => lib}/mm/hugepage_settings.c (76%) + rename tools/{testing/selftests => lib}/mm/hugepage_settings.h (90%) + rename tools/{testing/selftests/mm/gup_test.c => mm/gup_bench.c} (50%) + delete mode 100644 tools/testing/memblock/TODO + create mode 100644 tools/testing/memblock/tests/alloc_low_api.c + create mode 100644 tools/testing/memblock/tests/alloc_low_api.h + create mode 100644 tools/testing/selftests/mm/folio_order_check.c + create mode 100644 tools/testing/selftests/mm/gup.c + create mode 100644 tools/testing/selftests/mm/khugepaged_race.c + create mode 100644 tools/testing/selftests/mm/khugepaged_sync_check.c +Merging mm-nonmm-stable/mm-nonmm-stable (a243ede718463 Merge tag 'mtd/fixes-for-7.3-rc6' of git://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux) +$ git merge -m Merge branch 'mm-nonmm-stable' of https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm mm-nonmm-stable/mm-nonmm-stable +Already up to date. +Merging mm-nonmm-unstable/mm-nonmm-unstable (cbb17a7a0afe4 init: simplify initramfs.o build rule) +$ git merge -m Merge branch 'mm-nonmm-unstable' of https://git.kernel.org/pub/scm/linux/kernel/git/akpm/mm mm-nonmm-unstable/mm-nonmm-unstable +Auto-merging .mailmap +Auto-merging MAINTAINERS +Auto-merging arch/Kconfig +Auto-merging arch/s390/Kconfig +Auto-merging fs/proc/base.c +Auto-merging init/main.c +Auto-merging kernel/fork.c +Auto-merging mm/shmem.c +Merge made by the 'ort' strategy. + .mailmap | 4 + + CREDITS | 181 ++++++++-------- + Documentation/admin-guide/sysctl/kernel.rst | 5 +- + MAINTAINERS | 1 + + arch/Kconfig | 8 - + arch/alpha/include/uapi/asm/setup.h | 4 + + arch/arc/include/asm/setup.h | 2 +- + arch/arm/include/uapi/asm/setup.h | 6 +- + arch/arm64/include/uapi/asm/setup.h | 4 + + arch/loongarch/include/uapi/asm/setup.h | 4 + + arch/m68k/include/uapi/asm/setup.h | 6 +- + arch/microblaze/include/uapi/asm/setup.h | 4 + + arch/mips/include/uapi/asm/setup.h | 4 + + arch/parisc/include/uapi/asm/setup.h | 4 + + arch/powerpc/include/uapi/asm/setup.h | 4 + + arch/riscv/include/uapi/asm/setup.h | 4 + + arch/s390/Kconfig | 8 - + arch/s390/kernel/traps.c | 7 + + arch/sparc/include/uapi/asm/setup.h | 10 +- + arch/sparc/kernel/setup.c | 9 + + arch/um/include/asm/setup.h | 2 +- + arch/x86/include/asm/setup.h | 2 +- + arch/xtensa/include/uapi/asm/setup.h | 4 + + drivers/media/v4l2-core/v4l2-vp9.c | 30 +-- + drivers/rapidio/devices/rio_mport_cdev.c | 9 +- + drivers/usb/gadget/legacy/inode.c | 4 +- + drivers/watchdog/hpwdt.c | 2 +- + fs/fat/inode.c | 23 +- + fs/ocfs2/alloc.c | 36 +++- + fs/ocfs2/cluster/heartbeat.c | 12 +- + fs/ocfs2/dir.c | 55 +++++ + fs/ocfs2/dlmglue.c | 33 ++- + fs/ocfs2/dlmglue.h | 5 +- + fs/ocfs2/export.c | 10 +- + fs/ocfs2/inode.c | 58 ++++- + fs/ocfs2/journal.c | 3 +- + fs/ocfs2/journal.h | 1 - + fs/ocfs2/ocfs2.h | 17 ++ + fs/ocfs2/quota_local.c | 33 ++- + fs/ocfs2/refcounttree.c | 27 +++ + fs/ocfs2/suballoc.c | 235 +++++++++++++++++--- + fs/ocfs2/suballoc.h | 2 +- + fs/ocfs2/super.c | 32 ++- + fs/ocfs2/xattr.c | 204 ++++++++++++++---- + fs/proc/base.c | 8 +- + fs/squashfs/fragment.c | 6 +- + fs/squashfs/squashfs_fs.h | 2 +- + include/linux/kdev_t.h | 2 +- + include/linux/minmax.h | 6 +- + include/linux/panic.h | 5 +- + include/uapi/asm-generic/setup.h | 4 + + init/Kconfig | 19 +- + init/Makefile | 6 +- + init/main.c | 12 +- + init/version.c | 11 +- + ipc/mqueue.c | 7 + + kernel/fork.c | 3 +- + kernel/gcov/fs.c | 2 +- + kernel/hung_task.c | 55 ++++- + kernel/kallsyms.c | 111 ++++++---- + kernel/kallsyms_internal.h | 11 + + kernel/kcov.c | 8 + + kernel/panic.c | 104 +++++---- + kernel/resource.c | 14 +- + kernel/resource_kunit.c | 3 + + kernel/taskstats.c | 54 ++--- + lib/Kconfig | 14 +- + lib/Kconfig.debug | 16 +- + lib/bootconfig.c | 46 ++-- + lib/decompress_bunzip2.c | 6 +- + lib/decompress_unlz4.c | 4 + + lib/decompress_unxz.c | 12 +- + lib/dynamic_debug.c | 15 +- + lib/group_cpus.c | 87 +++++++- + lib/klist.c | 16 +- + lib/percpu_counter.c | 10 +- + lib/plist.c | 7 + + lib/raid/Kconfig | 2 + + lib/raid/raid6/x86/avx2.c | 6 + + lib/raid/raid6/x86/avx512.c | 6 + + lib/raid/raid6/x86/recov_avx2.c | 2 + + lib/raid/raid6/x86/recov_avx512.c | 2 + + lib/raid/xor/x86/xor-avx.c | 1 + + lib/string_helpers.c | 5 +- + lib/tests/Makefile | 1 + + lib/tests/errseq_kunit.c | 237 +++++++++++++++++++++ + lib/tests/string_helpers_kunit.c | 46 ++-- + mm/shmem.c | 5 + + scripts/checkpatch.pl | 38 +++- + scripts/checkstack.pl | 6 +- + scripts/kallsyms.c | 14 +- + scripts/spelling.txt | 4 +- + tools/bootconfig/main.c | 104 ++++----- + tools/bootconfig/test-bootconfig.sh | 12 ++ + tools/testing/selftests/core/unshare_test.c | 19 +- + tools/testing/selftests/filelock/ofdlocks.c | 2 +- + .../filesystems/epoll/epoll_wakeup_test.c | 32 ++- + .../selftests/membarrier/membarrier_test_impl.h | 17 +- + tools/testing/selftests/uevent/uevent_filtering.c | 8 - + 99 files changed, 1785 insertions(+), 588 deletions(-) + create mode 100644 lib/tests/errseq_kunit.c +Merging kbuild/kbuild-for-next (59ba9b8f8803c Merge branch 'kbuild-next-unstable' into kbuild-for-next) +$ git merge -m Merge branch 'kbuild-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git kbuild/kbuild-for-next +Auto-merging MAINTAINERS +Auto-merging Makefile +Auto-merging arch/x86/Kconfig +Auto-merging arch/x86/Makefile +Auto-merging init/Kconfig +Auto-merging scripts/kallsyms.c +CONFLICT (content): Merge conflict in scripts/kallsyms.c +Recorded preimage for 'scripts/kallsyms.c' +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +Recorded resolution for 'scripts/kallsyms.c'. +[master b7d4f7554b9d1] Merge branch 'kbuild-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/kbuild/linux.git +$ git diff -M --stat --summary HEAD^.. + .../early-userspace/early_userspace_support.rst | 16 +- + .../filesystems/ramfs-rootfs-initramfs.rst | 10 +- + Documentation/kbuild/kbuild.rst | 17 + + Documentation/kbuild/kconfig-language.rst | 13 +- + Documentation/kbuild/kconfig-macro-language.rst | 8 +- + Documentation/process/changes.rst | 8 + + MAINTAINERS | 2 + + Makefile | 51 +-- + arch/arm64/kernel/pi/Makefile | 2 +- + arch/riscv/kernel/pi/Makefile | 2 +- + arch/x86/Kconfig | 17 + + arch/x86/Makefile | 12 +- + arch/x86/boot/Makefile | 2 +- + arch/x86/boot/compressed/Makefile | 2 +- + drivers/firmware/efi/libstub/Makefile | 2 +- + drivers/firmware/qcom/Kconfig | 26 +- + drivers/firmware/qcom/qcom_tzmem.c | 4 +- + fs/erofs/Kconfig | 8 +- + include/asm-generic/vmlinux.lds.h | 2 +- + init/Kconfig | 197 +---------- + kernel/Makefile | 6 +- + scripts/.gitignore | 1 + + scripts/Kconfig.include | 29 +- + scripts/Kconfig.toolchain | 272 +++++++++++++++ + scripts/Makefile | 8 +- + scripts/Makefile.gcc-plugins | 4 + + scripts/Makefile.lib | 4 +- + scripts/Makefile.package | 2 +- + scripts/Makefile.vmlinux | 20 +- + scripts/Makefile.warn | 28 +- + scripts/check-function-names.sh | 3 +- + scripts/elf-parse.c | 103 +++++- + scripts/elf-parse.h | 54 +++ + {usr => scripts}/gen_init_cpio.c | 0 + {usr => scripts}/gen_initramfs.sh | 4 +- + scripts/jobserver-exec | 3 +- + scripts/kallsyms-sysmap.c | 248 +++++++++++++ + scripts/kallsyms.c | 382 +++++++++++++++------ + scripts/kallsyms.h | 44 +++ + scripts/kconfig/conf.c | 16 +- + scripts/kconfig/confdata.c | 6 + + scripts/kconfig/kconfig-sym-check.pl | 2 +- + scripts/kconfig/lexer.l | 3 + + scripts/kconfig/lkc.h | 2 +- + scripts/kconfig/lkc_proto.h | 1 + + scripts/kconfig/menu.c | 64 +++- + scripts/kconfig/parser.y | 8 +- + scripts/kconfig/symbol.c | 115 +++++-- + scripts/kconfig/tests/conftest.py | 27 +- + scripts/kconfig/tests/def_type/Kconfig | 28 ++ + scripts/kconfig/tests/def_type/__init__.py | 18 + + scripts/kconfig/tests/def_type/expected_guard_n | 9 + + scripts/kconfig/tests/def_type/expected_guard_y | 11 + + scripts/kconfig/tests/def_type/guard_n.config | 1 + + scripts/kconfig/tests/def_type/guard_y.config | 1 + + scripts/kconfig/tests/err_num_bounds/Kconfig | 79 +++++ + scripts/kconfig/tests/err_num_bounds/__init__.py | 12 + + .../kconfig/tests/err_num_bounds/expected_stderr | 10 + + scripts/kconfig/tests/err_num_mismatch/Kconfig | 38 ++ + scripts/kconfig/tests/err_num_mismatch/__init__.py | 9 + + .../kconfig/tests/err_num_mismatch/expected_stderr | 4 + + .../kconfig/tests/err_num_non_numeric_ref/Kconfig | 75 ++++ + .../tests/err_num_non_numeric_ref/__init__.py | 9 + + .../tests/err_num_non_numeric_ref/expected_stderr | 16 + + scripts/kconfig/tests/err_recursive_dep/Kconfig | 24 ++ + .../tests/err_recursive_dep/expected_stderr | 13 + + .../kconfig/tests/randconfig_probability/Kconfig | 12 + + .../tests/randconfig_probability/__init__.py | 130 +++++++ + scripts/kconfig/tests/savedefconfig_range/Kconfig | 60 ++++ + .../kconfig/tests/savedefconfig_range/__init__.py | 8 + + scripts/kconfig/tests/savedefconfig_range/config | 7 + + .../tests/savedefconfig_range/expected_defconfig | 0 + scripts/kconfig/tests/warn_changed_input/Kconfig | 5 + + .../kconfig/tests/warn_changed_input/__init__.py | 21 +- + .../tests/warn_changed_input/expected_config | 1 + + scripts/kconfig/tests/warn_num_bounds/Kconfig | 24 ++ + scripts/kconfig/tests/warn_num_bounds/__init__.py | 25 ++ + scripts/kconfig/tests/warn_num_bounds/config | 7 + + .../kconfig/tests/warn_num_bounds/expected_config | 11 + + .../tests/warn_num_bounds/expected_config_stderr | 3 + + .../tests/warn_num_bounds/expected_frontend_config | 11 + + .../tests/warn_num_bounds/expected_frontend_stderr | 0 + scripts/link-vmlinux.sh | 24 +- + scripts/mkcompile_h | 2 +- + scripts/mksysmap | 94 ----- + scripts/remove-stale-files | 2 + + .../sbom/tests/cmd_graph/test_savedcmd_parser.py | 2 +- + scripts/setlocalversion | 3 +- + scripts/sorttable.c | 15 +- + scripts/ver_linux | 1 + + tools/lib/python/jobserver.py | 13 + + tools/testing/selftests/kho/vmtest.sh | 2 +- + tools/testing/selftests/liveupdate/vmtest.sh | 4 +- + tools/testing/selftests/nolibc/Makefile.nolibc | 2 +- + usr/.gitignore | 4 +- + usr/Kconfig | 2 +- + usr/Makefile | 4 +- + 97 files changed, 2091 insertions(+), 625 deletions(-) + create mode 100644 scripts/Kconfig.toolchain + rename {usr => scripts}/gen_init_cpio.c (100%) + rename {usr => scripts}/gen_initramfs.sh (97%) + create mode 100644 scripts/kallsyms-sysmap.c + create mode 100644 scripts/kallsyms.h + create mode 100644 scripts/kconfig/tests/def_type/Kconfig + create mode 100644 scripts/kconfig/tests/def_type/__init__.py + create mode 100644 scripts/kconfig/tests/def_type/expected_guard_n + create mode 100644 scripts/kconfig/tests/def_type/expected_guard_y + create mode 100644 scripts/kconfig/tests/def_type/guard_n.config + create mode 100644 scripts/kconfig/tests/def_type/guard_y.config + create mode 100644 scripts/kconfig/tests/err_num_bounds/Kconfig + create mode 100644 scripts/kconfig/tests/err_num_bounds/__init__.py + create mode 100644 scripts/kconfig/tests/err_num_bounds/expected_stderr + create mode 100644 scripts/kconfig/tests/err_num_mismatch/Kconfig + create mode 100644 scripts/kconfig/tests/err_num_mismatch/__init__.py + create mode 100644 scripts/kconfig/tests/err_num_mismatch/expected_stderr + create mode 100644 scripts/kconfig/tests/err_num_non_numeric_ref/Kconfig + create mode 100644 scripts/kconfig/tests/err_num_non_numeric_ref/__init__.py + create mode 100644 scripts/kconfig/tests/err_num_non_numeric_ref/expected_stderr + create mode 100644 scripts/kconfig/tests/randconfig_probability/Kconfig + create mode 100644 scripts/kconfig/tests/randconfig_probability/__init__.py + create mode 100644 scripts/kconfig/tests/savedefconfig_range/Kconfig + create mode 100644 scripts/kconfig/tests/savedefconfig_range/__init__.py + create mode 100644 scripts/kconfig/tests/savedefconfig_range/config + create mode 100644 scripts/kconfig/tests/savedefconfig_range/expected_defconfig + create mode 100644 scripts/kconfig/tests/warn_num_bounds/Kconfig + create mode 100644 scripts/kconfig/tests/warn_num_bounds/__init__.py + create mode 100644 scripts/kconfig/tests/warn_num_bounds/config + create mode 100644 scripts/kconfig/tests/warn_num_bounds/expected_config + create mode 100644 scripts/kconfig/tests/warn_num_bounds/expected_config_stderr + create mode 100644 scripts/kconfig/tests/warn_num_bounds/expected_frontend_config + create mode 100644 scripts/kconfig/tests/warn_num_bounds/expected_frontend_stderr + delete mode 100755 scripts/mksysmap +Merging clang-fixes/clang-fixes-for-next (0bb666d5f5a23 once_lite: Simplify condition handling and fix context analysis) +$ git merge -m Merge branch 'clang-fixes-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/nathan/linux.git clang-fixes/clang-fixes-for-next +Merge made by the 'ort' strategy. + include/linux/once_lite.h | 8 +++++--- + 1 file changed, 5 insertions(+), 3 deletions(-) +Merging clang-format/clang-format (8f0b4cce4481f Linux 6.19-rc1) +$ git merge -m Merge branch 'clang-format' of https://github.com/ojeda/linux.git clang-format/clang-format +Already up to date. +Merging perf/perf-tools-next (45d15e89a783a perf symbols: Don't let a module's last symbol overlap the next module) +$ git merge -m Merge branch 'perf-tools-next' of https://git.kernel.org/pub/scm/linux/kernel/git/perf/perf-tools-next.git perf/perf-tools-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 2 +- + tools/arch/x86/include/asm/msr-index.h | 7 + + tools/arch/x86/include/uapi/asm/perf_regs.h | 53 + + tools/build/Makefile.feature | 8 +- + tools/build/feature/Makefile | 42 +- + tools/build/feature/test-all.c | 6 +- + tools/build/feature/test-gettid.c | 1 + + tools/build/feature/test-gtk2-infobar.c | 12 - + tools/build/feature/{test-gtk2.c => test-gtk4.c} | 4 +- + tools/build/feature/test-libdw.c | 10 +- + tools/build/feature/test-libperl.c | 10 - + tools/build/feature/test-libpython.c | 10 - + tools/build/feature/test-python-module.c | 13 + + tools/build/feature/test-timerfd.c | 2 +- + tools/include/tools/dis-asm-compat.h | 1 + + tools/include/uapi/linux/perf_event.h | 49 +- + tools/lib/api/fd/array.c | 4 + + tools/lib/perf/evlist.c | 5 +- + tools/lib/perf/include/internal/evsel.h | 2 + + tools/perf/.clang-format | 1 + + tools/perf/Build | 5 +- + tools/perf/Documentation/db-export.txt | 20 +- + tools/perf/Documentation/perf-annotate.txt | 3 + + tools/perf/Documentation/perf-c2c.txt | 14 +- + tools/perf/Documentation/perf-check.txt | 3 +- + tools/perf/Documentation/perf-config.txt | 10 + + tools/perf/Documentation/perf-mem.txt | 4 + + tools/perf/Documentation/perf-record.txt | 27 +- + tools/perf/Documentation/perf-report.txt | 13 +- + tools/perf/Documentation/perf-script-perl.txt | 216 - + tools/perf/Documentation/perf-script-python.txt | 752 +-- + tools/perf/Documentation/perf-script.txt | 82 +- + tools/perf/Documentation/perf-stat.txt | 12 +- + tools/perf/Documentation/perf-top.txt | 11 + + tools/perf/Documentation/perf.data-file-format.txt | 13 + + tools/perf/Documentation/tips.txt | 3 +- + tools/perf/Makefile | 35 +- + tools/perf/Makefile.config | 106 +- + tools/perf/Makefile.perf | 120 +- + tools/perf/arch/arm/tests/dwarf-unwind.c | 1 + + tools/perf/arch/arm64/tests/dwarf-unwind.c | 1 + + tools/perf/arch/loongarch/util/header.c | 2 +- + tools/perf/arch/powerpc/tests/dwarf-unwind.c | 1 + + tools/perf/arch/powerpc/util/skip-callchain-idx.c | 4 +- + tools/perf/arch/riscv/Build | 1 + + tools/perf/arch/riscv/include/arch-tests.h | 9 + + tools/perf/arch/riscv/include/perf_regs.h | 2 + + tools/perf/arch/riscv/tests/Build | 4 + + tools/perf/arch/riscv/tests/arch-tests.c | 10 + + tools/perf/arch/riscv/tests/dwarf-unwind.c | 64 + + tools/perf/arch/riscv/tests/regs_load.S | 58 + + tools/perf/arch/x86/tests/dwarf-unwind.c | 4 + + tools/perf/arch/x86/tests/topdown.c | 30 + + tools/perf/arch/x86/util/mem-events.c | 17 + + tools/perf/arch/x86/util/mem-events.h | 2 + + tools/perf/arch/x86/util/pmu.c | 226 +- + tools/perf/arch/x86/util/topdown.c | 3 +- + tools/perf/bench/futex.h | 11 + + tools/perf/bench/numa.c | 2 + + tools/perf/bench/sched-messaging.c | 4 +- + tools/perf/builtin-annotate.c | 96 +- + tools/perf/builtin-buildid-list.c | 17 +- + tools/perf/builtin-c2c.c | 229 +- + tools/perf/builtin-check.c | 3 +- + tools/perf/builtin-config.c | 18 +- + tools/perf/builtin-ftrace.c | 28 +- + tools/perf/builtin-help.c | 30 +- + tools/perf/builtin-inject.c | 4 + + tools/perf/builtin-kmem.c | 3 +- + tools/perf/builtin-kvm.c | 1 + + tools/perf/builtin-list.c | 2 +- + tools/perf/builtin-mem.c | 7 +- + tools/perf/builtin-record.c | 15 + + tools/perf/builtin-report.c | 43 +- + tools/perf/builtin-sched.c | 7 +- + tools/perf/builtin-script.c | 1124 +++-- + tools/perf/builtin-stat.c | 4 +- + tools/perf/builtin-timechart.c | 5 - + tools/perf/builtin-top.c | 38 +- + tools/perf/builtin-trace.c | 271 +- + tools/perf/builtin.h | 3 + + tools/perf/perf.c | 120 +- + tools/perf/pmu-events/Build | 10 +- + tools/perf/pmu-events/amd_metrics.py | 16 +- + .../perf/pmu-events/arch/x86/alderlake/cache.json | 4 +- + .../perf/pmu-events/arch/x86/alderlake/memory.json | 13 + + .../perf/pmu-events/arch/x86/alderlake/other.json | 10 + + .../pmu-events/arch/x86/alderlake/pipeline.json | 6 +- + .../perf/pmu-events/arch/x86/alderlaken/cache.json | 4 +- + .../perf/pmu-events/arch/x86/arrowlake/cache.json | 1 + + .../perf/pmu-events/arch/x86/arrowlake/other.json | 19 + + .../pmu-events/arch/x86/arrowlake/pipeline.json | 10 +- + .../pmu-events/arch/x86/broadwell/bdw-metrics.json | 51 +- + .../arch/x86/broadwell/metricgroups.json | 3 +- + .../arch/x86/broadwellde/bdwde-metrics.json | 49 +- + .../arch/x86/broadwellde/metricgroups.json | 3 +- + .../arch/x86/broadwellx/bdx-metrics.json | 63 +- + .../arch/x86/broadwellx/metricgroups.json | 3 +- + .../arch/x86/cascadelakex/clx-metrics.json | 94 +- + .../arch/x86/cascadelakex/metricgroups.json | 3 +- + .../arch/x86/cascadelakex/uncore-interconnect.json | 6 + + .../arch/x86/clearwaterforest/pipeline.json | 3 +- + .../arch/x86/clearwaterforest/uncore-cache.json | 5 + + .../x86/clearwaterforest/uncore-interconnect.json | 2 + + .../pmu-events/arch/x86/emeraldrapids/cache.json | 2 +- + .../pmu-events/arch/x86/emeraldrapids/memory.json | 35 + + .../pmu-events/arch/x86/emeraldrapids/other.json | 9 + + .../arch/x86/emeraldrapids/pipeline.json | 10 +- + .../arch/x86/emeraldrapids/uncore-cache.json | 42 + + .../x86/emeraldrapids/uncore-interconnect.json | 12 + + .../arch/x86/emeraldrapids/uncore-io.json | 54 +- + .../arch/x86/emeraldrapids/uncore-memory.json | 10 + + .../pmu-events/arch/x86/graniterapids/other.json | 9 + + .../arch/x86/graniterapids/pipeline.json | 6 +- + .../arch/x86/graniterapids/uncore-cache.json | 15 + + .../x86/graniterapids/uncore-interconnect.json | 2 + + .../arch/x86/graniterapids/uncore-io.json | 12 + + .../pmu-events/arch/x86/haswell/hsw-metrics.json | 49 +- + .../pmu-events/arch/x86/haswell/metricgroups.json | 3 +- + .../pmu-events/arch/x86/haswellx/hsx-metrics.json | 61 +- + .../pmu-events/arch/x86/haswellx/metricgroups.json | 3 +- + .../pmu-events/arch/x86/icelake/icl-metrics.json | 84 +- + tools/perf/pmu-events/arch/x86/icelake/memory.json | 24 + + .../pmu-events/arch/x86/icelake/metricgroups.json | 2 +- + tools/perf/pmu-events/arch/x86/icelake/other.json | 9 + + .../perf/pmu-events/arch/x86/icelake/pipeline.json | 9 +- + .../pmu-events/arch/x86/icelakex/icx-metrics.json | 101 +- + .../perf/pmu-events/arch/x86/icelakex/memory.json | 24 + + .../pmu-events/arch/x86/icelakex/metricgroups.json | 2 +- + tools/perf/pmu-events/arch/x86/icelakex/other.json | 9 + + .../pmu-events/arch/x86/icelakex/pipeline.json | 7 +- + .../pmu-events/arch/x86/icelakex/uncore-cache.json | 79 + + .../arch/x86/icelakex/uncore-interconnect.json | 22 + + .../pmu-events/arch/x86/ivybridge/ivb-metrics.json | 51 +- + .../arch/x86/ivybridge/metricgroups.json | 3 +- + .../pmu-events/arch/x86/ivytown/ivt-metrics.json | 63 +- + .../pmu-events/arch/x86/ivytown/metricgroups.json | 3 +- + .../pmu-events/arch/x86/jaketown/jkt-metrics.json | 41 +- + .../pmu-events/arch/x86/jaketown/metricgroups.json | 3 +- + .../perf/pmu-events/arch/x86/lunarlake/cache.json | 3 + + .../pmu-events/arch/x86/lunarlake/pipeline.json | 2 + + tools/perf/pmu-events/arch/x86/mapfile.csv | 26 +- + .../perf/pmu-events/arch/x86/meteorlake/other.json | 10 + + .../pmu-events/arch/x86/meteorlake/pipeline.json | 6 +- + tools/perf/pmu-events/arch/x86/novalake/cache.json | 500 +- + .../arch/x86/novalake/floating-point.json | 423 ++ + .../pmu-events/arch/x86/novalake/frontend.json | 147 +- + .../perf/pmu-events/arch/x86/novalake/memory.json | 24 + + tools/perf/pmu-events/arch/x86/novalake/other.json | 213 + + .../pmu-events/arch/x86/novalake/pipeline.json | 508 +- + .../arch/x86/novalake/virtual-memory.json | 173 + + .../pmu-events/arch/x86/pantherlake/cache.json | 3 + + .../pmu-events/arch/x86/pantherlake/other.json | 10 + + .../pmu-events/arch/x86/pantherlake/pipeline.json | 10 +- + .../pmu-events/arch/x86/rocketlake/memory.json | 24 + + .../arch/x86/rocketlake/metricgroups.json | 2 +- + .../perf/pmu-events/arch/x86/rocketlake/other.json | 9 + + .../pmu-events/arch/x86/rocketlake/pipeline.json | 9 +- + .../arch/x86/rocketlake/rkl-metrics.json | 84 +- + .../arch/x86/sandybridge/metricgroups.json | 3 +- + .../arch/x86/sandybridge/snb-metrics.json | 35 +- + .../pmu-events/arch/x86/sapphirerapids/cache.json | 2 +- + .../pmu-events/arch/x86/sapphirerapids/memory.json | 35 + + .../arch/x86/sapphirerapids/metricgroups.json | 2 +- + .../pmu-events/arch/x86/sapphirerapids/other.json | 9 + + .../arch/x86/sapphirerapids/pipeline.json | 10 +- + .../arch/x86/sapphirerapids/spr-metrics.json | 93 +- + .../arch/x86/sapphirerapids/uncore-cache.json | 42 + + .../x86/sapphirerapids/uncore-interconnect.json | 12 + + .../arch/x86/sapphirerapids/uncore-io.json | 54 +- + .../arch/x86/sapphirerapids/uncore-memory.json | 10 + + .../arch/x86/sierraforest/uncore-cache.json | 5 + + .../arch/x86/sierraforest/uncore-interconnect.json | 2 + + .../pmu-events/arch/x86/skylake/metricgroups.json | 3 +- + .../pmu-events/arch/x86/skylake/skl-metrics.json | 59 +- + .../pmu-events/arch/x86/skylakex/metricgroups.json | 3 +- + .../pmu-events/arch/x86/skylakex/skx-metrics.json | 94 +- + .../arch/x86/skylakex/uncore-interconnect.json | 6 + + .../arch/x86/snowridgex/uncore-cache.json | 68 + + .../arch/x86/snowridgex/uncore-interconnect.json | 10 + + .../arch/x86/tigerlake/metricgroups.json | 2 +- + .../perf/pmu-events/arch/x86/tigerlake/other.json | 9 + + .../pmu-events/arch/x86/tigerlake/pipeline.json | 7 +- + .../pmu-events/arch/x86/tigerlake/tgl-metrics.json | 82 +- + tools/perf/pmu-events/intel_metrics.py | 268 +- + tools/perf/pmu-events/jevents.py | 112 +- + tools/perf/pmu-events/make_legacy_cache.py | 34 +- + tools/perf/pmu-events/metric.py | 60 +- + tools/perf/python/SchedGui.py | 246 + + tools/perf/python/arm-cs-trace-disasm.py | 440 ++ + tools/perf/python/check-perf-trace.py | 223 + + tools/perf/python/compaction-times.py | 362 ++ + tools/perf/python/counting.py | 1 + + tools/perf/python/event_analyzing_sample.py | 317 ++ + tools/perf/python/export-to-postgresql.py | 1328 +++++ + tools/perf/python/export-to-sqlite.py | 946 ++++ + tools/perf/python/exported-sql-viewer.py | 5103 ++++++++++++++++++++ + tools/perf/python/failed-syscalls-by-pid.py | 161 + + tools/perf/python/failed-syscalls.py | 94 + + tools/perf/python/flamegraph.py | 283 ++ + tools/perf/python/futex-contention.py | 107 + + tools/perf/python/gecko.py | 411 ++ + tools/perf/python/ilist.py | 86 +- + tools/perf/python/intel-pt-events.py | 632 +++ + tools/perf/python/libxed.py | 123 + + tools/perf/python/mem-phys-addr.py | 136 + + tools/perf/python/net_dropmonitor.py | 173 + + tools/perf/python/netdev-times.py | 494 ++ + tools/perf/python/parallel-perf.py | 1250 +++++ + tools/perf/python/perf.pyi | 122 +- + tools/perf/python/perf_live.py | 10 +- + tools/perf/{scripts => }/python/powerpc-hcalls.py | 213 +- + tools/perf/python/rw-by-file.py | 113 + + tools/perf/python/rw-by-pid.py | 205 + + tools/perf/python/rwtop.py | 256 + + tools/perf/python/sched-migration.py | 496 ++ + tools/perf/python/sctop.py | 253 + + tools/perf/python/stackcollapse.py | 145 + + tools/perf/python/stat-cpi.py | 231 + + tools/perf/python/syscall-counts-by-pid.py | 112 + + tools/perf/python/syscall-counts.py | 95 + + tools/perf/python/task-analyzer.py | 908 ++++ + tools/perf/python/tracepoint.py | 5 +- + tools/perf/python/treport.py | 564 +++ + tools/perf/python/twatch.py | 77 +- + tools/perf/python/wakeup-latency.py | 103 + + tools/perf/scripts/Build | 30 - + tools/perf/scripts/install-build-deps.sh | 8 +- + tools/perf/scripts/perl/Perf-Trace-Util/Build | 9 - + tools/perf/scripts/perl/Perf-Trace-Util/Context.c | 122 - + tools/perf/scripts/perl/Perf-Trace-Util/Context.xs | 42 - + .../perf/scripts/perl/Perf-Trace-Util/Makefile.PL | 18 - + tools/perf/scripts/perl/Perf-Trace-Util/README | 59 - + .../perl/Perf-Trace-Util/lib/Perf/Trace/Context.pm | 55 - + .../perl/Perf-Trace-Util/lib/Perf/Trace/Core.pm | 192 - + .../perl/Perf-Trace-Util/lib/Perf/Trace/Util.pm | 94 - + tools/perf/scripts/perl/Perf-Trace-Util/typemap | 1 - + .../perf/scripts/perl/bin/check-perf-trace-record | 2 - + tools/perf/scripts/perl/bin/failed-syscalls-record | 3 - + tools/perf/scripts/perl/bin/failed-syscalls-report | 10 - + tools/perf/scripts/perl/bin/rw-by-file-record | 3 - + tools/perf/scripts/perl/bin/rw-by-file-report | 10 - + tools/perf/scripts/perl/bin/rw-by-pid-record | 2 - + tools/perf/scripts/perl/bin/rw-by-pid-report | 3 - + tools/perf/scripts/perl/bin/rwtop-record | 2 - + tools/perf/scripts/perl/bin/rwtop-report | 20 - + tools/perf/scripts/perl/bin/wakeup-latency-record | 6 - + tools/perf/scripts/perl/bin/wakeup-latency-report | 3 - + tools/perf/scripts/perl/check-perf-trace.pl | 106 - + tools/perf/scripts/perl/failed-syscalls.pl | 47 - + tools/perf/scripts/perl/rw-by-file.pl | 106 - + tools/perf/scripts/perl/rw-by-pid.pl | 184 - + tools/perf/scripts/perl/rwtop.pl | 203 - + tools/perf/scripts/perl/wakeup-latency.pl | 107 - + tools/perf/scripts/python/Perf-Trace-Util/Build | 4 - + .../perf/scripts/python/Perf-Trace-Util/Context.c | 225 - + .../python/Perf-Trace-Util/lib/Perf/Trace/Core.py | 116 - + .../Perf-Trace-Util/lib/Perf/Trace/EventClass.py | 97 - + .../Perf-Trace-Util/lib/Perf/Trace/SchedGui.py | 184 - + .../python/Perf-Trace-Util/lib/Perf/Trace/Util.py | 92 - + tools/perf/scripts/python/arm-cs-trace-disasm.py | 356 -- + .../scripts/python/bin/compaction-times-record | 2 - + .../scripts/python/bin/compaction-times-report | 4 - + .../python/bin/event_analyzing_sample-record | 8 - + .../python/bin/event_analyzing_sample-report | 3 - + .../scripts/python/bin/export-to-postgresql-record | 8 - + .../scripts/python/bin/export-to-postgresql-report | 29 - + .../scripts/python/bin/export-to-sqlite-record | 8 - + .../scripts/python/bin/export-to-sqlite-report | 29 - + .../python/bin/failed-syscalls-by-pid-record | 3 - + .../python/bin/failed-syscalls-by-pid-report | 10 - + tools/perf/scripts/python/bin/flamegraph-record | 2 - + tools/perf/scripts/python/bin/flamegraph-report | 3 - + .../scripts/python/bin/futex-contention-record | 2 - + .../scripts/python/bin/futex-contention-report | 4 - + tools/perf/scripts/python/bin/gecko-record | 2 - + tools/perf/scripts/python/bin/gecko-report | 7 - + .../perf/scripts/python/bin/intel-pt-events-record | 13 - + .../perf/scripts/python/bin/intel-pt-events-report | 3 - + tools/perf/scripts/python/bin/mem-phys-addr-record | 19 - + tools/perf/scripts/python/bin/mem-phys-addr-report | 3 - + .../perf/scripts/python/bin/net_dropmonitor-record | 2 - + .../perf/scripts/python/bin/net_dropmonitor-report | 4 - + tools/perf/scripts/python/bin/netdev-times-record | 8 - + tools/perf/scripts/python/bin/netdev-times-report | 5 - + .../perf/scripts/python/bin/powerpc-hcalls-record | 2 - + .../perf/scripts/python/bin/powerpc-hcalls-report | 2 - + .../perf/scripts/python/bin/sched-migration-record | 2 - + .../perf/scripts/python/bin/sched-migration-report | 3 - + tools/perf/scripts/python/bin/sctop-record | 3 - + tools/perf/scripts/python/bin/sctop-report | 24 - + tools/perf/scripts/python/bin/stackcollapse-record | 8 - + tools/perf/scripts/python/bin/stackcollapse-report | 3 - + .../python/bin/syscall-counts-by-pid-record | 3 - + .../python/bin/syscall-counts-by-pid-report | 10 - + .../perf/scripts/python/bin/syscall-counts-record | 3 - + .../perf/scripts/python/bin/syscall-counts-report | 10 - + tools/perf/scripts/python/bin/task-analyzer-record | 2 - + tools/perf/scripts/python/bin/task-analyzer-report | 3 - + tools/perf/scripts/python/check-perf-trace.py | 84 - + tools/perf/scripts/python/compaction-times.py | 311 -- + .../perf/scripts/python/event_analyzing_sample.py | 192 - + tools/perf/scripts/python/export-to-postgresql.py | 1114 ----- + tools/perf/scripts/python/export-to-sqlite.py | 799 --- + tools/perf/scripts/python/exported-sql-viewer.py | 5030 ------------------- + .../perf/scripts/python/failed-syscalls-by-pid.py | 79 - + tools/perf/scripts/python/flamegraph.py | 267 - + tools/perf/scripts/python/futex-contention.py | 57 - + tools/perf/scripts/python/gecko.py | 395 -- + tools/perf/scripts/python/intel-pt-events.py | 494 -- + tools/perf/scripts/python/libxed.py | 107 - + tools/perf/scripts/python/mem-phys-addr.py | 127 - + tools/perf/scripts/python/net_dropmonitor.py | 78 - + tools/perf/scripts/python/netdev-times.py | 473 -- + tools/perf/scripts/python/parallel-perf.py | 989 ---- + tools/perf/scripts/python/sched-migration.py | 462 -- + tools/perf/scripts/python/sctop.py | 89 - + tools/perf/scripts/python/stackcollapse.py | 127 - + tools/perf/scripts/python/stat-cpi.py | 79 - + tools/perf/scripts/python/syscall-counts-by-pid.py | 75 - + tools/perf/scripts/python/syscall-counts.py | 65 - + tools/perf/scripts/python/task-analyzer.py | 934 ---- + tools/perf/tests/Build | 7 +- + tools/perf/tests/builtin-test.c | 53 +- + tools/perf/tests/code-reading.c | 12 +- + tools/perf/tests/fdarray.c | 32 + + tools/perf/tests/hists_cumulate.c | 2 +- + tools/perf/tests/hists_filter.c | 2 +- + tools/perf/tests/hists_link.c | 2 +- + tools/perf/tests/hists_output.c | 2 +- + tools/perf/tests/hybrid-merge.c | 206 + + tools/perf/tests/make | 114 +- + tools/perf/tests/sample-parsing.c | 109 + + tools/perf/tests/shell/annotate.sh | 1 - + tools/perf/tests/shell/annotate_weight.sh | 63 + + .../tests/shell/base_probe/test_adding_kernel.sh | 1 - + tools/perf/tests/shell/c2c.sh | 114 + + tools/perf/tests/shell/coresight/callchain.sh | 7 +- + .../shell/coresight/test_arm_coresight_disasm.sh | 15 +- + tools/perf/tests/shell/data_type_profiling.sh | 46 +- + tools/perf/tests/shell/diff.sh | 1 - + tools/perf/tests/shell/inject_aslr.sh | 1 - + tools/perf/tests/shell/jitdump-python.sh | 1 - + tools/perf/tests/shell/lib/attr.py | 86 +- + tools/perf/tests/shell/lib/perf_brstack_max.py | 40 + + .../perf/tests/shell/lib/perf_json_output_lint.py | 36 +- + .../perf/tests/shell/lib/perf_metric_validation.py | 53 +- + tools/perf/tests/shell/lib/perf_record.sh | 2 +- + tools/perf/tests/shell/lib/probe_vfs_getname.sh | 12 +- + tools/perf/tests/shell/lib/setup_python.sh | 29 +- + tools/perf/tests/shell/lib/waiting.sh | 34 +- + tools/perf/tests/shell/list.sh | 1 - + tools/perf/tests/shell/pipe_test.sh | 1 - + tools/perf/tests/shell/probe_vfs_getname.sh | 2 - + tools/perf/tests/shell/python-use.sh | 1 - + .../tests/shell/record+probe_libc_inet_pton.sh | 2 - + .../tests/shell/record+script_probe_vfs_getname.sh | 2 - + tools/perf/tests/shell/record.sh | 278 +- + tools/perf/tests/shell/record_weak_term.sh | 1 - + tools/perf/tests/shell/report_hybrid_merge.sh | 121 + + tools/perf/tests/shell/script.sh | 46 +- + tools/perf/tests/shell/script_dlfilter.sh | 1 - + tools/perf/tests/shell/script_perl.sh | 102 - + tools/perf/tests/shell/script_python.sh | 113 - + tools/perf/tests/shell/stat+csv_output.sh | 1 - + tools/perf/tests/shell/stat+json_output.sh | 1 - + tools/perf/tests/shell/stat+std_output.sh | 1 - + tools/perf/tests/shell/stat.sh | 5 + + tools/perf/tests/shell/stat_metrics_cgrp.sh | 7 + + tools/perf/tests/shell/stat_metrics_values.sh | 1 - + tools/perf/tests/shell/test_arm_callgraph_fp.sh | 1 - + tools/perf/tests/shell/test_brstack.sh | 1 - + .../tests/shell/test_check_perf_trace_python.sh | 87 + + .../tests/shell/test_compaction_times_python.sh | 129 + + tools/perf/tests/shell/test_data_symbol.sh | 7 +- + .../shell/test_event_analyzing_sample_python.sh | 70 + + tools/perf/tests/shell/test_event_open_fallback.sh | 2 - + .../shell/test_export_to_postgresql_python.sh | 129 + + .../tests/shell/test_export_to_sqlite_python.sh | 115 + + .../shell/test_failed_syscalls_by_pid_python.sh | 106 + + .../tests/shell/test_failed_syscalls_python.sh | 90 + + tools/perf/tests/shell/test_flamegraph_python.sh | 123 + + .../tests/shell/test_futex_contention_python.sh | 116 + + tools/perf/tests/shell/test_gecko_python.sh | 96 + + tools/perf/tests/shell/test_intel_pt.sh | 46 +- + .../tests/shell/test_intel_pt_events_python.sh | 74 + + .../perf/tests/shell/test_mem_phys_addr_python.sh | 105 + + .../tests/shell/test_net_dropmonitor_python.sh | 103 + + tools/perf/tests/shell/test_netdev_times_python.sh | 104 + + .../tests/shell/test_perf_data_converter_json.sh | 1 - + .../perf/tests/shell/test_powerpc_hcalls_python.sh | 95 + + tools/perf/tests/shell/test_rw_by_file_python.sh | 69 + + tools/perf/tests/shell/test_rw_by_pid_python.sh | 75 + + tools/perf/tests/shell/test_rwtop_python.sh | 74 + + .../tests/shell/test_sched_migration_python.sh | 72 + + tools/perf/tests/shell/test_sctop_python.sh | 78 + + .../perf/tests/shell/test_stackcollapse_python.sh | 78 + + tools/perf/tests/shell/test_stat_cpi_python.sh | 121 + + .../shell/test_syscall_counts_by_pid_python.sh | 93 + + .../perf/tests/shell/test_syscall_counts_python.sh | 82 + + tools/perf/tests/shell/test_task_analyzer.sh | 93 +- + tools/perf/tests/shell/test_test_junit_output.sh | 1 - + .../tests/shell/test_uprobe_from_different_cu.sh | 1 - + .../perf/tests/shell/test_wakeup_latency_python.sh | 83 + + tools/perf/tests/shell/top.sh | 73 +- + tools/perf/tests/shell/trace+probe_vfs_getname.sh | 1 - + tools/perf/tests/shell/trace_btf_enum.sh | 1 - + tools/perf/tests/shell/trace_btf_general.sh | 1 - + tools/perf/tests/shell/trace_exit_race.sh | 1 - + tools/perf/tests/shell/trace_ksym_beautifier.sh | 38 + + tools/perf/tests/shell/trace_record_replay.sh | 2 - + tools/perf/tests/shell/trace_summary.sh | 1 - + tools/perf/tests/symbols.c | 96 +- + tools/perf/tests/tests.h | 1 + + tools/perf/tests/thread-maps-share.c | 3 +- + tools/perf/tests/tool_pmu.c | 79 + + tools/perf/tests/workloads/code_with_type.c | 2 +- + tools/perf/trace/beauty/arch_errno_names.sh | 4 +- + tools/perf/trace/beauty/beauty.h | 3 + + tools/perf/trace/beauty/include/uapi/linux/fs.h | 2 +- + tools/perf/trace/beauty/syscalltbl.sh | 2 +- + tools/perf/ui/browsers/annotate-data.c | 12 +- + tools/perf/ui/browsers/annotate.c | 21 +- + tools/perf/ui/browsers/hists.c | 8 +- + tools/perf/ui/browsers/scripts.c | 218 +- + tools/perf/ui/gtk/annotate.c | 36 +- + tools/perf/ui/gtk/browser.c | 96 +- + tools/perf/ui/gtk/gtk.h | 16 +- + tools/perf/ui/gtk/hists.c | 72 +- + tools/perf/ui/gtk/progress.c | 40 +- + tools/perf/ui/gtk/setup.c | 5 +- + tools/perf/ui/gtk/util.c | 97 +- + tools/perf/ui/hist.c | 248 +- + tools/perf/ui/libslang.h | 2 + + tools/perf/ui/setup.c | 2 +- + tools/perf/util/Build | 10 +- + tools/perf/util/addr2line.c | 42 +- + tools/perf/util/annotate-arch/Build | 1 + + tools/perf/util/annotate-arch/annotate-alpha.c | 185 + + tools/perf/util/annotate-arch/annotate-arm64.c | 435 +- + tools/perf/util/annotate-arch/annotate-loongarch.c | 4 + + tools/perf/util/annotate-arch/annotate-powerpc.c | 9 + + tools/perf/util/annotate-arch/annotate-x86.c | 79 + + tools/perf/util/annotate-data.c | 303 +- + tools/perf/util/annotate-data.h | 17 +- + tools/perf/util/annotate.c | 240 +- + tools/perf/util/annotate.h | 121 +- + tools/perf/util/arm-spe.c | 17 +- + tools/perf/util/aslr.c | 41 +- + tools/perf/util/aslr.h | 2 + + tools/perf/util/auxtrace.c | 12 +- + tools/perf/util/auxtrace.h | 7 +- + tools/perf/util/bpf-filter.l | 1 + + tools/perf/util/bpf_trace_augment.c | 2 + + tools/perf/util/build-id.c | 12 +- + tools/perf/util/c2c-function.c | 9 +- + tools/perf/util/c2c.h | 2 + + tools/perf/util/cache.h | 31 - + tools/perf/util/capstone.c | 186 +- + tools/perf/util/config.c | 52 +- + tools/perf/util/cs-etm.c | 2 +- + tools/perf/util/debug.c | 2 +- + tools/perf/util/debuginfo.c | 51 +- + tools/perf/util/debuginfo.h | 13 +- + tools/perf/util/disasm.c | 28 +- + tools/perf/util/disasm.h | 6 + + tools/perf/util/dso.c | 194 +- + tools/perf/util/dso.h | 59 +- + tools/perf/util/dwarf-aux.c | 247 +- + tools/perf/util/dwarf-aux.h | 16 + + tools/perf/util/dwarf-regs-arch/dwarf-regs-arm64.c | 25 + + tools/perf/util/dwarf-regs-arch/dwarf-regs-csky.c | 2 +- + .../perf/util/dwarf-regs-arch/dwarf-regs-powerpc.c | 2 +- + tools/perf/util/dwarf-regs-arch/dwarf-regs-s390.c | 2 +- + tools/perf/util/dwarf-regs-arch/dwarf-regs-x86.c | 140 +- + tools/perf/util/dwarf-regs.c | 11 +- + tools/perf/util/env.c | 1 + + tools/perf/util/env.h | 10 + + tools/perf/util/evlist.c | 236 +- + tools/perf/util/evlist.h | 3 + + tools/perf/util/evsel.c | 277 +- + tools/perf/util/evsel.h | 9 + + tools/perf/util/expr.c | 5 +- + tools/perf/util/genelf_debug.c | 28 +- + tools/perf/util/header.c | 336 +- + tools/perf/util/header.h | 1 + + tools/perf/util/help-unknown-cmd.c | 19 +- + tools/perf/util/help-unknown-cmd.h | 0 + tools/perf/util/hist.h | 21 +- + tools/perf/util/hwmon_pmu.c | 2 +- + tools/perf/util/include/dwarf-regs.h | 8 +- + tools/perf/util/intel-bts.c | 2 +- + tools/perf/util/intel-pt.c | 3 +- + tools/perf/util/jitdump.c | 204 +- + tools/perf/util/libbfd.c | 94 +- + tools/perf/util/libbfd.h | 9 + + tools/perf/util/libdw.c | 39 +- + tools/perf/util/llvm.c | 56 +- + tools/perf/util/machine.c | 4 +- + tools/perf/util/machine.h | 10 +- + tools/perf/util/mem-events.c | 94 +- + tools/perf/util/mem-events.h | 5 +- + tools/perf/util/metricgroup.c | 37 +- + tools/perf/util/parse-events.c | 5 +- + tools/perf/util/parse-regs-options.c | 211 +- + tools/perf/util/path.c | 10 +- + tools/perf/util/path.h | 6 +- + tools/perf/util/perf-regs-arch/perf_regs_x86.c | 438 +- + tools/perf/util/perf_event_attr_fprintf.c | 13 + + tools/perf/util/perf_regs.c | 84 +- + tools/perf/util/perf_regs.h | 21 +- + tools/perf/util/pmu.c | 48 +- + tools/perf/util/pmu.h | 3 + + tools/perf/util/pmus.c | 21 +- + tools/perf/util/powerpc-vpadtl.c | 2 +- + tools/perf/util/probe-event.c | 4 +- + tools/perf/util/python.c | 1042 +++- + tools/perf/util/record.h | 7 + + tools/perf/util/sample.h | 5 + + tools/perf/util/scripting-engines/Build | 9 - + .../perf/util/scripting-engines/trace-event-perl.c | 770 --- + .../util/scripting-engines/trace-event-python.c | 2224 --------- + tools/perf/util/session.c | 119 +- + tools/perf/util/srcline.c | 66 +- + tools/perf/util/strbuf.c | 14 +- + tools/perf/util/symbol.c | 23 +- + tools/perf/util/symbol_conf.h | 16 +- + tools/perf/util/synthetic-events.c | 56 +- + tools/perf/util/thread.c | 18 +- + tools/perf/util/tool_pmu.c | 2 +- + tools/perf/util/tp_pmu.c | 8 +- + tools/perf/util/trace-event-parse.c | 65 - + tools/perf/util/trace-event-scripting.c | 407 -- + tools/perf/util/trace-event.c | 90 +- + tools/perf/util/trace-event.h | 73 +- + tools/perf/util/trace_augment.h | 1 + + tools/perf/util/unwind-libdw.c | 59 +- + tools/perf/util/unwind-libunwind.c | 62 +- + tools/perf/util/usage.c | 34 - + tools/perf/util/util.h | 4 - + 540 files changed, 32637 insertions(+), 23820 deletions(-) + delete mode 100644 tools/build/feature/test-gtk2-infobar.c + rename tools/build/feature/{test-gtk2.c => test-gtk4.c} (76%) + delete mode 100644 tools/build/feature/test-libperl.c + delete mode 100644 tools/build/feature/test-libpython.c + create mode 100644 tools/build/feature/test-python-module.c + delete mode 100644 tools/perf/Documentation/perf-script-perl.txt + create mode 100644 tools/perf/arch/riscv/include/arch-tests.h + create mode 100644 tools/perf/arch/riscv/tests/Build + create mode 100644 tools/perf/arch/riscv/tests/arch-tests.c + create mode 100644 tools/perf/arch/riscv/tests/dwarf-unwind.c + create mode 100644 tools/perf/arch/riscv/tests/regs_load.S + create mode 100755 tools/perf/python/SchedGui.py + create mode 100755 tools/perf/python/arm-cs-trace-disasm.py + create mode 100755 tools/perf/python/check-perf-trace.py + create mode 100755 tools/perf/python/compaction-times.py + create mode 100755 tools/perf/python/event_analyzing_sample.py + create mode 100755 tools/perf/python/export-to-postgresql.py + create mode 100755 tools/perf/python/export-to-sqlite.py + create mode 100755 tools/perf/python/exported-sql-viewer.py + create mode 100755 tools/perf/python/failed-syscalls-by-pid.py + create mode 100755 tools/perf/python/failed-syscalls.py + create mode 100755 tools/perf/python/flamegraph.py + create mode 100755 tools/perf/python/futex-contention.py + create mode 100755 tools/perf/python/gecko.py + create mode 100755 tools/perf/python/intel-pt-events.py + create mode 100755 tools/perf/python/libxed.py + create mode 100755 tools/perf/python/mem-phys-addr.py + create mode 100755 tools/perf/python/net_dropmonitor.py + create mode 100755 tools/perf/python/netdev-times.py + create mode 100755 tools/perf/python/parallel-perf.py + rename tools/perf/{scripts => }/python/powerpc-hcalls.py (54%) + mode change 100644 => 100755 + create mode 100755 tools/perf/python/rw-by-file.py + create mode 100755 tools/perf/python/rw-by-pid.py + create mode 100755 tools/perf/python/rwtop.py + create mode 100755 tools/perf/python/sched-migration.py + create mode 100755 tools/perf/python/sctop.py + create mode 100755 tools/perf/python/stackcollapse.py + create mode 100755 tools/perf/python/stat-cpi.py + create mode 100755 tools/perf/python/syscall-counts-by-pid.py + create mode 100755 tools/perf/python/syscall-counts.py + create mode 100755 tools/perf/python/task-analyzer.py + create mode 100755 tools/perf/python/treport.py + create mode 100755 tools/perf/python/wakeup-latency.py + delete mode 100644 tools/perf/scripts/Build + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/Build + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/Context.c + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/Context.xs + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/Makefile.PL + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/README + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/lib/Perf/Trace/Context.pm + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/lib/Perf/Trace/Core.pm + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/lib/Perf/Trace/Util.pm + delete mode 100644 tools/perf/scripts/perl/Perf-Trace-Util/typemap + delete mode 100644 tools/perf/scripts/perl/bin/check-perf-trace-record + delete mode 100644 tools/perf/scripts/perl/bin/failed-syscalls-record + delete mode 100644 tools/perf/scripts/perl/bin/failed-syscalls-report + delete mode 100644 tools/perf/scripts/perl/bin/rw-by-file-record + delete mode 100644 tools/perf/scripts/perl/bin/rw-by-file-report + delete mode 100644 tools/perf/scripts/perl/bin/rw-by-pid-record + delete mode 100644 tools/perf/scripts/perl/bin/rw-by-pid-report + delete mode 100644 tools/perf/scripts/perl/bin/rwtop-record + delete mode 100644 tools/perf/scripts/perl/bin/rwtop-report + delete mode 100644 tools/perf/scripts/perl/bin/wakeup-latency-record + delete mode 100644 tools/perf/scripts/perl/bin/wakeup-latency-report + delete mode 100644 tools/perf/scripts/perl/check-perf-trace.pl + delete mode 100644 tools/perf/scripts/perl/failed-syscalls.pl + delete mode 100644 tools/perf/scripts/perl/rw-by-file.pl + delete mode 100644 tools/perf/scripts/perl/rw-by-pid.pl + delete mode 100644 tools/perf/scripts/perl/rwtop.pl + delete mode 100644 tools/perf/scripts/perl/wakeup-latency.pl + delete mode 100644 tools/perf/scripts/python/Perf-Trace-Util/Build + delete mode 100644 tools/perf/scripts/python/Perf-Trace-Util/Context.c + delete mode 100644 tools/perf/scripts/python/Perf-Trace-Util/lib/Perf/Trace/Core.py + delete mode 100755 tools/perf/scripts/python/Perf-Trace-Util/lib/Perf/Trace/EventClass.py + delete mode 100644 tools/perf/scripts/python/Perf-Trace-Util/lib/Perf/Trace/SchedGui.py + delete mode 100644 tools/perf/scripts/python/Perf-Trace-Util/lib/Perf/Trace/Util.py + delete mode 100755 tools/perf/scripts/python/arm-cs-trace-disasm.py + delete mode 100644 tools/perf/scripts/python/bin/compaction-times-record + delete mode 100644 tools/perf/scripts/python/bin/compaction-times-report + delete mode 100644 tools/perf/scripts/python/bin/event_analyzing_sample-record + delete mode 100644 tools/perf/scripts/python/bin/event_analyzing_sample-report + delete mode 100644 tools/perf/scripts/python/bin/export-to-postgresql-record + delete mode 100644 tools/perf/scripts/python/bin/export-to-postgresql-report + delete mode 100644 tools/perf/scripts/python/bin/export-to-sqlite-record + delete mode 100644 tools/perf/scripts/python/bin/export-to-sqlite-report + delete mode 100644 tools/perf/scripts/python/bin/failed-syscalls-by-pid-record + delete mode 100644 tools/perf/scripts/python/bin/failed-syscalls-by-pid-report + delete mode 100755 tools/perf/scripts/python/bin/flamegraph-record + delete mode 100755 tools/perf/scripts/python/bin/flamegraph-report + delete mode 100644 tools/perf/scripts/python/bin/futex-contention-record + delete mode 100644 tools/perf/scripts/python/bin/futex-contention-report + delete mode 100644 tools/perf/scripts/python/bin/gecko-record + delete mode 100755 tools/perf/scripts/python/bin/gecko-report + delete mode 100644 tools/perf/scripts/python/bin/intel-pt-events-record + delete mode 100644 tools/perf/scripts/python/bin/intel-pt-events-report + delete mode 100644 tools/perf/scripts/python/bin/mem-phys-addr-record + delete mode 100644 tools/perf/scripts/python/bin/mem-phys-addr-report + delete mode 100755 tools/perf/scripts/python/bin/net_dropmonitor-record + delete mode 100755 tools/perf/scripts/python/bin/net_dropmonitor-report + delete mode 100644 tools/perf/scripts/python/bin/netdev-times-record + delete mode 100644 tools/perf/scripts/python/bin/netdev-times-report + delete mode 100644 tools/perf/scripts/python/bin/powerpc-hcalls-record + delete mode 100644 tools/perf/scripts/python/bin/powerpc-hcalls-report + delete mode 100644 tools/perf/scripts/python/bin/sched-migration-record + delete mode 100644 tools/perf/scripts/python/bin/sched-migration-report + delete mode 100644 tools/perf/scripts/python/bin/sctop-record + delete mode 100644 tools/perf/scripts/python/bin/sctop-report + delete mode 100755 tools/perf/scripts/python/bin/stackcollapse-record + delete mode 100755 tools/perf/scripts/python/bin/stackcollapse-report + delete mode 100644 tools/perf/scripts/python/bin/syscall-counts-by-pid-record + delete mode 100644 tools/perf/scripts/python/bin/syscall-counts-by-pid-report + delete mode 100644 tools/perf/scripts/python/bin/syscall-counts-record + delete mode 100644 tools/perf/scripts/python/bin/syscall-counts-report + delete mode 100755 tools/perf/scripts/python/bin/task-analyzer-record + delete mode 100755 tools/perf/scripts/python/bin/task-analyzer-report + delete mode 100644 tools/perf/scripts/python/check-perf-trace.py + delete mode 100644 tools/perf/scripts/python/compaction-times.py + delete mode 100644 tools/perf/scripts/python/event_analyzing_sample.py + delete mode 100644 tools/perf/scripts/python/export-to-postgresql.py + delete mode 100644 tools/perf/scripts/python/export-to-sqlite.py + delete mode 100755 tools/perf/scripts/python/exported-sql-viewer.py + delete mode 100644 tools/perf/scripts/python/failed-syscalls-by-pid.py + delete mode 100755 tools/perf/scripts/python/flamegraph.py + delete mode 100644 tools/perf/scripts/python/futex-contention.py + delete mode 100644 tools/perf/scripts/python/gecko.py + delete mode 100644 tools/perf/scripts/python/intel-pt-events.py + delete mode 100644 tools/perf/scripts/python/libxed.py + delete mode 100644 tools/perf/scripts/python/mem-phys-addr.py + delete mode 100755 tools/perf/scripts/python/net_dropmonitor.py + delete mode 100644 tools/perf/scripts/python/netdev-times.py + delete mode 100755 tools/perf/scripts/python/parallel-perf.py + delete mode 100644 tools/perf/scripts/python/sched-migration.py + delete mode 100644 tools/perf/scripts/python/sctop.py + delete mode 100755 tools/perf/scripts/python/stackcollapse.py + delete mode 100644 tools/perf/scripts/python/stat-cpi.py + delete mode 100644 tools/perf/scripts/python/syscall-counts-by-pid.py + delete mode 100644 tools/perf/scripts/python/syscall-counts.py + delete mode 100755 tools/perf/scripts/python/task-analyzer.py + create mode 100644 tools/perf/tests/hybrid-merge.c + create mode 100755 tools/perf/tests/shell/annotate_weight.sh + create mode 100644 tools/perf/tests/shell/lib/perf_brstack_max.py + create mode 100755 tools/perf/tests/shell/report_hybrid_merge.sh + delete mode 100755 tools/perf/tests/shell/script_perl.sh + delete mode 100755 tools/perf/tests/shell/script_python.sh + create mode 100755 tools/perf/tests/shell/test_check_perf_trace_python.sh + create mode 100755 tools/perf/tests/shell/test_compaction_times_python.sh + create mode 100755 tools/perf/tests/shell/test_event_analyzing_sample_python.sh + create mode 100755 tools/perf/tests/shell/test_export_to_postgresql_python.sh + create mode 100755 tools/perf/tests/shell/test_export_to_sqlite_python.sh + create mode 100755 tools/perf/tests/shell/test_failed_syscalls_by_pid_python.sh + create mode 100755 tools/perf/tests/shell/test_failed_syscalls_python.sh + create mode 100755 tools/perf/tests/shell/test_flamegraph_python.sh + create mode 100755 tools/perf/tests/shell/test_futex_contention_python.sh + create mode 100755 tools/perf/tests/shell/test_gecko_python.sh + create mode 100755 tools/perf/tests/shell/test_intel_pt_events_python.sh + create mode 100755 tools/perf/tests/shell/test_mem_phys_addr_python.sh + create mode 100755 tools/perf/tests/shell/test_net_dropmonitor_python.sh + create mode 100755 tools/perf/tests/shell/test_netdev_times_python.sh + create mode 100755 tools/perf/tests/shell/test_powerpc_hcalls_python.sh + create mode 100755 tools/perf/tests/shell/test_rw_by_file_python.sh + create mode 100755 tools/perf/tests/shell/test_rw_by_pid_python.sh + create mode 100755 tools/perf/tests/shell/test_rwtop_python.sh + create mode 100755 tools/perf/tests/shell/test_sched_migration_python.sh + create mode 100755 tools/perf/tests/shell/test_sctop_python.sh + create mode 100755 tools/perf/tests/shell/test_stackcollapse_python.sh + create mode 100755 tools/perf/tests/shell/test_stat_cpi_python.sh + create mode 100755 tools/perf/tests/shell/test_syscall_counts_by_pid_python.sh + create mode 100755 tools/perf/tests/shell/test_syscall_counts_python.sh + create mode 100755 tools/perf/tests/shell/test_wakeup_latency_python.sh + create mode 100755 tools/perf/tests/shell/trace_ksym_beautifier.sh + create mode 100644 tools/perf/util/annotate-arch/annotate-alpha.c + delete mode 100644 tools/perf/util/cache.h + delete mode 100644 tools/perf/util/help-unknown-cmd.h + delete mode 100644 tools/perf/util/scripting-engines/Build + delete mode 100644 tools/perf/util/scripting-engines/trace-event-perl.c + delete mode 100644 tools/perf/util/scripting-engines/trace-event-python.c + delete mode 100644 tools/perf/util/trace-event-scripting.c + delete mode 100644 tools/perf/util/usage.c +Merging compiler-attributes/compiler-attributes (8f0b4cce4481f Linux 6.19-rc1) +$ git merge -m Merge branch 'compiler-attributes' of https://github.com/ojeda/linux.git compiler-attributes/compiler-attributes +Already up to date. +Merging dma-mapping/dma-mapping-for-next (57a57ae077f0f MAINTAINERS: Add files to the DMA MAPPING HELPERS entry) +$ git merge -m Merge branch 'dma-mapping-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mszyprowski/linux.git dma-mapping/dma-mapping-for-next +Auto-merging MAINTAINERS +Auto-merging drivers/iommu/dma-iommu.c +Auto-merging drivers/nvme/host/pci.c +Auto-merging mm/page_alloc.c +Merge made by the 'ort' strategy. + Documentation/core-api/dma-api.rst | 2 +- + MAINTAINERS | 1 + + drivers/iommu/dma-iommu.c | 12 +++++++----- + drivers/nvme/host/pci.c | 2 +- + drivers/scsi/scsi_transport_sas.c | 12 ++++++------ + include/linux/dma-map-ops.h | 4 ++-- + include/linux/dma-mapping.h | 4 ++-- + include/linux/gfp.h | 2 +- + include/linux/iommu-dma.h | 2 +- + kernel/dma/contiguous.c | 6 +++--- + kernel/dma/direct.c | 8 +++++--- + kernel/dma/mapping.c | 10 +++++----- + kernel/dma/ops_helpers.c | 11 +++++++---- + mm/page_alloc.c | 2 +- + 14 files changed, 43 insertions(+), 35 deletions(-) +Merging asm-generic/master (adbbd9714f805 scripts: headers_install.sh: Remove config leak ignore machinery) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/arnd/asm-generic asm-generic/master +Already up to date. +Merging alpha/alpha-next (d58041d2c63e0 MAINTAINERS: Add Magnus Lindholm as maintainer for alpha port) +$ git merge -m Merge branch 'alpha-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mattst88/alpha.git alpha/alpha-next +Already up to date. +Merging arm/for-next (1a89abc009cb5 Merge branches 'fixes' and 'misc' into for-linus) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rmk/linux.git arm/for-next +Already up to date. +Merging arm64/for-next/core (6c360697fd4d1 Merge branches 'for-next/misc', 'for-next/be-gone', 'for-next/smccc-bus' and 'for-next/preemptible-this-cpu' into for-next/core) +$ git merge -m Merge branch 'for-next/core' of https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux arm64/for-next/core +Auto-merging Documentation/arch/arm64/booting.rst +Auto-merging MAINTAINERS +Auto-merging arch/arm64/Kconfig +Auto-merging arch/arm64/include/asm/io.h +Auto-merging arch/arm64/kernel/kexec_image.c +Auto-merging arch/arm64/mm/pageattr.c +CONFLICT (content): Merge conflict in arch/arm64/mm/pageattr.c +Auto-merging arch/arm64/net/bpf_jit_comp.c +Resolved 'arch/arm64/mm/pageattr.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 1e2c745c23532] Merge branch 'for-next/core' of https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux +$ git diff -M --stat --summary HEAD^.. + Documentation/ABI/testing/sysfs-firmware-cca | 7 + + Documentation/arch/arm64/booting.rst | 2 +- + MAINTAINERS | 6 + + arch/arm/include/asm/archrandom.h | 2 +- + arch/arm64/Kconfig | 36 +- + arch/arm64/Makefile | 12 +- + arch/arm64/crypto/aes-ce-ccm-core.S | 4 +- + arch/arm64/crypto/aes-neonbs-core.S | 4 +- + arch/arm64/crypto/ghash-ce-core.S | 6 +- + arch/arm64/crypto/sm4-ce-cipher-core.S | 4 +- + arch/arm64/include/asm/alternative-macros.h | 30 +- + arch/arm64/include/asm/archrandom.h | 2 +- + arch/arm64/include/asm/assembler.h | 31 -- + arch/arm64/include/asm/atomic_ll_sc.h | 27 +- + arch/arm64/include/asm/atomic_lse.h | 4 +- + arch/arm64/include/asm/compat.h | 14 - + arch/arm64/include/asm/elf.h | 12 - + arch/arm64/include/asm/gpr-num.h | 9 + + arch/arm64/include/asm/io.h | 2 +- + arch/arm64/include/asm/mem_encrypt.h | 5 +- + arch/arm64/include/asm/mte-kasan.h | 10 +- + arch/arm64/include/asm/percpu.h | 462 ++++++++++++++++----- + arch/arm64/include/asm/pgtable-prot.h | 4 +- + arch/arm64/include/asm/preempt.h | 31 +- + arch/arm64/include/asm/processor.h | 2 - + arch/arm64/include/asm/ptrace.h | 18 +- + arch/arm64/include/asm/rsi.h | 70 ---- + arch/arm64/include/asm/stage2_pgtable.h | 2 +- + arch/arm64/include/asm/sysreg.h | 52 +-- + arch/arm64/include/asm/thread_info.h | 6 +- + arch/arm64/include/asm/word-at-a-time.h | 6 - + arch/arm64/include/asm/xwreg.h | 16 + + arch/arm64/include/uapi/asm/byteorder.h | 4 - + arch/arm64/include/uapi/asm/ptrace.h | 4 +- + arch/arm64/kernel/Makefile | 2 +- + arch/arm64/kernel/asm-offsets.c | 2 + + arch/arm64/kernel/cpufeature.c | 6 +- + arch/arm64/kernel/entry-common.c | 38 ++ + arch/arm64/kernel/entry.S | 41 +- + arch/arm64/kernel/fpsimd.c | 21 +- + arch/arm64/kernel/head.S | 3 +- + arch/arm64/kernel/image.h | 18 +- + arch/arm64/kernel/kexec_image.c | 5 +- + arch/arm64/kernel/kgdb.c | 6 +- + arch/arm64/kernel/ptrace.c | 8 +- + arch/arm64/kernel/setup.c | 2 +- + arch/arm64/kernel/signal32.c | 13 - + arch/arm64/kernel/sys32.c | 5 - + arch/arm64/kernel/vdso32/Makefile | 4 - + arch/arm64/kvm/hyp/include/nvhe/spinlock.h | 4 - + arch/arm64/kvm/hyp/nvhe/gen-hyprel.c | 16 - + arch/arm64/lib/csum.c | 17 - + arch/arm64/lib/memchr.S | 2 +- + arch/arm64/lib/memcmp.S | 2 - + arch/arm64/lib/strcmp.S | 34 +- + arch/arm64/lib/strlen.S | 26 -- + arch/arm64/lib/strncmp.S | 93 +---- + arch/arm64/lib/strnlen.S | 16 +- + arch/arm64/mm/extable.c | 5 - + arch/arm64/mm/init.c | 3 +- + arch/arm64/mm/pageattr.c | 4 +- + arch/arm64/net/bpf_jit_comp.c | 11 - + drivers/char/hw_random/arm_smccc_trng.c | 32 +- + drivers/crypto/hisilicon/sec2/sec_main.c | 2 +- + drivers/firmware/Kconfig | 1 + + drivers/firmware/Makefile | 1 + + drivers/firmware/arm_rmm/Kconfig | 17 + + drivers/firmware/arm_rmm/Makefile | 2 + + .../kernel => drivers/firmware/arm_rmm}/rsi.c | 79 +++- + drivers/firmware/smccc/Makefile | 2 +- + drivers/firmware/smccc/bus.c | 158 +++++++ + drivers/firmware/smccc/smccc.c | 66 ++- + drivers/hv/Kconfig | 3 +- + drivers/misc/vmw_vmci/Kconfig | 2 +- + drivers/net/ethernet/google/Kconfig | 2 +- + drivers/net/ethernet/microsoft/Kconfig | 2 +- + drivers/soc/tegra/Kconfig | 5 - + drivers/virt/coco/arm-cca-guest/Kconfig | 3 +- + drivers/virt/coco/arm-cca-guest/Makefile | 2 + + .../coco/arm-cca-guest/{arm-cca-guest.c => main.c} | 52 +-- + .../asm/rsi_cmds.h => include/linux/arm-rsi-cmds.h | 76 +++- + include/linux/arm-smccc-bus.h | 48 +++ + .../asm/rsi_smc.h => include/linux/arm-smccc-rsi.h | 6 +- + include/linux/device-id/arm_smccc.h | 19 + + include/linux/mod_devicetable.h | 1 + + scripts/mod/devicetable-offsets.c | 3 + + scripts/mod/file2alias.c | 8 + + tools/testing/selftests/arm64/fp/fp-ptrace.c | 19 +- + .../testcases/fake_sigreturn_sve_change_vl.c | 2 +- + tools/testing/selftests/rseq/rseq-arm64.h | 5 - + 90 files changed, 1107 insertions(+), 824 deletions(-) + create mode 100644 Documentation/ABI/testing/sysfs-firmware-cca + delete mode 100644 arch/arm64/include/asm/rsi.h + create mode 100644 arch/arm64/include/asm/xwreg.h + create mode 100644 drivers/firmware/arm_rmm/Kconfig + create mode 100644 drivers/firmware/arm_rmm/Makefile + rename {arch/arm64/kernel => drivers/firmware/arm_rmm}/rsi.c (71%) + create mode 100644 drivers/firmware/smccc/bus.c + rename drivers/virt/coco/arm-cca-guest/{arm-cca-guest.c => main.c} (82%) + rename arch/arm64/include/asm/rsi_cmds.h => include/linux/arm-rsi-cmds.h (69%) + create mode 100644 include/linux/arm-smccc-bus.h + rename arch/arm64/include/asm/rsi_smc.h => include/linux/arm-smccc-rsi.h (98%) + create mode 100644 include/linux/device-id/arm_smccc.h +$ git am -3 ../patches/0001-arm64-fixup-merge-with-dropped-copy-of-code-getting-.patch +Applying: arm64: fixup merge with dropped copy of code getting kept +$ git reset HEAD^ +Unstaged changes after reset: +M arch/arm64/mm/pageattr.c +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master b7c7aa664e16b] Merge branch 'for-next/core' of https://git.kernel.org/pub/scm/linux/kernel/git/arm64/linux + Date: Wed Sep 30 12:47:14 2026 +0100 +Merging arm-perf/for-next/perf (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next/perf' of https://git.kernel.org/pub/scm/linux/kernel/git/will/linux.git arm-perf/for-next/perf +Already up to date. +Merging arm-soc/for-next (b9bb4134a9583 soc: document merges) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/soc/soc.git arm-soc/for-next +Merge made by the 'ort' strategy. + arch/arm/arm-soc-for-next-contents.txt | 22 ++++++++++++++++++++++ + 1 file changed, 22 insertions(+) + create mode 100644 arch/arm/arm-soc-for-next-contents.txt +Merging amlogic/for-next (a2530c83c458a Merge branch 'v7.4/arm64-dt' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/amlogic/linux.git amlogic/for-next +Merge made by the 'ort' strategy. + arch/arm64/boot/dts/amlogic/amlogic-t7.dtsi | 15 +++++++++------ + .../arm64/boot/dts/amlogic/meson-g12b-odroid-go-ultra.dts | 2 +- + arch/arm64/boot/dts/amlogic/meson-gxbb-nanopi-k2.dts | 2 +- + arch/arm64/boot/dts/amlogic/meson-gxbb.dtsi | 2 +- + arch/arm64/boot/dts/amlogic/meson-gxl.dtsi | 2 +- + 5 files changed, 13 insertions(+), 10 deletions(-) +Merging asahi-soc/asahi-soc/for-next (9378cd5ddcebc Merge branch 'apple-soc/drivers-7.4' into asahi-soc/for-next) +$ git merge -m Merge branch 'asahi-soc/for-next' of https://github.com/AsahiLinux/linux.git asahi-soc/asahi-soc/for-next +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/apple.yaml | 32 + + .../devicetree/bindings/arm/apple/apple,pmgr.yaml | 2 + + Documentation/devicetree/bindings/arm/cpus.yaml | 4 + + .../devicetree/bindings/i2c/apple,i2c.yaml | 1 + + .../bindings/interrupt-controller/apple,aic2.yaml | 2 + + .../devicetree/bindings/pinctrl/apple,pinctrl.yaml | 1 + + .../bindings/power/apple,pmgr-pwrstate.yaml | 2 + + .../devicetree/bindings/pwm/apple,s5l-fpwm.yaml | 1 + + arch/arm64/Kconfig.platforms | 1 + + arch/arm64/boot/dts/apple/Makefile | 7 + + arch/arm64/boot/dts/apple/t6001.dtsi | 11 +- + arch/arm64/boot/dts/apple/t6002.dtsi | 18 +- + arch/arm64/boot/dts/apple/t600x-die0.dtsi | 3 +- + arch/arm64/boot/dts/apple/t600x-dieX.dtsi | 2 + + arch/arm64/boot/dts/apple/t600x-nvme.dtsi | 2 + + arch/arm64/boot/dts/apple/t6021.dtsi | 10 +- + arch/arm64/boot/dts/apple/t6022.dtsi | 18 +- + arch/arm64/boot/dts/apple/t602x-common.dtsi | 2 +- + arch/arm64/boot/dts/apple/t602x-die0.dtsi | 2 + + arch/arm64/boot/dts/apple/t602x-dieX.dtsi | 2 + + arch/arm64/boot/dts/apple/t602x-nvme.dtsi | 2 + + arch/arm64/boot/dts/apple/t6030-pmgr.dtsi | 1 - + arch/arm64/boot/dts/apple/t6030.dtsi | 277 ++--- + arch/arm64/boot/dts/apple/t6031-base.dtsi | 191 ++-- + arch/arm64/boot/dts/apple/t6031-die0.dtsi | 63 +- + arch/arm64/boot/dts/apple/t6031-dieX.dtsi | 94 +- + arch/arm64/boot/dts/apple/t6031-gpio-pins.dtsi | 18 +- + arch/arm64/boot/dts/apple/t6031-pmgr.dtsi | 6 + + arch/arm64/boot/dts/apple/t6031.dtsi | 16 +- + arch/arm64/boot/dts/apple/t6032-j575d.dts | 10 +- + arch/arm64/boot/dts/apple/t6032.dtsi | 193 ++-- + arch/arm64/boot/dts/apple/t603x-j514-j516.dtsi | 15 +- + arch/arm64/boot/dts/apple/t8122-j504.dts | 10 +- + arch/arm64/boot/dts/apple/t8122-j613.dts | 9 +- + arch/arm64/boot/dts/apple/t8122-j615.dts | 9 +- + arch/arm64/boot/dts/apple/t8122-jxxx.dtsi | 5 +- + arch/arm64/boot/dts/apple/t8122-pmgr.dtsi | 1 - + arch/arm64/boot/dts/apple/t8122.dtsi | 256 +++-- + arch/arm64/boot/dts/apple/t8132-j604.dts | 35 + + arch/arm64/boot/dts/apple/t8132-j623.dts | 18 + + arch/arm64/boot/dts/apple/t8132-j624.dts | 18 + + arch/arm64/boot/dts/apple/t8132-j713.dts | 35 + + arch/arm64/boot/dts/apple/t8132-j715.dts | 35 + + arch/arm64/boot/dts/apple/t8132-j773g.dts | 25 + + arch/arm64/boot/dts/apple/t8132-jxxx.dtsi | 48 + + arch/arm64/boot/dts/apple/t8132-pmgr.dtsi | 1130 ++++++++++++++++++++ + arch/arm64/boot/dts/apple/t8132.dtsi | 467 ++++++++ + arch/arm64/boot/dts/apple/t8140-j700.dts | 54 + + arch/arm64/boot/dts/apple/t8140-pmgr.dtsi | 777 ++++++++++++++ + arch/arm64/boot/dts/apple/t8140.dtsi | 207 ++++ + drivers/soc/apple/rtkit-crashlog.c | 141 ++- + 51 files changed, 3715 insertions(+), 574 deletions(-) + create mode 100644 arch/arm64/boot/dts/apple/t8132-j604.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j623.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j624.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j713.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j715.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-j773g.dts + create mode 100644 arch/arm64/boot/dts/apple/t8132-jxxx.dtsi + create mode 100644 arch/arm64/boot/dts/apple/t8132-pmgr.dtsi + create mode 100644 arch/arm64/boot/dts/apple/t8132.dtsi + create mode 100644 arch/arm64/boot/dts/apple/t8140-j700.dts + create mode 100644 arch/arm64/boot/dts/apple/t8140-pmgr.dtsi + create mode 100644 arch/arm64/boot/dts/apple/t8140.dtsi +Merging at91/at91-next (bd49a1c72abbd Merge branch 'microchip-dt64' into at91-next) +$ git merge -m Merge branch 'at91-next' of https://git.kernel.org/pub/scm/linux/kernel/git/at91/linux.git at91/at91-next +Merge made by the 'ort' strategy. + arch/arm/mach-at91/pm.h | 2 +- + arch/arm64/boot/dts/microchip/lan9696-ev23x71a.dts | 13 +++++++++++++ + drivers/clk/at91/clk-main.c | 2 +- + drivers/soc/atmel/soc.c | 4 +--- + 4 files changed, 16 insertions(+), 5 deletions(-) +Merging bmc/for-next (cd7d1ef7d74ed Merge branches 'aspeed/drivers', 'aspeed/arm/dt', 'aspeed/fixes/drivers', 'aspeed/maintainers', 'nuvoton/arm/dt', 'nuvoton/arm/fixes' and 'nuvoton/arm64/dt' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/bmc/linux.git bmc/for-next +Merge made by the 'ort' strategy. +Merging broadcom/next (c0f2033a9ce4b Merge branch 'devicetree-arm64/next' into next) +$ git merge -m Merge branch 'next' of https://github.com/Broadcom/stblinux.git broadcom/next +Merge made by the 'ort' strategy. + .../devicetree/bindings/arm/bcm/bcm2835.yaml | 6 ++ + arch/arm/boot/dts/broadcom/bcm-ns.dtsi | 13 +++ + .../dts/broadcom/bcm4708-linksys-ea6300-v1.dts | 4 + + .../boot/dts/broadcom/bcm4708-smartrg-sr400ac.dts | 18 ++++ + .../boot/dts/broadcom/bcm4709-linksys-ea9200.dts | 62 +++++++++++ + .../boot/dts/broadcom/bcm4709-netgear-r8000.dts | 12 +++ + .../boot/dts/broadcom/bcm47094-dlink-dir-890l.dts | 6 +- + .../dts/broadcom/bcm47094-linksys-panamera.dts | 2 +- + arch/arm/boot/dts/broadcom/bcm47189-tenda-ac9.dts | 20 ++++ + arch/arm/boot/dts/broadcom/bcm53573.dtsi | 9 +- + arch/arm/boot/dts/broadcom/bcm7445.dtsi | 2 +- + arch/arm64/boot/dts/broadcom/bcm2712-d-rpi-5-b.dts | 5 + + .../boot/dts/broadcom/bcm2712-rpi-5-b-base.dtsi | 41 +++++++ + arch/arm64/boot/dts/broadcom/bcm2712.dtsi | 77 +++++++++++++ + arch/arm64/boot/dts/broadcom/rp1-common.dtsi | 119 +++++++++++++++++++++ + 15 files changed, 392 insertions(+), 4 deletions(-) +Merging cix/for-next (a0cffbd8878c5 Merge remote-tracking branch 'cix/dt' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/peter.chen/cix.git cix/for-next +Merge made by the 'ort' strategy. +Merging davinci/davinci/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'davinci/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git davinci/davinci/for-next +Already up to date. +Merging drivers-memory/for-next (a22355280361d memory: stm32_omm: fix child clock leak on set_amcr() error path) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-mem-ctrl.git drivers-memory/for-next +Merge made by the 'ort' strategy. + .../memory-controllers/mediatek,smi-common.yaml | 21 +- + .../memory-controllers/mediatek,smi-larb.yaml | 3 + + drivers/memory/brcmstb_dpfe.c | 41 ++- + drivers/memory/emif.c | 6 +- + drivers/memory/emif.h | 4 +- + drivers/memory/mtk-smi.c | 45 +++ + drivers/memory/renesas-rpc-if.c | 87 ++--- + drivers/memory/samsung/exynos5422-dmc.c | 11 +- + drivers/memory/stm32_omm.c | 2 + + drivers/memory/tegra/tegra264.c | 392 +++++++++++++-------- + 10 files changed, 398 insertions(+), 214 deletions(-) +Merging fsl/soc_fsl (7b3b0598c00e6 soc: fsl: cpm1: tsa: Fix clock leak in tsa_of_parse_tdms()) +$ git merge -m Merge branch 'soc_fsl' of https://git.kernel.org/pub/scm/linux/kernel/git/chleroy/linux.git fsl/soc_fsl +Merge made by the 'ort' strategy. + drivers/bus/fsl-mc/fsl-mc-bus.c | 47 +++++++++++++++++------------ + drivers/bus/fsl-mc/fsl-mc-msi.c | 2 +- + drivers/soc/fsl/qe/gpio.c | 67 ++++++++++++++++------------------------- + drivers/soc/fsl/qe/tsa.c | 8 ++--- + drivers/usb/host/fhci-hcd.c | 2 +- + include/soc/fsl/qe/qe.h | 6 ++-- + 6 files changed, 64 insertions(+), 68 deletions(-) +Merging imx-mxs/for-next (5d034f17b2464 Merge branch 'imx/dt64' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/frank.li/linux.git imx-mxs/for-next +Merge made by the 'ort' strategy. + Documentation/ABI/testing/se-cdev | 45 + + Documentation/devicetree/bindings/arm/fsl.yaml | 68 +- + .../bindings/display/bridge/fsl,ldb.yaml | 23 +- + .../devicetree/bindings/display/fsl,lcdif.yaml | 1 + + .../devicetree/bindings/firmware/fsl,imx-se.yaml | 91 + + .../bindings/soc/imx/fsl,imx-iomuxc-gpr.yaml | 62 + + .../devicetree/bindings/vendor-prefixes.yaml | 2 + + .../driver-api/firmware/other_interfaces.rst | 238 ++ + arch/arm/boot/dts/nxp/imx/Makefile | 49 + + arch/arm/boot/dts/nxp/imx/e60k02.dtsi | 8 +- + arch/arm/boot/dts/nxp/imx/e70k02.dtsi | 8 +- + arch/arm/boot/dts/nxp/imx/imx50-kobo-aura.dts | 7 +- + arch/arm/boot/dts/nxp/imx/imx51-zii-rdu1.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx53-qsb.dts | 2 +- + .../arm/boot/dts/nxp/imx/imx53-voipac-dmm-668.dtsi | 2 +- + arch/arm/boot/dts/nxp/imx/imx6dl-b125pv2.dts | 2 + + arch/arm/boot/dts/nxp/imx/imx6dl-b125v2.dts | 2 + + arch/arm/boot/dts/nxp/imx/imx6dl-mamoj.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6dl-plym2m.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6dl-prtvt7.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6dl-victgo.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6q-bosch-acc.dts | 4 +- + arch/arm/boot/dts/nxp/imx/imx6qdl-emcon.dtsi | 1 - + arch/arm/boot/dts/nxp/imx/imx6qdl-gw560x.dtsi | 20 +- + arch/arm/boot/dts/nxp/imx/imx6qdl-pico.dtsi | 4 +- + arch/arm/boot/dts/nxp/imx/imx6qdl-skov-cpu.dtsi | 2 +- + arch/arm/boot/dts/nxp/imx/imx6qdl-wandboard.dtsi | 4 +- + arch/arm/boot/dts/nxp/imx/imx6qdl-zii-rdu2.dtsi | 10 +- + arch/arm/boot/dts/nxp/imx/imx6qdl.dtsi | 10 +- + arch/arm/boot/dts/nxp/imx/imx6sl-kobo-aura2.dts | 8 +- + .../boot/dts/nxp/imx/imx6sl-tolino-shine2hd.dts | 8 +- + arch/arm/boot/dts/nxp/imx/imx6sll-evk.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx6sll.dtsi | 11 - + arch/arm/boot/dts/nxp/imx/imx6ul-14x14-evk.dtsi | 1 - + .../boot/dts/nxp/imx/imx6ul-kontron-bl-common.dtsi | 38 +- + arch/arm/boot/dts/nxp/imx/imx6ul-tx6ul.dtsi | 2 +- + .../arm/boot/dts/nxp/imx/imx7-colibri-iris-v2.dtsi | 1 - + arch/arm/boot/dts/nxp/imx/imx7-tqma7.dtsi | 52 +- + ...nel-cap-touch-7inch-parallel-touch-adapter.dtso | 58 + + ...olibri-emmc-panel-cap-touch-7inch-parallel.dtso | 43 + + ...olibri-emmc-panel-res-touch-7inch-parallel.dtso | 32 + + .../boot/dts/nxp/imx/imx7d-colibri-emmc-vga.dtso | 59 + + arch/arm/boot/dts/nxp/imx/imx7d-meerkat96.dts | 2 +- + arch/arm/boot/dts/nxp/imx/imx7d-nitrogen7.dts | 1 - + .../nxp/imx/imx7d-var-som-emmc-mx7customboard.dts | 23 + + .../imx7d-var-som-emmc-wm8731-mx7customboard.dts | 23 + + arch/arm/boot/dts/nxp/imx/imx7d-var-som-emmc.dtsi | 56 + + .../dts/nxp/imx/imx7d-var-som-mx7customboard.dtsi | 358 +++ + .../nxp/imx/imx7d-var-som-nand-mx7customboard.dts | 23 + + .../imx7d-var-som-nand-wm8731-mx7customboard.dts | 23 + + arch/arm/boot/dts/nxp/imx/imx7d-var-som-nand.dtsi | 73 + + .../imx/imx7d-var-som-v2-emmc-mx7customboard.dts | 24 + + .../imx/imx7d-var-som-v2-nand-mx7customboard.dts | 24 + + arch/arm/boot/dts/nxp/imx/imx7d-var-som-v2.dtsi | 66 + + .../arm/boot/dts/nxp/imx/imx7d-var-som-wm8731.dtsi | 102 + + .../arm/boot/dts/nxp/imx/imx7d-var-som-wm8904.dtsi | 118 + + arch/arm/boot/dts/nxp/imx/imx7d-var-som.dtsi | 533 +++++ + arch/arm/boot/dts/nxp/imx/imx7s-warp.dts | 2 +- + arch/arm/boot/dts/nxp/imx/mba6ulx.dtsi | 36 +- + arch/arm/boot/dts/nxp/ls/ls1021a.dtsi | 10 +- + arch/arm/boot/dts/nxp/vf/Makefile | 2 + + arch/arm/boot/dts/nxp/vf/vf-colibri-iris.dtsi | 126 + + arch/arm/boot/dts/nxp/vf/vf-colibri.dtsi | 56 + + arch/arm/boot/dts/nxp/vf/vf500-colibri-iris.dts | 14 + + arch/arm/boot/dts/nxp/vf/vf500.dtsi | 6 + + arch/arm/boot/dts/nxp/vf/vf610-colibri-iris.dts | 14 + + arch/arm/boot/dts/nxp/vf/vf610.dtsi | 3 + + arch/arm/include/asm/linkage.h | 29 + + arch/arm/mach-imx/hardware.h | 2 +- + arch/arm/mach-imx/mach-imx6q.c | 30 - + arch/arm/mach-imx/mxc.h | 2 +- + arch/arm/mach-imx/pm-imx6.c | 26 +- + arch/arm/mach-imx/suspend-imx6.S | 6 +- + arch/arm64/boot/dts/freescale/Makefile | 266 ++- + arch/arm64/boot/dts/freescale/fsl-ls1028a.dtsi | 2 +- + .../arm64/boot/dts/freescale/fsl-lx2160a-nbxv3.dts | 34 + + .../boot/dts/freescale/fsl-lx2160a-nbxv3.dtsi | 330 +++ + arch/arm64/boot/dts/freescale/fsl-lx216x.dtsi | 19 +- + .../arm64/boot/dts/freescale/imx8-apalis-eval.dtsi | 6 +- + .../boot/dts/freescale/imx8-apalis-ixora-v1.1.dtsi | 6 +- + .../boot/dts/freescale/imx8-apalis-ixora-v1.2.dtsi | 6 +- + .../arm64/boot/dts/freescale/imx8-apalis-v1.1.dtsi | 16 +- + arch/arm64/boot/dts/freescale/imx8dxl-evk.dts | 14 +- + arch/arm64/boot/dts/freescale/imx8dxl-sr-som.dtsi | 14 +- + .../dts/freescale/imx8mm-beacon-baseboard.dtsi | 2 +- + .../imx8mm-data-modul-edm-sbc-overlay-cm4.dtso | 56 + + ...odul-edm-sbc-overlay-edm-mod-imx8mm-common.dtsi | 59 + + ...-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtsi | 99 + + ...-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtso | 79 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtsi | 69 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtso | 59 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtsi | 94 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtso | 17 + + ...edm-sbc-overlay-edm-mod-imx8mm-lvds-common.dtsi | 118 + + ...l-edm-sbc-overlay-edm-mod-imx8mm-lvds-dual.dtsi | 32 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtsi | 12 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtsi | 12 + + ...bc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtsi | 12 + + ...bc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtsi | 12 + + ...bc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtso | 7 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtsi | 12 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtso | 7 + + ...c-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtsi | 12 + + ...c-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtso | 7 + + ...verlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtsi | 12 + + ...verlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtso | 7 + + ...edm-sbc-overlay-edm-mod-imx8mm-lvds-single.dtsi | 20 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds.dtsi | 22 + + ...odul-edm-sbc-overlay-edm-sbc-imx8mm-rev900.dtso | 18 + + ...imx8mm-data-modul-edm-sbc-overlay-lvds-3v3.dtsi | 19 + + ...imx8mm-data-modul-edm-sbc-overlay-lvds-5v0.dtsi | 19 + + ...mx8mm-data-modul-edm-sbc-overlay-lvds-dual.dtsi | 29 + + ...data-modul-edm-sbc-overlay-lvds-g070y2-l01.dtsi | 31 + + ...ata-modul-edm-sbc-overlay-lvds-g101ice-l01.dtsi | 31 + + ...ata-modul-edm-sbc-overlay-lvds-g121xce-l01.dtsi | 31 + + ...ata-modul-edm-sbc-overlay-lvds-g156hce-l01.dtsi | 31 + + ...data-modul-edm-sbc-overlay-lvds-g215hvn011.dtsi | 30 + + ...ta-modul-edm-sbc-overlay-lvds-mi0700a2t-30.dtsi | 31 + + ...modul-edm-sbc-overlay-lvds-mi1010z1t-1cp11.dtsi | 31 + + ...8mm-data-modul-edm-sbc-overlay-lvds-single.dtsi | 13 + + .../dts/freescale/imx8mm-data-modul-edm-sbc.dts | 64 +- + .../arm64/boot/dts/freescale/imx8mm-kontron-bl.dts | 107 +- + arch/arm64/boot/dts/freescale/imx8mm-var-dart.dtsi | 21 +- + .../boot/dts/freescale/imx8mm-verdin-ivy.dtsi | 4 +- + .../dts/freescale/imx8mn-solidsense-n8-compact.dts | 2 - + .../freescale/imx8mn-vhip4-evalboard-common.dtsi | 3 +- + .../dts/freescale/imx8mn-vhip4-evalboard-v1.dts | 8 + + .../dts/freescale/imx8mn-vhip4-evalboard-v2.dts | 28 +- + .../imx8mn-vhip4-overlay-eeprom-1000.dtso | 103 + + .../imx8mn-vhip4-overlay-eeprom-1100.dtso | 103 + + .../imx8mn-vhip4-overlay-eeprom-2000.dtso | 107 + + .../imx8mp-data-modul-edm-sbc-overlay-cm7.dtso | 57 + + ...-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtso | 67 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtso | 46 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtso | 17 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtso | 7 + + ...bc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtso | 7 + + ...sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtso | 11 + + ...c-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtso | 7 + + ...verlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtso | 7 + + ...-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds.dtsi | 35 + + ...sbc-overlay-edm-sbc-imx8mp-lvds-g070y2-l01.dtso | 24 + + ...bc-overlay-edm-sbc-imx8mp-lvds-g101ice-l01.dtso | 24 + + ...bc-overlay-edm-sbc-imx8mp-lvds-g121xce-l01.dtso | 24 + + ...bc-overlay-edm-sbc-imx8mp-lvds-g156hce-l01.dtso | 32 + + ...sbc-overlay-edm-sbc-imx8mp-lvds-g215hvn011.dtso | 36 + + ...c-overlay-edm-sbc-imx8mp-lvds-mi0700a2t-30.dtso | 24 + + ...verlay-edm-sbc-imx8mp-lvds-mi1010z1t-1cp11.dtso | 24 + + ...edm-sbc-overlay-edm-sbc-imx8mp-lvds-rev900.dtso | 41 + + ...edm-sbc-overlay-edm-sbc-imx8mp-lvds-rev902.dtso | 14 + + ...-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds.dtsi | 79 + + ...odul-edm-sbc-overlay-edm-sbc-imx8mp-rev900.dtso | 97 + + ...odul-edm-sbc-overlay-edm-sbc-imx8mp-rev902.dtso | 69 + + .../dts/freescale/imx8mp-data-modul-edm-sbc.dts | 176 +- + .../dts/freescale/imx8mp-icore-mx8mp-edimm2.2.dts | 2 +- + .../boot/dts/freescale/imx8mp-icore-mx8mp.dtsi | 2 +- + .../boot/dts/freescale/imx8mp-kontron-dl.dtso | 11 + + .../imx8mp-phyboard-pollux-peb-av-10.dtsi | 5 +- + .../imx8mp-phyboard-pollux-peb-wlbt-07.dtso | 76 + + .../boot/dts/freescale/imx8mp-toradex-smarc.dtsi | 2 +- + ...8mp-tqma8mpqs-mb-smarc-2-lvds0-tm070jvhg33.dtso | 2 +- + ...8mp-tqma8mpqs-mb-smarc-2-lvds1-tm070jvhg33.dtso | 2 +- + .../boot/dts/freescale/imx8mp-var-dart-sonata.dts | 1 + + .../boot/dts/freescale/imx8mp-verdin-ivy.dtsi | 4 +- + .../boot/dts/freescale/imx8mp-verdin-yavia.dtsi | 7 +- + arch/arm64/boot/dts/freescale/imx8mp.dtsi | 2 +- + arch/arm64/boot/dts/freescale/imx8mq-evk.dts | 6 + + .../boot/dts/freescale/imx8mq-librem5-devkit.dts | 4 +- + arch/arm64/boot/dts/freescale/imx8mq-librem5.dtsi | 4 +- + arch/arm64/boot/dts/freescale/imx8mq-nitrogen.dts | 2 +- + .../arm64/boot/dts/freescale/imx8mq-zii-ultra.dtsi | 16 +- + arch/arm64/boot/dts/freescale/imx8mq.dtsi | 168 +- + arch/arm64/boot/dts/freescale/imx8qm-mek.dts | 32 +- + arch/arm64/boot/dts/freescale/imx8qm-ss-conn.dtsi | 39 + + .../boot/dts/freescale/imx8qm-var-som-symphony.dts | 2 +- + arch/arm64/boot/dts/freescale/imx8qm-var-som.dtsi | 6 +- + arch/arm64/boot/dts/freescale/imx8qxp-mek.dts | 20 +- + arch/arm64/boot/dts/freescale/imx8ulp-evk.dts | 7 +- + .../arm64/boot/dts/freescale/imx8ulp-firmware.dtsi | 31 + + arch/arm64/boot/dts/freescale/imx8ulp.dtsi | 12 +- + arch/arm64/boot/dts/freescale/imx9-mqs.dtso | 57 + + arch/arm64/boot/dts/freescale/imx91-11x11-evk.dts | 55 +- + .../boot/dts/freescale/imx91-11x11-frdm-s.dts | 2 - + arch/arm64/boot/dts/freescale/imx91-11x11-frdm.dts | 2 - + .../freescale/imx91-9x9-qsb-tianma-tm050rdh03.dtso | 104 + + arch/arm64/boot/dts/freescale/imx91-9x9-qsb.dts | 52 + + .../dts/freescale/imx91-lino-verdin-dahlia.dts | 24 + + .../boot/dts/freescale/imx91-lino-verdin-dev.dts | 24 + + arch/arm64/boot/dts/freescale/imx91-lino.dtsi | 305 +++ + .../boot/dts/freescale/imx91-toradex-osm-dev.dts | 22 + + .../boot/dts/freescale/imx91-toradex-osm.dtsi | 303 +++ + arch/arm64/boot/dts/freescale/imx91.dtsi | 11 + + arch/arm64/boot/dts/freescale/imx91_93_common.dtsi | 6 + + .../boot/dts/freescale/imx93-11x11-evk-common.dtsi | 5 +- + arch/arm64/boot/dts/freescale/imx93-11x11-evk.dts | 54 +- + .../dts/freescale/imx93-11x11-frdm-common.dtsi | 681 ++++++ + arch/arm64/boot/dts/freescale/imx93-11x11-frdm.dts | 682 +----- + arch/arm64/boot/dts/freescale/imx93-14x14-evk.dts | 59 +- + arch/arm64/boot/dts/freescale/imx93-9x9-qsb.dts | 7 +- + .../dts/freescale/imx93-imx91-lino-common.dtsi | 536 +++++ + .../freescale/imx93-imx91-lino-verdin-common.dtsi | 175 ++ + .../imx93-imx91-lino-verdin-dahlia-common.dtsi | 198 ++ + .../imx93-imx91-lino-verdin-dev-common.dtsi | 216 ++ + .../freescale/imx93-imx91-toradex-osm-common.dtsi | 526 +++++ + .../imx93-imx91-toradex-osm-dev-common.dtsi | 324 +++ + .../boot/dts/freescale/imx93-kontron-bl-osm-s.dts | 6 +- + .../boot/dts/freescale/imx93-kontron-osm-s.dtsi | 32 +- + .../dts/freescale/imx93-lino-verdin-dahlia.dts | 24 + + .../boot/dts/freescale/imx93-lino-verdin-dev.dts | 24 + + arch/arm64/boot/dts/freescale/imx93-lino.dtsi | 359 +++ + .../boot/dts/freescale/imx93-toradex-osm-dev.dts | 22 + + .../boot/dts/freescale/imx93-toradex-osm.dtsi | 357 +++ + arch/arm64/boot/dts/freescale/imx93w-frdm.dts | 23 + + arch/arm64/boot/dts/freescale/imx94.dtsi | 5 +- + .../boot/dts/freescale/imx943-evk-sdwifi.dtso | 25 + + arch/arm64/boot/dts/freescale/imx943-evk.dts | 138 +- + arch/arm64/boot/dts/freescale/imx943.dtsi | 2 +- + arch/arm64/boot/dts/freescale/imx95-15x15-evk.dts | 48 +- + arch/arm64/boot/dts/freescale/imx95-15x15-frdm.dts | 38 +- + arch/arm64/boot/dts/freescale/imx95-19x19-evk.dts | 55 +- + .../boot/dts/freescale/imx95-19x19-frdm-pro.dts | 2 - + .../boot/dts/freescale/imx95-aquila-clover.dts | 18 + + arch/arm64/boot/dts/freescale/imx95-aquila-dev.dts | 18 + + arch/arm64/boot/dts/freescale/imx95-aquila.dtsi | 15 +- + .../boot/dts/freescale/imx95-toradex-osm-dev.dts | 404 ++++ + .../boot/dts/freescale/imx95-toradex-osm.dtsi | 1032 +++++++++ + .../boot/dts/freescale/imx95-toradex-smarc-dev.dts | 8 + + .../boot/dts/freescale/imx95-toradex-smarc.dtsi | 8 + + .../boot/dts/freescale/imx95-var-dart-sonata.dts | 2 +- + arch/arm64/boot/dts/freescale/imx95.dtsi | 18 +- + arch/arm64/boot/dts/freescale/imx952-evk.dts | 77 +- + arch/arm64/boot/dts/freescale/imx952-frdm.dts | 9 + + arch/arm64/boot/dts/freescale/imx952-frdm.dtsi | 726 ++++++ + arch/arm64/boot/dts/freescale/imx952.dtsi | 19 +- + arch/arm64/boot/dts/freescale/s32g2.dtsi | 2 +- + arch/arm64/boot/dts/freescale/s32g3.dtsi | 2 +- + drivers/firmware/imx/Kconfig | 12 + + drivers/firmware/imx/Makefile | 2 + + drivers/firmware/imx/ele_base_msg.c | 395 ++++ + drivers/firmware/imx/ele_base_msg.h | 119 + + drivers/firmware/imx/ele_common.c | 1012 ++++++++ + drivers/firmware/imx/ele_common.h | 147 ++ + drivers/firmware/imx/ele_fw_api.c | 406 ++++ + drivers/firmware/imx/ele_fw_api.h | 104 + + drivers/firmware/imx/ele_msg_addr_field.c | 619 +++++ + drivers/firmware/imx/imx-dsp.c | 6 +- + drivers/firmware/imx/se_ctrl.c | 2434 ++++++++++++++++++++ + drivers/firmware/imx/se_ctrl.h | 309 +++ + include/linux/firmware/imx/se_api.h | 14 + + include/linux/micrel_phy.h | 5 - + include/soc/imx/cpu.h | 2 +- + include/uapi/linux/se_ioctl.h | 97 + + 257 files changed, 19003 insertions(+), 1271 deletions(-) + create mode 100644 Documentation/ABI/testing/se-cdev + create mode 100644 Documentation/devicetree/bindings/firmware/fsl,imx-se.yaml + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel-touch-adapter.dtso + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-cap-touch-7inch-parallel.dtso + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-panel-res-touch-7inch-parallel.dtso + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-colibri-emmc-vga.dtso + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-emmc-mx7customboard.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-emmc-wm8731-mx7customboard.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-emmc.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-mx7customboard.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-nand-mx7customboard.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-nand-wm8731-mx7customboard.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-nand.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-v2-emmc-mx7customboard.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-v2-nand-mx7customboard.dts + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-v2.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-wm8731.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som-wm8904.dtsi + create mode 100644 arch/arm/boot/dts/nxp/imx/imx7d-var-som.dtsi + create mode 100644 arch/arm/boot/dts/nxp/vf/vf-colibri-iris.dtsi + create mode 100644 arch/arm/boot/dts/nxp/vf/vf500-colibri-iris.dts + create mode 100644 arch/arm/boot/dts/nxp/vf/vf610-colibri-iris.dts + create mode 100644 arch/arm64/boot/dts/freescale/fsl-lx2160a-nbxv3.dts + create mode 100644 arch/arm64/boot/dts/freescale/fsl-lx2160a-nbxv3.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-cm4.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-dual.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-single.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-edm-sbc-imx8mm-rev900.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-3v3.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-5v0.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-dual.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g070y2-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g101ice-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g121xce-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g156hce-l01.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-g215hvn011.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-mi0700a2t-30.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-mi1010z1t-1cp11.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mm-data-modul-edm-sbc-overlay-lvds-single.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mn-vhip4-overlay-eeprom-1000.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mn-vhip4-overlay-eeprom-1100.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mn-vhip4-overlay-eeprom-2000.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-cm7.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1-audio.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-fio1.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-hdmi.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g070y2-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g101ice-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g121xce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g156hce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-g215hvn011.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi0700a2t-30.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds-mi1010z1t-1cp11.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-mod-imx8mm-lvds.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g070y2-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g101ice-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g121xce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g156hce-l01.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-g215hvn011.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-mi0700a2t-30.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-mi1010z1t-1cp11.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-rev900.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds-rev902.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-lvds.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-rev900.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-data-modul-edm-sbc-overlay-edm-sbc-imx8mp-rev902.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8mp-phyboard-pollux-peb-wlbt-07.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx8ulp-firmware.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx9-mqs.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx91-9x9-qsb-tianma-tm050rdh03.dtso + create mode 100644 arch/arm64/boot/dts/freescale/imx91-lino-verdin-dahlia.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx91-lino-verdin-dev.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx91-lino.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx91-toradex-osm-dev.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx91-toradex-osm.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-11x11-frdm-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-imx91-lino-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-imx91-lino-verdin-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-imx91-lino-verdin-dahlia-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-imx91-lino-verdin-dev-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-imx91-toradex-osm-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-imx91-toradex-osm-dev-common.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-lino-verdin-dahlia.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx93-lino-verdin-dev.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx93-lino.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93-toradex-osm-dev.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx93-toradex-osm.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx93w-frdm.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx95-toradex-osm-dev.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx95-toradex-osm.dtsi + create mode 100644 arch/arm64/boot/dts/freescale/imx952-frdm.dts + create mode 100644 arch/arm64/boot/dts/freescale/imx952-frdm.dtsi + create mode 100644 drivers/firmware/imx/ele_base_msg.c + create mode 100644 drivers/firmware/imx/ele_base_msg.h + create mode 100644 drivers/firmware/imx/ele_common.c + create mode 100644 drivers/firmware/imx/ele_common.h + create mode 100644 drivers/firmware/imx/ele_fw_api.c + create mode 100644 drivers/firmware/imx/ele_fw_api.h + create mode 100644 drivers/firmware/imx/ele_msg_addr_field.c + create mode 100644 drivers/firmware/imx/se_ctrl.c + create mode 100644 drivers/firmware/imx/se_ctrl.h + create mode 100644 include/linux/firmware/imx/se_api.h + create mode 100644 include/uapi/linux/se_ioctl.h +Merging mediatek/for-next (4bfec57314396 Merge branches 'v7.3-next/dts32', 'v7.3-next/dts64' and 'v7.3-next/soc' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mediatek/linux.git mediatek/for-next +Merge made by the 'ort' strategy. + .../devicetree/bindings/arm/mediatek.yaml | 1 + + arch/arm/boot/dts/mediatek/Makefile | 1 + + arch/arm/boot/dts/mediatek/mt8127-amazon-ford.dts | 54 ++++++ + arch/arm/boot/dts/mediatek/mt8127.dtsi | 7 + + arch/arm64/boot/dts/mediatek/mt6878-pinfunc.h | 2 +- + arch/arm64/boot/dts/mediatek/mt6893-pinfunc.h | 2 +- + .../boot/dts/mediatek/mt7986a-bananapi-bpi-r3.dts | 13 ++ + arch/arm64/boot/dts/mediatek/mt7986a.dtsi | 1 + + .../mediatek/mt7988a-bananapi-bpi-r4-pro-cn13.dtso | 1 + + .../mediatek/mt7988a-bananapi-bpi-r4-pro-cn14.dtso | 1 + + .../mediatek/mt7988a-bananapi-bpi-r4-pro-cn15.dtso | 1 + + .../mediatek/mt7988a-bananapi-bpi-r4-pro-cn18.dtso | 1 + + .../dts/mediatek/mt7988a-bananapi-bpi-r4-pro.dtsi | 18 ++ + .../boot/dts/mediatek/mt7988a-bananapi-bpi-r4.dtsi | 13 ++ + arch/arm64/boot/dts/mediatek/mt8173-elm.dtsi | 5 + + arch/arm64/boot/dts/mediatek/mt8173.dtsi | 35 +++- + arch/arm64/boot/dts/mediatek/mt8183-kukui.dtsi | 5 + + arch/arm64/boot/dts/mediatek/mt8186-corsola.dtsi | 5 + + arch/arm64/boot/dts/mediatek/mt8188-geralt.dtsi | 7 +- + arch/arm64/boot/dts/mediatek/mt8192-asurada.dtsi | 5 + + arch/arm64/boot/dts/mediatek/mt8195-cherry.dtsi | 5 + + drivers/soc/mediatek/mt8167-mmsys.h | 183 ++++++++++++++++++++- + drivers/soc/mediatek/mtk-regulator-coupler.c | 1 + + 23 files changed, 353 insertions(+), 14 deletions(-) + create mode 100644 arch/arm/boot/dts/mediatek/mt8127-amazon-ford.dts +Merging mvebu/for-next (6dfda79476538 ARM: dts: marvell: armada-388: use onnn,pca9655 compatible) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gclement/mvebu.git mvebu/for-next +Merge made by the 'ort' strategy. + .../boot/dts/marvell/armada-385-clearfog-gtr.dtsi | 2 +- + arch/arm/boot/dts/marvell/armada-388-clearfog.dtsi | 7 +---- + arch/arm/boot/dts/marvell/armada-388-helios4.dts | 36 ++++++++-------------- + arch/arm/boot/dts/marvell/armada-38x.dtsi | 1 + + arch/arm/boot/dts/marvell/armada-395-gp.dts | 1 - + arch/arm/boot/dts/marvell/armada-39x.dtsi | 1 + + arch/arm/boot/dts/marvell/dove.dtsi | 22 ++++++------- + arch/arm/boot/dts/marvell/kirkwood-6282.dtsi | 15 +++++---- + .../dts/marvell/kirkwood-guruplug-server-plus.dts | 4 +-- + arch/arm/boot/dts/marvell/kirkwood-lsxl.dtsi | 4 +-- + arch/arm/boot/dts/marvell/kirkwood-synology.dtsi | 12 ++++---- + .../arm/boot/dts/marvell/orion5x-rd88f5182-nas.dts | 2 +- + 12 files changed, 46 insertions(+), 61 deletions(-) +Merging omap/for-next (7cdd46c9d6c5a Merge branch 'omap-for-v7.4/soc' into tmp/omap-next-20260921.091322) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/khilman/linux-omap.git omap/for-next +Merge made by the 'ort' strategy. + .../devicetree/bindings/input/omap-keypad.txt | 28 ------- + .../devicetree/bindings/input/ti,omap4-keypad.yaml | 63 +++++++++++++++ + .../boot/dts/ti/omap/am335x-boneblack-hdmi.dtsi | 2 +- + arch/arm/boot/dts/ti/omap/am335x-bonegreen-eco.dts | 2 +- + arch/arm/boot/dts/ti/omap/am335x-shc.dts | 4 - + arch/arm/boot/dts/ti/omap/am437x-l4.dtsi | 2 +- + arch/arm/boot/dts/ti/omap/am57-pruss.dtsi | 11 +++ + arch/arm/boot/dts/ti/omap/am571x-idk.dts | 65 +++++++++++++++- + arch/arm/boot/dts/ti/omap/am5729-beagleboneai.dts | 2 +- + .../boot/dts/ti/omap/am57xx-beagle-x15-common.dtsi | 2 +- + arch/arm/boot/dts/ti/omap/am57xx-idk-common.dtsi | 3 +- + arch/arm/boot/dts/ti/omap/dra7-l4.dtsi | 2 +- + .../boot/dts/ti/omap/motorola-mapphone-common.dtsi | 8 +- + .../dts/ti/omap/motorola-mapphone-mz607-mz617.dtsi | 2 + + arch/arm/boot/dts/ti/omap/omap3-igep.dtsi | 2 +- + arch/arm/boot/dts/ti/omap/omap3-n9.dts | 5 +- + arch/arm/boot/dts/ti/omap/omap3-n900.dts | 3 +- + arch/arm/boot/dts/ti/omap/omap3-n950.dts | 5 +- + arch/arm/boot/dts/ti/omap/omap3-tao3530.dtsi | 8 +- + .../boot/dts/ti/omap/omap4-droid-bionic-xt875.dts | 2 + + arch/arm/boot/dts/ti/omap/omap4-droid4-xt894.dts | 2 + + arch/arm/boot/dts/ti/omap/omap4-epson-embt2ws.dts | 2 + + arch/arm/boot/dts/ti/omap/omap4-l4.dtsi | 1 + + .../dts/ti/omap/omap4-samsung-espresso-common.dtsi | 6 +- + arch/arm/boot/dts/ti/omap/omap4-sdp.dts | 2 + + arch/arm/boot/dts/ti/omap/omap4.dtsi | 6 +- + arch/arm/boot/dts/ti/omap/omap5-l4.dtsi | 1 + + arch/arm/boot/dts/ti/omap/omap5.dtsi | 2 +- + arch/arm/configs/multi_v7_defconfig | 1 + + arch/arm/configs/omap2plus_defconfig | 3 + + arch/arm/mach-omap2/control.h | 8 +- + arch/arm/mach-omap2/ctrl_module_wkup_44xx.h | 89 ---------------------- + arch/arm/mach-omap2/soc.h | 4 +- + arch/arm/mach-omap2/sram.h | 4 +- + 34 files changed, 195 insertions(+), 157 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/input/omap-keypad.txt + create mode 100644 Documentation/devicetree/bindings/input/ti,omap4-keypad.yaml + delete mode 100644 arch/arm/mach-omap2/ctrl_module_wkup_44xx.h +Merging qcom/for-next (e081f1e68766c Merge branches 'arm32-for-7.4', 'arm64-defconfig-for-7.4', 'arm64-fixes-for-7.3', 'arm64-for-7.4', 'clk-fixes-for-7.3', 'clk-for-7.4', 'drivers-fixes-for-7.3' and 'drivers-for-7.4' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux.git qcom/for-next +Auto-merging Documentation/devicetree/bindings/arm/cpus.yaml +Auto-merging Documentation/devicetree/bindings/vendor-prefixes.yaml +Auto-merging MAINTAINERS +Auto-merging arch/arm/configs/multi_v7_defconfig +Auto-merging drivers/firmware/qcom/Kconfig +Auto-merging drivers/firmware/qcom/qcom_tzmem.c +Auto-merging include/linux/mod_devicetable.h +CONFLICT (content): Merge conflict in include/linux/mod_devicetable.h +Resolved 'include/linux/mod_devicetable.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 3857b8771c684] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/qcom/linux.git +$ git diff -M --stat --summary HEAD^.. + Documentation/devicetree/bindings/arm/cpus.yaml | 4 + + .../bindings/arm/qcom,coresight-ctcu.yaml | 1 + + .../devicetree/bindings/arm/qcom-soc.yaml | 4 +- + Documentation/devicetree/bindings/arm/qcom.yaml | 60 + + .../devicetree/bindings/cache/qcom,llcc.yaml | 55 +- + .../bindings/clock/qcom,gcc-msm8952.yaml | 65 + + .../devicetree/bindings/clock/qcom,gcc-sdm845.yaml | 6 +- + .../devicetree/bindings/clock/qcom,gcc-sm6350.yaml | 7 + + .../devicetree/bindings/clock/qcom,gcc-sm8150.yaml | 7 + + .../devicetree/bindings/clock/qcom,gcc-sm8250.yaml | 7 + + .../devicetree/bindings/clock/qcom,gcc-sm8350.yaml | 7 + + .../devicetree/bindings/clock/qcom,gcc-sm8450.yaml | 7 + + .../devicetree/bindings/clock/qcom,hawi-gpucc.yaml | 80 + + .../bindings/clock/qcom,ipq9574-cmn-pll.yaml | 1 + + .../bindings/clock/qcom,kaanapali-gxclkctl.yaml | 23 +- + .../devicetree/bindings/clock/qcom,kuno-gcc.yaml | 52 + + .../bindings/clock/qcom,milos-camcc.yaml | 39 +- + .../bindings/clock/qcom,milos-videocc.yaml | 29 +- + .../devicetree/bindings/clock/qcom,qcs615-gcc.yaml | 7 + + .../bindings/clock/qcom,qcs8300-gcc.yaml | 7 + + .../devicetree/bindings/clock/qcom,rpmcc.yaml | 5 + + .../devicetree/bindings/clock/qcom,rpmhcc.yaml | 2 + + .../devicetree/bindings/clock/qcom,sdx75-gcc.yaml | 7 + + .../devicetree/bindings/clock/qcom,sm4450-gcc.yaml | 7 + + .../devicetree/bindings/clock/qcom,sm6375-gcc.yaml | 9 +- + .../devicetree/bindings/clock/qcom,sm7150-gcc.yaml | 53 - + .../bindings/clock/qcom,sm8450-camcc.yaml | 47 +- + .../bindings/clock/qcom,sm8450-gpucc.yaml | 3 + + .../bindings/clock/qcom,sm8450-videocc.yaml | 2 + + .../devicetree/bindings/clock/qcom,sm8550-gcc.yaml | 7 + + .../bindings/clock/qcom,sm8550-tcsr.yaml | 1 - + .../devicetree/bindings/clock/qcom,sm8650-gcc.yaml | 7 + + .../devicetree/bindings/clock/qcom,sm8750-gcc.yaml | 7 + + .../bindings/clock/qcom,x1e80100-tcsr.yaml | 114 + + .../embedded-controller/qcom,hamoa-crd-ec.yaml | 3 + + .../devicetree/bindings/firmware/qcom,scm.yaml | 4 + + .../bindings/interconnect/qcom,msm8998-bwmon.yaml | 1 + + .../bindings/mailbox/qcom,cpucp-mbox.yaml | 1 + + Documentation/devicetree/bindings/mfd/syscon.yaml | 1 + + .../devicetree/bindings/mmc/qcom,sdhci-msm.yaml | 1 + + .../bindings/net/bluetooth/qcom,wcn6750-bt.yaml | 10 +- + .../bindings/net/wireless/qcom,ath11k.yaml | 16 +- + .../devicetree/bindings/soc/qcom/qcom,apr.yaml | 4 +- + .../devicetree/bindings/soc/qcom/qcom-stats.yaml | 25 +- + .../bindings/spmi/qcom,glymur-spmi-pmic-arb.yaml | 1 + + .../devicetree/bindings/sram/qcom,imem.yaml | 1 - + Documentation/devicetree/bindings/sram/sram.yaml | 1 + + .../devicetree/bindings/usb/qcom,snps-dwc3.yaml | 3 + + .../devicetree/bindings/vendor-prefixes.yaml | 8 + + MAINTAINERS | 2 + + .../boot/dts/qcom/qcom-msm8960-sony-huashan.dts | 15 + + arch/arm/boot/dts/qcom/qcom-msm8960.dtsi | 169 +- + arch/arm/boot/dts/qcom/qcom-msm8974.dtsi | 3 + + arch/arm/boot/dts/qcom/qcom-sdx55-t55.dts | 2 +- + arch/arm/configs/multi_v7_defconfig | 9 - + arch/arm/configs/qcom_defconfig | 14 - + arch/arm64/boot/dts/qcom/Makefile | 44 + + arch/arm64/boot/dts/qcom/agatti.dtsi | 89 +- + arch/arm64/boot/dts/qcom/apq8096-db820c.dtsi | 15 +- + arch/arm64/boot/dts/qcom/eliza-cqs-som.dtsi | 15 + + arch/arm64/boot/dts/qcom/eliza-evk.dtsi | 243 + + arch/arm64/boot/dts/qcom/eliza-mtp.dts | 34 + + arch/arm64/boot/dts/qcom/eliza.dtsi | 774 ++- + .../dts/qcom/glymur-asus-zenbook-a14-ux3407na.dts | 1140 ++++ + .../dts/qcom/glymur-asus-zenbook-a16-ux3607oa.dts | 3 +- + arch/arm64/boot/dts/qcom/glymur-crd.dts | 8 + + arch/arm64/boot/dts/qcom/glymur-crd.dtsi | 36 +- + .../boot/dts/qcom/glymur-hp-elitebook-x-g2q.dts | 940 +++ + .../boot/dts/qcom/glymur-lenovo-yoga-slim7x.dts | 1221 ++++ + arch/arm64/boot/dts/qcom/glymur.dtsi | 496 +- + .../boot/dts/qcom/hamoa-advantech-som-6820.dtsi | 54 + + .../boot/dts/qcom/hamoa-advantech-som-db5830.dts | 119 + + arch/arm64/boot/dts/qcom/hamoa-iot-evk.dts | 267 +- + arch/arm64/boot/dts/qcom/hamoa-iot-som.dtsi | 21 + + .../qcom/hamoa-lenovo-ideacentre-mini-01q8x10.dts | 21 + + arch/arm64/boot/dts/qcom/hamoa-pmics.dtsi | 1 + + arch/arm64/boot/dts/qcom/hamoa.dtsi | 44 +- + arch/arm64/boot/dts/qcom/hawi-ipcc.h | 61 + + arch/arm64/boot/dts/qcom/hawi-mtp.dts | 997 +++ + arch/arm64/boot/dts/qcom/hawi.dtsi | 6844 ++++++++++++++++++++ + arch/arm64/boot/dts/qcom/ipq5018-rdp432-c2.dts | 10 + + arch/arm64/boot/dts/qcom/ipq5018.dtsi | 21 +- + arch/arm64/boot/dts/qcom/ipq5332-rdp-common.dtsi | 2 + + arch/arm64/boot/dts/qcom/ipq5424-rdp466.dts | 2 + + arch/arm64/boot/dts/qcom/ipq5424.dtsi | 16 + + arch/arm64/boot/dts/qcom/ipq9574-rdp-common.dtsi | 2 + + arch/arm64/boot/dts/qcom/ipq9574.dtsi | 3 +- + arch/arm64/boot/dts/qcom/ipq9650-rdp488.dts | 119 + + arch/arm64/boot/dts/qcom/ipq9650.dtsi | 666 +- + arch/arm64/boot/dts/qcom/kaanapali-mtp.dts | 10 +- + arch/arm64/boot/dts/qcom/kaanapali-qrd.dts | 8 + + arch/arm64/boot/dts/qcom/kaanapali.dtsi | 470 +- + arch/arm64/boot/dts/qcom/kodiak.dtsi | 1444 +++-- + arch/arm64/boot/dts/qcom/lemans-el2.dtso | 4 +- + .../boot/dts/qcom/lemans-evk-ifp-mezzanine.dtso | 15 +- + arch/arm64/boot/dts/qcom/lemans-evk.dts | 16 +- + arch/arm64/boot/dts/qcom/lemans-pmics.dtsi | 116 + + arch/arm64/boot/dts/qcom/lemans-ride-common.dtsi | 21 +- + arch/arm64/boot/dts/qcom/lemans.dtsi | 24 +- + arch/arm64/boot/dts/qcom/mahua-crd.dts | 5 + + .../qcom/mahua-lenovo-thinkpad-t14s-gen7-lcd.dts | 1272 ++++ + arch/arm64/boot/dts/qcom/mahua.dtsi | 89 + + arch/arm64/boot/dts/qcom/milos-fairphone-fp6.dts | 191 + + arch/arm64/boot/dts/qcom/milos.dtsi | 479 +- + arch/arm64/boot/dts/qcom/monaco-ac-evk.dts | 938 +++ + arch/arm64/boot/dts/qcom/monaco-arduino-monza.dts | 201 +- + .../boot/dts/qcom/monaco-evk-ifp-mezzanine.dtso | 12 +- + arch/arm64/boot/dts/qcom/monaco-evk.dts | 69 +- + arch/arm64/boot/dts/qcom/monaco-pmics.dtsi | 62 + + arch/arm64/boot/dts/qcom/monaco.dtsi | 15 +- + arch/arm64/boot/dts/qcom/msm8917-xiaomi-riva.dts | 35 + + arch/arm64/boot/dts/qcom/msm8939-asus-z00t.dts | 8 + + .../boot/dts/qcom/msm8939-longcheer-l9100.dts | 8 + + arch/arm64/boot/dts/qcom/msm8939.dtsi | 26 + + arch/arm64/boot/dts/qcom/msm8976.dtsi | 2 +- + .../boot/dts/qcom/msm8996-oneplus-common.dtsi | 5 +- + .../boot/dts/qcom/msm8996-sony-xperia-tone.dtsi | 7 +- + .../arm64/boot/dts/qcom/msm8996-xiaomi-common.dtsi | 6 +- + arch/arm64/boot/dts/qcom/msm8996.dtsi | 45 +- + arch/arm64/boot/dts/qcom/msm8998.dtsi | 6 +- + arch/arm64/boot/dts/qcom/nord-embedded.dtsi | 1928 ++++++ + arch/arm64/boot/dts/qcom/nord-gearvm.dtsi | 2848 ++++++++ + arch/arm64/boot/dts/qcom/nord-ride.dts | 254 + + arch/arm64/boot/dts/qcom/nord-rrd.dts | 424 ++ + arch/arm64/boot/dts/qcom/nord.dtsi | 4650 +++++++++++++ + arch/arm64/boot/dts/qcom/pmh0104-hawi.dtsi | 103 + + arch/arm64/boot/dts/qcom/pmi8996.dtsi | 12 +- + arch/arm64/boot/dts/qcom/purwa-iot-evk.dts | 224 +- + arch/arm64/boot/dts/qcom/purwa-iot-som.dtsi | 21 + + arch/arm64/boot/dts/qcom/qcm6490-fairphone-fp5.dts | 54 + + arch/arm64/boot/dts/qcom/qcm6490-idp.dts | 6 + + .../boot/dts/qcom/qcm6490-particle-tachyon.dts | 3 +- + arch/arm64/boot/dts/qcom/qcs404-evb.dtsi | 6 +- + arch/arm64/boot/dts/qcom/qcs404.dtsi | 7 +- + arch/arm64/boot/dts/qcom/qcs615-ride.dts | 5 +- + .../boot/dts/qcom/qcs6490-radxa-dragon-q6a.dts | 4 +- + .../qcom/qcs6490-rb3gen2-industrial-mezzanine.dtso | 30 +- + .../dts/qcom/qcs6490-rb3gen2-vision-mezzanine.dtso | 13 +- + arch/arm64/boot/dts/qcom/qcs6490-rb3gen2.dts | 24 +- + .../dts/qcom/qcs6490-thundercomm-minipc-g1iot.dts | 16 +- + .../qcs6490-thundercomm-rubikpi3-cam1-imx219.dtso | 115 + + .../qcs6490-thundercomm-rubikpi3-cam2-imx219.dtso | 115 + + .../boot/dts/qcom/qcs6490-thundercomm-rubikpi3.dts | 25 +- + .../boot/dts/qcom/qcs6490-vicharak-axon-mini.dts | 16 +- + arch/arm64/boot/dts/qcom/qcs8300-ride.dts | 8 +- + arch/arm64/boot/dts/qcom/qcs8550-aim300.dtsi | 16 +- + .../arm64/boot/dts/qcom/qcs8550-ayntec-common.dtsi | 1819 ++++++ + .../boot/dts/qcom/qcs8550-ayntec-odin2mini.dts | 44 + + .../boot/dts/qcom/qcs8550-ayntec-odin2portal.dts | 99 + + arch/arm64/boot/dts/qcom/qcs8550-ayntec-thor.dts | 250 + + arch/arm64/boot/dts/qcom/qcs8550-imdt-sbc.dts | 401 ++ + arch/arm64/boot/dts/qcom/qcs8550-imdt-som.dtsi | 315 + + arch/arm64/boot/dts/qcom/qcs8550-rb5gen2.dts | 24 +- + arch/arm64/boot/dts/qcom/qdu1000.dtsi | 6 +- + arch/arm64/boot/dts/qcom/sa8540p-ride.dts | 4 +- + arch/arm64/boot/dts/qcom/sar2130p-qar2130p.dts | 6 +- + arch/arm64/boot/dts/qcom/sar2130p.dtsi | 13 +- + .../boot/dts/qcom/sc7180-trogdor-homestar.dtsi | 2 +- + .../qcom/sc7180-trogdor-lazor-limozeen-nots-r4.dts | 2 +- + .../boot/dts/qcom/sc7180-trogdor-lazor-r1.dts | 4 +- + .../boot/dts/qcom/sc7180-trogdor-pompom-r1.dts | 4 +- + arch/arm64/boot/dts/qcom/sc7180-trogdor-r1.dts | 4 +- + .../arm64/boot/dts/qcom/sc8180x-lenovo-flex-5g.dts | 7 +- + arch/arm64/boot/dts/qcom/sc8180x-primus.dts | 7 +- + arch/arm64/boot/dts/qcom/sc8180x.dtsi | 24 +- + arch/arm64/boot/dts/qcom/sc8280xp-crd.dts | 3 + + .../boot/dts/qcom/sc8280xp-huawei-gaokun3.dts | 13 +- + .../dts/qcom/sc8280xp-lenovo-thinkpad-x13s.dts | 37 + + .../boot/dts/qcom/sc8280xp-microsoft-arcata.dts | 5 +- + .../boot/dts/qcom/sc8280xp-microsoft-blackrock.dts | 3 + + arch/arm64/boot/dts/qcom/sc8280xp.dtsi | 690 +- + arch/arm64/boot/dts/qcom/sdm630.dtsi | 16 +- + arch/arm64/boot/dts/qcom/sdm670-google-common.dtsi | 50 +- + arch/arm64/boot/dts/qcom/sdm670.dtsi | 271 + + arch/arm64/boot/dts/qcom/sdm845-db845c.dts | 13 +- + arch/arm64/boot/dts/qcom/sdm845-google-common.dtsi | 36 +- + arch/arm64/boot/dts/qcom/sdm845-lg-common.dtsi | 13 +- + arch/arm64/boot/dts/qcom/sdm845-mtp.dts | 12 +- + .../arm64/boot/dts/qcom/sdm845-oneplus-common.dtsi | 7 +- + .../boot/dts/qcom/sdm845-oneplus-enchilada.dts | 5 + + arch/arm64/boot/dts/qcom/sdm845-oneplus-fajita.dts | 1 + + .../boot/dts/qcom/sdm845-sony-xperia-tama.dtsi | 2 + + .../dts/qcom/sdm845-xiaomi-beryllium-common.dtsi | 6 + + arch/arm64/boot/dts/qcom/sdm845-xiaomi-polaris.dts | 6 + + arch/arm64/boot/dts/qcom/sdm845.dtsi | 20 +- + arch/arm64/boot/dts/qcom/sdx75.dtsi | 2 - + .../dts/qcom/shikra-cqm-cqs-evk-imx577-camera.dtso | 72 + + arch/arm64/boot/dts/qcom/shikra-cqm-evk.dts | 36 + + arch/arm64/boot/dts/qcom/shikra-cqm-som.dtsi | 1 - + arch/arm64/boot/dts/qcom/shikra-cqs-evk.dts | 32 + + arch/arm64/boot/dts/qcom/shikra-evk.dtsi | 41 + + .../dts/qcom/shikra-iqs-evk-imx577-camera.dtso | 72 + + arch/arm64/boot/dts/qcom/shikra-iqs-evk.dts | 32 + + arch/arm64/boot/dts/qcom/shikra-iqs-som.dtsi | 3 +- + arch/arm64/boot/dts/qcom/shikra.dtsi | 2034 +++++- + arch/arm64/boot/dts/qcom/sm4450.dtsi | 1 - + arch/arm64/boot/dts/qcom/sm6115.dtsi | 2 +- + arch/arm64/boot/dts/qcom/sm6350.dtsi | 233 + + arch/arm64/boot/dts/qcom/sm7225-fairphone-fp4.dts | 4 + + .../boot/dts/qcom/sm7325-nothing-spacewar.dts | 2 + + arch/arm64/boot/dts/qcom/sm7325-xiaomi-lisa.dts | 1108 ++++ + arch/arm64/boot/dts/qcom/sm8150.dtsi | 25 +- + arch/arm64/boot/dts/qcom/sm8250.dtsi | 49 +- + arch/arm64/boot/dts/qcom/sm8350-hdk.dts | 18 +- + .../boot/dts/qcom/sm8350-sony-xperia-sagami.dtsi | 17 +- + arch/arm64/boot/dts/qcom/sm8350.dtsi | 15 +- + arch/arm64/boot/dts/qcom/sm8450.dtsi | 25 +- + arch/arm64/boot/dts/qcom/sm8550-hdk.dts | 28 +- + arch/arm64/boot/dts/qcom/sm8550-mtp.dts | 16 +- + arch/arm64/boot/dts/qcom/sm8550-qrd.dts | 20 +- + arch/arm64/boot/dts/qcom/sm8550-samsung-q5q.dts | 7 +- + .../dts/qcom/sm8550-sony-xperia-yodo-pdx234.dts | 8 +- + arch/arm64/boot/dts/qcom/sm8550.dtsi | 12 +- + .../boot/dts/qcom/sm8650-ayaneo-pocket-s2.dts | 243 +- + arch/arm64/boot/dts/qcom/sm8650-hdk.dts | 22 +- + arch/arm64/boot/dts/qcom/sm8650-mtp.dts | 16 +- + arch/arm64/boot/dts/qcom/sm8650-qrd.dts | 14 +- + arch/arm64/boot/dts/qcom/sm8650-valve-deckard.dts | 1288 ++++ + arch/arm64/boot/dts/qcom/sm8650.dtsi | 18 +- + arch/arm64/boot/dts/qcom/sm8750-mtp.dts | 10 +- + arch/arm64/boot/dts/qcom/sm8750-qrd.dts | 8 + + arch/arm64/boot/dts/qcom/sm8750.dtsi | 407 +- + .../boot/dts/qcom/talos-evk-rpi-display-2-5in.dtso | 88 + + arch/arm64/boot/dts/qcom/talos-evk-som.dtsi | 9 +- + arch/arm64/boot/dts/qcom/talos-evk.dts | 2 +- + arch/arm64/boot/dts/qcom/talos.dtsi | 11 +- + arch/arm64/boot/dts/qcom/x1-asus-vivobook-s15.dtsi | 22 + + arch/arm64/boot/dts/qcom/x1-asus-zenbook-a14.dtsi | 22 + + arch/arm64/boot/dts/qcom/x1-crd.dtsi | 22 + + arch/arm64/boot/dts/qcom/x1-dell-thena.dtsi | 24 +- + arch/arm64/boot/dts/qcom/x1-hp-omnibook-x14.dtsi | 22 + + arch/arm64/boot/dts/qcom/x1-microsoft-denali.dtsi | 93 +- + arch/arm64/boot/dts/qcom/x1e001de-devkit.dts | 24 +- + .../dts/qcom/x1e78100-lenovo-thinkpad-t14s.dtsi | 22 + + arch/arm64/boot/dts/qcom/x1e80100-crd.dts | 119 + + .../qcom/x1e80100-dell-inspiron-14-plus-7441.dts | 31 +- + .../boot/dts/qcom/x1e80100-dell-xps13-9345.dts | 22 + + .../dts/qcom/x1e80100-honor-magicbook-art-14.dts | 24 +- + .../boot/dts/qcom/x1e80100-lenovo-yoga-slim7x.dts | 44 + + .../dts/qcom/x1e80100-medion-sprchrgd-14-s1.dts | 26 +- + .../boot/dts/qcom/x1e80100-microsoft-romulus.dtsi | 21 + + arch/arm64/boot/dts/qcom/x1e80100-qcp.dts | 22 + + .../boot/dts/qcom/x1p42100-lenovo-thinkbook-16.dts | 22 + + .../boot/dts/qcom/x1p42100-microsoft-sp12in.dts | 21 + + arch/arm64/configs/defconfig | 127 +- + drivers/clk/qcom/Kconfig | 167 +- + drivers/clk/qcom/Makefile | 9 + + drivers/clk/qcom/cambistmclkcc-eliza.c | 464 ++ + drivers/clk/qcom/cambistmclkcc-kaanapali.c | 7 + + drivers/clk/qcom/cambistmclkcc-sm8750.c | 7 + + drivers/clk/qcom/camcc-eliza.c | 2803 ++++++++ + drivers/clk/qcom/camcc-glymur.c | 7 + + drivers/clk/qcom/camcc-hawi.c | 3152 +++++++++ + drivers/clk/qcom/camcc-nord.c | 2941 +++++++++ + drivers/clk/qcom/camcc-sa8775p.c | 36 +- + drivers/clk/qcom/camcc-sc8280xp.c | 13 +- + drivers/clk/qcom/camcc-sm7150.c | 13 +- + drivers/clk/qcom/camcc-sm8150.c | 13 +- + drivers/clk/qcom/clk-alpha-pll.c | 9 + + drivers/clk/qcom/clk-rcg2.c | 35 +- + drivers/clk/qcom/clk-regmap-divider.c | 16 +- + drivers/clk/qcom/clk-regmap-divider.h | 1 + + drivers/clk/qcom/clk-rpmh.c | 44 + + drivers/clk/qcom/dispcc-sc7280.c | 13 +- + drivers/clk/qcom/dispcc-sc8280xp.c | 18 +- + drivers/clk/qcom/dispcc-sm4450.c | 15 +- + drivers/clk/qcom/dispcc-sm6115.c | 13 +- + drivers/clk/qcom/dispcc-sm7150.c | 13 +- + drivers/clk/qcom/dispcc-sm8250.c | 13 +- + drivers/clk/qcom/dispcc-sm8450.c | 40 +- + drivers/clk/qcom/dispcc-sm8550.c | 13 +- + drivers/clk/qcom/dispcc-sm8750.c | 19 +- + drivers/clk/qcom/dispcc-x1e80100.c | 27 +- + drivers/clk/qcom/dispcc0-sa8775p.c | 15 +- + drivers/clk/qcom/dispcc1-sa8775p.c | 15 +- + drivers/clk/qcom/gcc-eliza.c | 19 +- + drivers/clk/qcom/gcc-glymur.c | 81 +- + drivers/clk/qcom/gcc-hawi.c | 18 +- + drivers/clk/qcom/gcc-ipq5018.c | 10 + + drivers/clk/qcom/gcc-ipq5210.c | 1 + + drivers/clk/qcom/gcc-ipq5332.c | 1 + + drivers/clk/qcom/gcc-ipq5424.c | 4 +- + drivers/clk/qcom/gcc-ipq9650.c | 22 + + drivers/clk/qcom/gcc-kaanapali.c | 19 +- + drivers/clk/qcom/gcc-kuno.c | 1473 +++++ + drivers/clk/qcom/gcc-milos.c | 12 +- + drivers/clk/qcom/gcc-msm8939.c | 4 + + drivers/clk/qcom/gcc-msm8952.c | 3548 ++++++++++ + drivers/clk/qcom/gcc-nord.c | 12 +- + drivers/clk/qcom/gcc-qcm2290.c | 12 +- + drivers/clk/qcom/gcc-qcs615.c | 69 +- + drivers/clk/qcom/gcc-qcs8300.c | 167 +- + drivers/clk/qcom/gcc-qdu1000.c | 12 +- + drivers/clk/qcom/gcc-sa8775p.c | 56 +- + drivers/clk/qcom/gcc-sar2130p.c | 58 +- + drivers/clk/qcom/gcc-sc7180.c | 60 +- + drivers/clk/qcom/gcc-sc7280.c | 67 +- + drivers/clk/qcom/gcc-sc8280xp.c | 47 +- + drivers/clk/qcom/gcc-sdx55.c | 43 +- + drivers/clk/qcom/gcc-sdx65.c | 43 +- + drivers/clk/qcom/gcc-sdx75.c | 44 +- + drivers/clk/qcom/gcc-shikra.c | 12 +- + drivers/clk/qcom/gcc-sm4450.c | 77 +- + drivers/clk/qcom/gcc-sm6115.c | 12 +- + drivers/clk/qcom/gcc-sm6125.c | 12 +- + drivers/clk/qcom/gcc-sm6375.c | 12 +- + drivers/clk/qcom/gcc-sm7150.c | 72 +- + drivers/clk/qcom/gcc-sm8150.c | 12 +- + drivers/clk/qcom/gcc-sm8250.c | 69 +- + drivers/clk/qcom/gcc-sm8350.c | 65 +- + drivers/clk/qcom/gcc-sm8450.c | 64 +- + drivers/clk/qcom/gcc-sm8550.c | 70 +- + drivers/clk/qcom/gcc-sm8650.c | 72 +- + drivers/clk/qcom/gcc-sm8750.c | 89 +- + drivers/clk/qcom/gcc-x1e80100.c | 80 +- + drivers/clk/qcom/gpucc-eliza.c | 606 ++ + drivers/clk/qcom/gpucc-glymur.c | 2 +- + drivers/clk/qcom/gpucc-hawi.c | 477 ++ + drivers/clk/qcom/gpucc-kaanapali.c | 2 +- + drivers/clk/qcom/gpucc-sar2130p.c | 13 +- + drivers/clk/qcom/gpucc-sc7280.c | 22 +- + drivers/clk/qcom/gpucc-sc8280xp.c | 15 +- + drivers/clk/qcom/gpucc-sm4450.c | 17 +- + drivers/clk/qcom/gpucc-sm8550.c | 15 +- + drivers/clk/qcom/gpucc-x1e80100.c | 13 +- + drivers/clk/qcom/gpucc-x1p42100.c | 17 +- + drivers/clk/qcom/ipq-cmn-pll.c | 538 +- + drivers/clk/qcom/lpasscc-sc7280.c | 12 +- + drivers/clk/qcom/lpasscc-sdm845.c | 12 +- + drivers/clk/qcom/lpasscc-sm6115.c | 4 +- + drivers/clk/qcom/lpasscorecc-sc7180.c | 31 +- + drivers/clk/qcom/mmcc-msm8960.c | 3 +- + drivers/clk/qcom/mmcc-msm8996.c | 4 +- + drivers/clk/qcom/mmcc-sdm660.c | 2 +- + drivers/clk/qcom/negcc-nord.c | 5 +- + drivers/clk/qcom/nwgcc-nord.c | 5 +- + drivers/clk/qcom/segcc-nord.c | 40 +- + drivers/clk/qcom/tcsrcc-eliza.c | 12 +- + drivers/clk/qcom/tcsrcc-glymur.c | 12 +- + drivers/clk/qcom/tcsrcc-kaanapali.c | 12 +- + drivers/clk/qcom/tcsrcc-sm8550.c | 12 +- + drivers/clk/qcom/tcsrcc-sm8650.c | 12 +- + drivers/clk/qcom/tcsrcc-sm8750.c | 12 +- + drivers/clk/qcom/tcsrcc-x1e80100.c | 349 +- + drivers/clk/qcom/videocc-eliza.c | 404 ++ + drivers/clk/qcom/videocc-glymur.c | 38 + + drivers/clk/qcom/videocc-sa8775p.c | 17 +- + drivers/clk/qcom/videocc-sm6350.c | 13 +- + drivers/clk/qcom/videocc-sm7150.c | 13 +- + drivers/clk/qcom/videocc-sm8250.c | 15 +- + drivers/clk/qcom/videocc-sm8350.c | 31 +- + drivers/clk/qcom/videocc-sm8750.c | 12 +- + drivers/clk/renesas/r9a06g032-clocks.c | 8 +- + drivers/clk/renesas/renesas-cpg-mssr.c | 7 +- + drivers/clk/renesas/rzg2l-cpg.c | 7 +- + drivers/clk/renesas/rzv2h-cpg.c | 7 +- + drivers/firmware/qcom/Kconfig | 5 +- + drivers/firmware/qcom/qcom_scm-legacy.c | 2 +- + drivers/firmware/qcom/qcom_scm.c | 187 +- + drivers/firmware/qcom/qcom_tzmem.c | 3 +- + drivers/soc/qcom/Kconfig | 1 - + drivers/soc/qcom/apr.c | 75 +- + drivers/soc/qcom/llcc-qcom.c | 199 +- + drivers/soc/qcom/pmic_glink.c | 3 +- + drivers/soc/qcom/pmic_glink_altmode.c | 4 +- + drivers/soc/qcom/pmic_pdcharger_ulog.c | 2 +- + drivers/soc/qcom/qcom-geni-se.c | 65 +- + drivers/soc/qcom/qcom_aoss.c | 4 +- + drivers/soc/qcom/qcom_stats.c | 9 + + drivers/soc/qcom/rpmh-internal.h | 16 +- + drivers/soc/qcom/rpmh-rsc.c | 4 +- + drivers/soc/qcom/rpmh.c | 40 +- + drivers/soc/qcom/smem_dramc.c | 81 +- + drivers/soc/qcom/smp2p.c | 22 +- + drivers/soc/qcom/smsm.c | 36 +- + drivers/soc/qcom/socinfo.c | 3 + + drivers/soc/qcom/ubwc_config.c | 12 + + include/dt-bindings/arm/qcom,ids.h | 3 + + .../dt-bindings/clock/qcom,eliza-cambistmclkcc.h | 32 + + include/dt-bindings/clock/qcom,eliza-camcc.h | 151 + + include/dt-bindings/clock/qcom,eliza-gpucc.h | 51 + + include/dt-bindings/clock/qcom,eliza-videocc.h | 37 + + include/dt-bindings/clock/qcom,gcc-msm8952.h | 215 + + include/dt-bindings/clock/qcom,hawi-camcc.h | 165 + + include/dt-bindings/clock/qcom,hawi-gpucc.h | 47 + + include/dt-bindings/clock/qcom,ipq5210-cmn-pll.h | 30 + + include/dt-bindings/clock/qcom,kuno-gcc.h | 100 + + include/dt-bindings/clock/qcom,nord-camcc.h | 167 + + include/dt-bindings/clock/qcom,nord-negcc.h | 1 + + include/dt-bindings/clock/qcom,nord-nwgcc.h | 3 + + include/dt-bindings/clock/qcom,nord-segcc.h | 2 + + include/dt-bindings/clock/qcom,qcs8300-gcc.h | 6 + + include/dt-bindings/interconnect/qcom,ipq9650.h | 28 + + include/linux/device-id/apr.h | 21 - + include/linux/device/driver.h | 31 + + include/linux/firmware/qcom/qcom_scm.h | 29 - + include/linux/mod_devicetable.h | 1 - + include/linux/platform_device.h | 30 + + include/linux/soc/qcom/apr.h | 4 +- + include/linux/soc/qcom/llcc-qcom.h | 6 + + include/linux/soc/qcom/ubwc.h | 3 + + 401 files changed, 60568 insertions(+), 3709 deletions(-) + create mode 100644 Documentation/devicetree/bindings/clock/qcom,gcc-msm8952.yaml + create mode 100644 Documentation/devicetree/bindings/clock/qcom,hawi-gpucc.yaml + create mode 100644 Documentation/devicetree/bindings/clock/qcom,kuno-gcc.yaml + delete mode 100644 Documentation/devicetree/bindings/clock/qcom,sm7150-gcc.yaml + create mode 100644 Documentation/devicetree/bindings/clock/qcom,x1e80100-tcsr.yaml + create mode 100644 arch/arm64/boot/dts/qcom/glymur-asus-zenbook-a14-ux3407na.dts + create mode 100644 arch/arm64/boot/dts/qcom/glymur-hp-elitebook-x-g2q.dts + create mode 100644 arch/arm64/boot/dts/qcom/glymur-lenovo-yoga-slim7x.dts + create mode 100644 arch/arm64/boot/dts/qcom/hamoa-advantech-som-6820.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/hamoa-advantech-som-db5830.dts + create mode 100644 arch/arm64/boot/dts/qcom/hawi-ipcc.h + create mode 100644 arch/arm64/boot/dts/qcom/hawi-mtp.dts + create mode 100644 arch/arm64/boot/dts/qcom/hawi.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/mahua-lenovo-thinkpad-t14s-gen7-lcd.dts + create mode 100644 arch/arm64/boot/dts/qcom/monaco-ac-evk.dts + create mode 100644 arch/arm64/boot/dts/qcom/nord-embedded.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/nord-gearvm.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/nord-ride.dts + create mode 100644 arch/arm64/boot/dts/qcom/nord-rrd.dts + create mode 100644 arch/arm64/boot/dts/qcom/nord.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/pmh0104-hawi.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/qcs6490-thundercomm-rubikpi3-cam1-imx219.dtso + create mode 100644 arch/arm64/boot/dts/qcom/qcs6490-thundercomm-rubikpi3-cam2-imx219.dtso + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-ayntec-common.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-ayntec-odin2mini.dts + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-ayntec-odin2portal.dts + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-ayntec-thor.dts + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-imdt-sbc.dts + create mode 100644 arch/arm64/boot/dts/qcom/qcs8550-imdt-som.dtsi + create mode 100644 arch/arm64/boot/dts/qcom/shikra-cqm-cqs-evk-imx577-camera.dtso + create mode 100644 arch/arm64/boot/dts/qcom/shikra-iqs-evk-imx577-camera.dtso + create mode 100644 arch/arm64/boot/dts/qcom/sm7325-xiaomi-lisa.dts + create mode 100644 arch/arm64/boot/dts/qcom/sm8650-valve-deckard.dts + create mode 100644 arch/arm64/boot/dts/qcom/talos-evk-rpi-display-2-5in.dtso + create mode 100644 drivers/clk/qcom/cambistmclkcc-eliza.c + create mode 100644 drivers/clk/qcom/camcc-eliza.c + create mode 100644 drivers/clk/qcom/camcc-hawi.c + create mode 100644 drivers/clk/qcom/camcc-nord.c + create mode 100644 drivers/clk/qcom/gcc-kuno.c + create mode 100644 drivers/clk/qcom/gcc-msm8952.c + create mode 100644 drivers/clk/qcom/gpucc-eliza.c + create mode 100644 drivers/clk/qcom/gpucc-hawi.c + create mode 100644 drivers/clk/qcom/videocc-eliza.c + create mode 100644 include/dt-bindings/clock/qcom,eliza-cambistmclkcc.h + create mode 100644 include/dt-bindings/clock/qcom,eliza-camcc.h + create mode 100644 include/dt-bindings/clock/qcom,eliza-gpucc.h + create mode 100644 include/dt-bindings/clock/qcom,eliza-videocc.h + create mode 100644 include/dt-bindings/clock/qcom,gcc-msm8952.h + create mode 100644 include/dt-bindings/clock/qcom,hawi-camcc.h + create mode 100644 include/dt-bindings/clock/qcom,hawi-gpucc.h + create mode 100644 include/dt-bindings/clock/qcom,ipq5210-cmn-pll.h + create mode 100644 include/dt-bindings/clock/qcom,kuno-gcc.h + create mode 100644 include/dt-bindings/clock/qcom,nord-camcc.h + create mode 100644 include/dt-bindings/interconnect/qcom,ipq9650.h + delete mode 100644 include/linux/device-id/apr.h +Merging realtek/for-next (3c778f0c9fa36 arm64: dts: realtek: Add GPIO support for RTD1625) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/yu_chun/linux.git realtek/for-next +Already up to date. +Merging renesas/next (7687cd01a980a Merge branches 'renesas-drivers-for-v7.4' and 'renesas-dts-for-v7.4' into renesas-next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-devel.git renesas/next +Auto-merging arch/arm/configs/shmobile_defconfig +Merge made by the 'ort' strategy. + .../devicetree/bindings/power/qcom,rpmpd.yaml | 2 + + .../bindings/power/renesas,rcar-sysc.yaml | 7 +- + .../bindings/power/riscv,rpmi-device-power.yaml | 65 +++ + .../power/riscv,rpmi-mpxy-device-power.yaml | 65 +++ + .../devicetree/bindings/soc/renesas/renesas.yaml | 37 ++ + arch/arm/boot/dts/renesas/r7s72100-gr-peach.dts | 2 + + .../boot/dts/renesas/r8a7740-armadillo800eva.dts | 2 + + arch/arm/boot/dts/renesas/r8a7743-sk-rzg1m.dts | 2 + + arch/arm/boot/dts/renesas/r8a7743.dtsi | 6 +- + arch/arm/boot/dts/renesas/r8a7744.dtsi | 6 +- + arch/arm/boot/dts/renesas/r8a7745-sk-rzg1e.dts | 2 + + arch/arm/boot/dts/renesas/r8a7745.dtsi | 6 +- + arch/arm/boot/dts/renesas/r8a77470.dtsi | 6 +- + arch/arm/boot/dts/renesas/r8a7790-lager.dts | 28 +- + arch/arm/boot/dts/renesas/r8a7790-stout.dts | 4 +- + arch/arm/boot/dts/renesas/r8a7790.dtsi | 4 +- + arch/arm/boot/dts/renesas/r8a7791-koelsch.dts | 2 + + arch/arm/boot/dts/renesas/r8a7791-porter.dts | 2 + + arch/arm/boot/dts/renesas/r8a7791.dtsi | 4 +- + arch/arm/boot/dts/renesas/r8a7792.dtsi | 4 +- + arch/arm/boot/dts/renesas/r8a7793-gose.dts | 2 + + arch/arm/boot/dts/renesas/r8a7793.dtsi | 4 +- + arch/arm/boot/dts/renesas/r8a7794-alt.dts | 2 + + arch/arm/boot/dts/renesas/r8a7794-silk.dts | 2 + + arch/arm/boot/dts/renesas/r8a7794.dtsi | 4 +- + .../arm/boot/dts/renesas/r9a06g032-rzn1d400-db.dts | 1 - + .../arm/boot/dts/renesas/r9a06g032-rzn1d400-eb.dts | 14 +- + arch/arm/boot/dts/renesas/r9a06g032.dtsi | 2 - + arch/arm/configs/shmobile_defconfig | 3 +- + arch/arm64/boot/dts/renesas/Makefile | 83 +++ + .../dts/renesas/aistarvision-mipi-adapter-2.1.dtsi | 2 +- + .../boot/dts/renesas/beacon-renesom-baseboard.dtsi | 6 +- + .../arm64/boot/dts/renesas/beacon-renesom-som.dtsi | 12 +- + arch/arm64/boot/dts/renesas/cat875.dtsi | 2 + + arch/arm64/boot/dts/renesas/condor-common.dtsi | 2 +- + arch/arm64/boot/dts/renesas/draak.dtsi | 2 +- + arch/arm64/boot/dts/renesas/ebisu.dtsi | 6 +- + arch/arm64/boot/dts/renesas/gray-hawk-single.dtsi | 8 +- + arch/arm64/boot/dts/renesas/hihope-common.dtsi | 4 +- + arch/arm64/boot/dts/renesas/hihope-rev2.dtsi | 8 +- + arch/arm64/boot/dts/renesas/hihope-rev4.dtsi | 4 +- + arch/arm64/boot/dts/renesas/hihope-rzg2-ex.dtsi | 6 +- + arch/arm64/boot/dts/renesas/r8a774a1.dtsi | 14 +- + .../boot/dts/renesas/r8a774a3-hihope-rzg2m-ex.dts | 14 + + .../boot/dts/renesas/r8a774a3-hihope-rzg2m.dts | 49 ++ + arch/arm64/boot/dts/renesas/r8a774a3.dtsi | 302 +++++++++++ + arch/arm64/boot/dts/renesas/r8a774b1.dtsi | 12 +- + arch/arm64/boot/dts/renesas/r8a774c0-cat874.dts | 4 +- + arch/arm64/boot/dts/renesas/r8a774c0.dtsi | 12 +- + arch/arm64/boot/dts/renesas/r8a774e1.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a77951.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a77960.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a77961.dtsi | 14 +- + arch/arm64/boot/dts/renesas/r8a77965.dtsi | 12 +- + .../renesas/r8a77970-eagle-function-expansion.dtso | 2 +- + arch/arm64/boot/dts/renesas/r8a77970-eagle.dts | 2 +- + arch/arm64/boot/dts/renesas/r8a77970-v3msk.dts | 2 +- + arch/arm64/boot/dts/renesas/r8a77970.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r8a77980-v3hsk.dts | 2 +- + arch/arm64/boot/dts/renesas/r8a77980.dtsi | 4 +- + arch/arm64/boot/dts/renesas/r8a77990.dtsi | 10 +- + arch/arm64/boot/dts/renesas/r8a77995.dtsi | 6 +- + .../boot/dts/renesas/r8a779a0-falcon-cpu.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r8a779a0-falcon.dts | 4 +- + arch/arm64/boot/dts/renesas/r8a779a0.dtsi | 2 +- + .../boot/dts/renesas/r8a779f0-spider-cpu.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r8a779f0.dtsi | 52 +- + arch/arm64/boot/dts/renesas/r8a779f4-s4sk.dts | 2 +- + .../arm64/boot/dts/renesas/r8a779g0-white-hawk.dts | 94 ++++ + arch/arm64/boot/dts/renesas/r8a779g0.dtsi | 85 +++- + .../renesas/r8a779g3-sparrow-hawk-fan-argon40.dtso | 1 - + .../r8a779g3-sparrow-hawk-ws-2ch-canfd.dtso | 135 +++++ + .../boot/dts/renesas/r8a779g3-sparrow-hawk.dts | 19 +- + arch/arm64/boot/dts/renesas/r8a779h0.dtsi | 34 +- + arch/arm64/boot/dts/renesas/r8a779md-geist.dts | 12 +- + arch/arm64/boot/dts/renesas/r8a78000-ironhide.dts | 12 +- + arch/arm64/boot/dts/renesas/r8a78000.dtsi | 516 ++++++++++++++++++- + .../renesas/r9a07g043u12-hummingboard-ripple.dts | 89 ++++ + .../dts/renesas/r9a07g044c2-hummingboard-iiot.dts | 20 + + .../renesas/r9a07g044c2-hummingboard-ripple.dts | 93 ++++ + .../dts/renesas/r9a07g044l2-hummingboard-iiot.dts | 16 + + .../renesas/r9a07g044l2-hummingboard-ripple.dts | 18 + + .../arm64/boot/dts/renesas/r9a07g044l2-remi-pi.dts | 2 +- + .../dts/renesas/r9a07g054l2-hummingboard-iiot.dts | 16 + + .../renesas/r9a07g054l2-hummingboard-ripple.dts | 18 + + arch/arm64/boot/dts/renesas/r9a08g045.dtsi | 5 +- + arch/arm64/boot/dts/renesas/r9a08g046.dtsi | 335 +++++++++++- + arch/arm64/boot/dts/renesas/r9a08g046l48-smarc.dts | 36 ++ + arch/arm64/boot/dts/renesas/r9a09g011.dtsi | 4 +- + arch/arm64/boot/dts/renesas/r9a09g047.dtsi | 126 ++++- + arch/arm64/boot/dts/renesas/r9a09g047e57-smarc.dts | 58 ++- + arch/arm64/boot/dts/renesas/r9a09g056.dtsi | 2 +- + arch/arm64/boot/dts/renesas/r9a09g057.dtsi | 2 +- + .../boot/dts/renesas/r9a09g057h44-rzv2h-evk.dts | 6 +- + arch/arm64/boot/dts/renesas/r9a09g077.dtsi | 89 +++- + .../dts/renesas/r9a09g077m44-evk-cn15-lcdc.dtso | 53 ++ + .../boot/dts/renesas/r9a09g077m44-rzt2h-evk.dts | 18 +- + arch/arm64/boot/dts/renesas/r9a09g087.dtsi | 89 +++- + .../dts/renesas/r9a09g087m44-evk-cn20-lcdc.dtso | 63 +++ + .../boot/dts/renesas/r9a09g087m44-rzn2h-evk.dts | 32 +- + arch/arm64/boot/dts/renesas/renesas-smarc2.dtsi | 38 +- + .../renesas/rzg2l-hummingboard-iiot-common.dtsi | 560 +++++++++++++++++++++ + .../renesas/rzg2l-hummingboard-iiot-microsd.dtso | 26 + + ...hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtso | 79 +++ + .../renesas/rzg2l-hummingboard-iiot-rs485-a.dtso | 17 + + .../renesas/rzg2l-hummingboard-iiot-rs485-b.dtso | 17 + + .../boot/dts/renesas/rzg2l-hummingboard-iiot.dtsi | 49 ++ + .../renesas/rzg2l-hummingboard-pulse-common.dtsi | 116 +++++ + .../rzg2l-hummingboard-pulse-micro-hdmi.dtsi | 79 +++ + .../dts/renesas/rzg2l-hummingboard-ripple.dtsi | 122 +++++ + .../boot/dts/renesas/rzg2l-smarc-pinfunction.dtsi | 16 +- + arch/arm64/boot/dts/renesas/rzg2l-smarc-som.dtsi | 20 +- + arch/arm64/boot/dts/renesas/rzg2l-smarc.dtsi | 2 +- + arch/arm64/boot/dts/renesas/rzg2l-sr-som-emmc.dtso | 44 ++ + arch/arm64/boot/dts/renesas/rzg2l-sr-som.dtsi | 473 +++++++++++++++++ + .../renesas/rzg2lc-hummingboard-pulse-leds.dtso | 48 ++ + .../boot/dts/renesas/rzg2lc-smarc-pinfunction.dtsi | 16 +- + arch/arm64/boot/dts/renesas/rzg2lc-smarc-som.dtsi | 20 +- + arch/arm64/boot/dts/renesas/rzg2lc-smarc.dtsi | 2 +- + arch/arm64/boot/dts/renesas/rzg2lc-sr-som.dtsi | 436 ++++++++++++++++ + .../renesas/rzg2ul-hummingboard-pulse-leds.dtso | 48 ++ + .../boot/dts/renesas/rzg2ul-smarc-pinfunction.dtsi | 16 +- + arch/arm64/boot/dts/renesas/rzg2ul-smarc-som.dtsi | 20 +- + arch/arm64/boot/dts/renesas/rzg2ul-sr-som.dtsi | 418 +++++++++++++++ + arch/arm64/boot/dts/renesas/rzg3l-smarc-som.dtsi | 25 + + arch/arm64/boot/dts/renesas/rzg3s-smarc-som.dtsi | 25 +- + arch/arm64/boot/dts/renesas/rzg3s-smarc-switches.h | 4 + + arch/arm64/boot/dts/renesas/rzg3s-smarc.dtsi | 1 - + .../boot/dts/renesas/rzt2h-n2h-evk-common.dtsi | 7 + + .../boot/dts/renesas/rzt2h-n2h-evk-du-adv7513.dtsi | 72 +++ + arch/arm64/boot/dts/renesas/salvator-common.dtsi | 14 +- + arch/arm64/boot/dts/renesas/salvator-xs.dtsi | 2 +- + .../ulcb-kf-audio-graph-card2-mix+split.dtsi | 18 +- + arch/arm64/boot/dts/renesas/ulcb-kf.dtsi | 2 +- + arch/arm64/boot/dts/renesas/ulcb.dtsi | 8 +- + .../boot/dts/renesas/white-hawk-cpu-common.dtsi | 6 +- + drivers/soc/renesas/Kconfig | 22 +- + drivers/soc/renesas/r9a08g046-sysc.c | 1 + + drivers/soc/renesas/rcar-mfis.c | 62 ++- + drivers/soc/renesas/rcar-rst.c | 1 + + drivers/soc/renesas/renesas-soc.c | 3 + + drivers/soc/renesas/rz-sysc.c | 5 + + drivers/soc/renesas/rz-sysc.h | 2 + + .../dt-bindings/clock/renesas,r8a774a3-cpg-mssr.h | 59 +++ + include/dt-bindings/clock/renesas,r8a78000-cpg.h | 2 + + include/dt-bindings/power/qcom,rpmhpd.h | 1 + + include/dt-bindings/power/renesas,r8a774a3-sysc.h | 30 ++ + 147 files changed, 5734 insertions(+), 422 deletions(-) + create mode 100644 Documentation/devicetree/bindings/power/riscv,rpmi-device-power.yaml + create mode 100644 Documentation/devicetree/bindings/power/riscv,rpmi-mpxy-device-power.yaml + create mode 100644 arch/arm64/boot/dts/renesas/r8a774a3-hihope-rzg2m-ex.dts + create mode 100644 arch/arm64/boot/dts/renesas/r8a774a3-hihope-rzg2m.dts + create mode 100644 arch/arm64/boot/dts/renesas/r8a774a3.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/r8a779g3-sparrow-hawk-ws-2ch-canfd.dtso + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g043u12-hummingboard-ripple.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g044c2-hummingboard-iiot.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g044c2-hummingboard-ripple.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g044l2-hummingboard-iiot.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g044l2-hummingboard-ripple.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g054l2-hummingboard-iiot.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a07g054l2-hummingboard-ripple.dts + create mode 100644 arch/arm64/boot/dts/renesas/r9a09g077m44-evk-cn15-lcdc.dtso + create mode 100644 arch/arm64/boot/dts/renesas/r9a09g087m44-evk-cn20-lcdc.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-common.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-microsd.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-panel-dsi-WJ70N3TYJHMNG0.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-rs485-a.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot-rs485-b.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-iiot.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-pulse-common.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-pulse-micro-hdmi.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-hummingboard-ripple.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-sr-som-emmc.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2l-sr-som.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2lc-hummingboard-pulse-leds.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2lc-sr-som.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzg2ul-hummingboard-pulse-leds.dtso + create mode 100644 arch/arm64/boot/dts/renesas/rzg2ul-sr-som.dtsi + create mode 100644 arch/arm64/boot/dts/renesas/rzt2h-n2h-evk-du-adv7513.dtsi + create mode 100644 include/dt-bindings/clock/renesas,r8a774a3-cpg-mssr.h + create mode 100644 include/dt-bindings/power/renesas,r8a774a3-sysc.h +Merging reset/reset/next (d373605cd5148 Merge tag 'reset-fixes-for-v7.0-2' into reset/next) +$ git merge -m Merge branch 'reset/next' of https://git.kernel.org/pub/scm/linux/kernel/git/pza/linux reset/reset/next +Already up to date. +Merging rockchip/for-next (23d8f49fcae5d Merge branch 'v7.4-armsoc/dts64' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mmind/linux-rockchip.git rockchip/for-next +Merge made by the 'ort' strategy. + .../devicetree/bindings/arm/rockchip.yaml | 7 + + arch/arm/boot/dts/rockchip/rk3229-xms6.dts | 4 +- + arch/arm64/boot/dts/rockchip/Makefile | 6 + + .../boot/dts/rockchip/rk3399pro-vmarc-som.dtsi | 2 + + .../boot/dts/rockchip/rk3576-armsom-cm5-io.dts | 4 +- + arch/arm64/boot/dts/rockchip/rk3576.dtsi | 7 +- + arch/arm64/boot/dts/rockchip/rk3588-base.dtsi | 39 + + .../rockchip/rk3588-jaguar-can1-can2-uart4.dtso | 26 + + arch/arm64/boot/dts/rockchip/rk3588-jaguar.dts | 16 + + .../boot/dts/rockchip/rk3588-lubancat-5-btb.dtsi | 534 ++++++++++ + .../boot/dts/rockchip/rk3588-lubancat-5io.dts | 1032 ++++++++++++++++++++ + .../boot/dts/rockchip/rk3588-nanopc-t6-lts.dts | 17 - + arch/arm64/boot/dts/rockchip/rk3588-nanopc-t6.dtsi | 175 ++-- + .../boot/dts/rockchip/rk3588-tiger-haikou.dts | 4 + + arch/arm64/boot/dts/rockchip/rk3588-tiger.dtsi | 5 + + .../boot/dts/rockchip/rk3588s-gameforce-ace.dts | 7 +- + 16 files changed, 1760 insertions(+), 125 deletions(-) + create mode 100644 arch/arm64/boot/dts/rockchip/rk3588-jaguar-can1-can2-uart4.dtso + create mode 100644 arch/arm64/boot/dts/rockchip/rk3588-lubancat-5-btb.dtsi + create mode 100644 arch/arm64/boot/dts/rockchip/rk3588-lubancat-5io.dts +Merging samsung-krzk/for-next (08df370772f32 Merge branch 'for-v7.4/google-lga' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux.git samsung-krzk/for-next +Auto-merging Documentation/devicetree/bindings/sram/sram.yaml +Auto-merging MAINTAINERS +Auto-merging arch/arm64/Kconfig.platforms +Auto-merging arch/arm64/configs/defconfig +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/google.yaml | 48 +- + .../bindings/arm/samsung/samsung-boards.yaml | 1 + + .../bindings/clock/samsung,exynos5515-cmu.yaml | 145 ++ + .../devicetree/bindings/gpu/arm,mali-bifrost.yaml | 1 + + .../bindings/hwinfo/samsung,exynos-chipid.yaml | 1 + + .../bindings/input/samsung,s3c6410-keypad.yaml | 53 +- + Documentation/devicetree/bindings/sram/sram.yaml | 2 + + MAINTAINERS | 1 + + arch/arm/boot/dts/samsung/exynos4210-i9100.dts | 9 +- + arch/arm/boot/dts/samsung/exynos4210-trats.dts | 5 + + .../boot/dts/samsung/exynos4210-universal_c210.dts | 5 + + arch/arm/boot/dts/samsung/exynos4412-midas.dtsi | 5 +- + arch/arm/mach-exynos/smc.h | 4 +- + arch/arm/mach-s3c/Kconfig | 5 - + arch/arm/mach-s3c/Kconfig.s3c64xx | 7 - + arch/arm/mach-s3c/Makefile.s3c64xx | 1 - + arch/arm/mach-s3c/devs.c | 27 - + arch/arm/mach-s3c/devs.h | 1 - + arch/arm/mach-s3c/gpio-core.h | 3 + + arch/arm/mach-s3c/gpio-samsung-s3c64xx.h | 5 + + arch/arm/mach-s3c/gpio-samsung.c | 72 +- + arch/arm/mach-s3c/keypad.h | 27 - + arch/arm/mach-s3c/mach-crag6410.c | 135 +- + arch/arm/mach-s3c/setup-keypad-s3c64xx.c | 20 - + arch/arm64/Kconfig.platforms | 8 + + arch/arm64/boot/dts/Makefile | 1 + + arch/arm64/boot/dts/exynos/Makefile | 1 + + arch/arm64/boot/dts/exynos/axis/artpec8.dtsi | 36 +- + arch/arm64/boot/dts/exynos/axis/artpec9.dtsi | 36 +- + arch/arm64/boot/dts/exynos/exynos2200-g0s.dts | 8 +- + arch/arm64/boot/dts/exynos/exynos2200-pinctrl.dtsi | 2 - + arch/arm64/boot/dts/exynos/exynos2200.dtsi | 1 - + .../boot/dts/exynos/exynos5433-tm2-common.dtsi | 93 +- + arch/arm64/boot/dts/exynos/exynos5433.dtsi | 108 +- + .../arm64/boot/dts/exynos/exynos7870-a2corelte.dts | 39 +- + arch/arm64/boot/dts/exynos/exynos7870-j5y17lte.dts | 34 +- + arch/arm64/boot/dts/exynos/exynos7870-j6lte.dts | 34 +- + arch/arm64/boot/dts/exynos/exynos7870-j7xelte.dts | 35 +- + arch/arm64/boot/dts/exynos/exynos7870-on7xelte.dts | 39 +- + arch/arm64/boot/dts/exynos/exynos7870.dtsi | 108 +- + arch/arm64/boot/dts/exynos/exynos850-a217f.dts | 200 +++ + arch/arm64/boot/dts/exynos/exynos850.dtsi | 29 +- + arch/arm64/boot/dts/exynos/exynos8855-smdk.dts | 2 +- + arch/arm64/boot/dts/exynos/exynosautov920.dtsi | 2 +- + .../boot/dts/exynos/google/gs101-pixel-common.dtsi | 51 +- + arch/arm64/boot/dts/exynos/google/gs101.dtsi | 96 +- + arch/arm64/boot/dts/google/Makefile | 6 + + arch/arm64/boot/dts/google/lga-blazer.dts | 15 + + arch/arm64/boot/dts/google/lga-frankel.dts | 15 + + arch/arm64/boot/dts/google/lga-mustang.dts | 15 + + arch/arm64/boot/dts/google/lga-pixel-common.dtsi | 22 + + arch/arm64/boot/dts/google/lga.dtsi | 411 ++++++ + arch/arm64/boot/dts/tesla/fsd-evb.dts | 10 +- + arch/arm64/boot/dts/tesla/fsd-pinctrl.dtsi | 16 +- + arch/arm64/boot/dts/tesla/fsd.dtsi | 1162 +++++++-------- + arch/arm64/configs/defconfig | 1 + + drivers/clk/samsung/Makefile | 1 + + drivers/clk/samsung/clk-acpm.c | 61 +- + drivers/clk/samsung/clk-exynos5515.c | 1484 ++++++++++++++++++++ + drivers/clk/samsung/clk-pll.c | 1 + + drivers/clk/samsung/clk-pll.h | 1 + + drivers/input/keyboard/samsung-keypad.c | 200 ++- + drivers/soc/samsung/exynos-chipid.c | 1 + + include/dt-bindings/clock/samsung,exynos5515-cmu.h | 191 +++ + include/linux/input/samsung-keypad.h | 39 - + 65 files changed, 3922 insertions(+), 1276 deletions(-) + create mode 100644 Documentation/devicetree/bindings/clock/samsung,exynos5515-cmu.yaml + delete mode 100644 arch/arm/mach-s3c/keypad.h + delete mode 100644 arch/arm/mach-s3c/setup-keypad-s3c64xx.c + create mode 100644 arch/arm64/boot/dts/exynos/exynos850-a217f.dts + create mode 100644 arch/arm64/boot/dts/google/Makefile + create mode 100644 arch/arm64/boot/dts/google/lga-blazer.dts + create mode 100644 arch/arm64/boot/dts/google/lga-frankel.dts + create mode 100644 arch/arm64/boot/dts/google/lga-mustang.dts + create mode 100644 arch/arm64/boot/dts/google/lga-pixel-common.dtsi + create mode 100644 arch/arm64/boot/dts/google/lga.dtsi + create mode 100644 drivers/clk/samsung/clk-exynos5515.c + create mode 100644 include/dt-bindings/clock/samsung,exynos5515-cmu.h + delete mode 100644 include/linux/input/samsung-keypad.h +Merging scmi/for-linux-next (0dcfd0f40b5be Merge branches 'for-next/juno/updates' and 'for-next/scmi/updates' of git://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux) +$ git merge -m Merge branch 'for-linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sudeep.holla/linux.git scmi/for-linux-next +Auto-merging MAINTAINERS +Auto-merging scripts/mod/devicetable-offsets.c +Auto-merging scripts/mod/file2alias.c +Merge made by the 'ort' strategy. + MAINTAINERS | 1 + + arch/arm64/boot/dts/arm/fvp-base-revc.dts | 42 +- + drivers/firmware/arm_scmi/bus.c | 109 ++- + drivers/firmware/arm_scmi/common.h | 61 +- + drivers/firmware/arm_scmi/driver.c | 262 +++++--- + drivers/firmware/arm_scmi/protocols.h | 9 +- + drivers/firmware/arm_scmi/raw_mode.c | 30 +- + drivers/firmware/arm_scmi/reset.c | 36 +- + drivers/firmware/arm_scmi/transports/Kconfig | 12 + + drivers/firmware/arm_scmi/transports/Makefile | 2 + + drivers/firmware/arm_scmi/transports/mailbox.c | 7 +- + drivers/firmware/arm_scmi/transports/optee.c | 7 +- + drivers/firmware/arm_scmi/transports/pcc.c | 875 +++++++++++++++++++++++++ + drivers/firmware/arm_scmi/transports/smc.c | 10 +- + drivers/firmware/arm_scmi/transports/virtio.c | 3 +- + include/linux/device-id/scmi.h | 17 + + include/linux/scmi_protocol.h | 20 +- + include/trace/events/scmi.h | 8 +- + scripts/mod/devicetable-offsets.c | 5 + + scripts/mod/file2alias.c | 12 + + 20 files changed, 1342 insertions(+), 186 deletions(-) + create mode 100644 drivers/firmware/arm_scmi/transports/pcc.c + create mode 100644 include/linux/device-id/scmi.h +Merging sophgo/for-next (76acfee87c74d Merge branch 'dt/arm' into for-next) +$ git merge -m Merge branch 'for-next' of https://github.com/sophgo/linux.git sophgo/for-next +Merge made by the 'ort' strategy. +Merging sophgo-soc/soc-for-next (c8754c7deab4c soc: sophgo: cv1800: rtcsys: New driver (handling RTC only)) +$ git merge -m Merge branch 'soc-for-next' of https://github.com/sophgo/linux.git sophgo-soc/soc-for-next +Already up to date. +Merging spacemit/for-next (4b98722be2911 Merge branch 'spacemit-misc-for-next' into spacemit-for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/spacemit/linux spacemit/for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 9 ++ + arch/riscv/boot/dts/spacemit/k1-musepi-pro.dts | 35 ++++++- + arch/riscv/boot/dts/spacemit/k1-orangepi-r2s.dts | 65 +++++++++++- + arch/riscv/boot/dts/spacemit/k1-orangepi-rv2.dts | 126 +++++++++++++++++------ + arch/riscv/boot/dts/spacemit/k3-com260.dtsi | 17 +++ + arch/riscv/boot/dts/spacemit/k3-pinctrl.dtsi | 11 ++ + arch/riscv/boot/dts/spacemit/k3.dtsi | 123 +++++++++++++++++++++- + 7 files changed, 350 insertions(+), 36 deletions(-) +Merging stm32/stm32-next (e6cdc9f49eebe arm64: dts: st: fix sai3b unit address on stm32mp231) +$ git merge -m Merge branch 'stm32-next' of https://git.kernel.org/pub/scm/linux/kernel/git/atorgue/stm32.git stm32/stm32-next +Auto-merging MAINTAINERS +Auto-merging arch/arm64/configs/defconfig +Merge made by the 'ort' strategy. + .../devicetree/bindings/arm/stm32/stm32.yaml | 39 +- + MAINTAINERS | 3 +- + arch/arm/boot/dts/st/stm32mp131.dtsi | 24 + + arch/arm/boot/dts/st/stm32mp135.dtsi | 44 +- + arch/arm/boot/dts/st/stm32mp135f-dk.dts | 28 +- + arch/arm/boot/dts/st/stm32mp15-pinctrl.dtsi | 34 + + arch/arm/boot/dts/st/stm32mp151.dtsi | 20 +- + .../arm/boot/dts/st/stm32mp153c-lxa-fairytux2.dtsi | 2 +- + arch/arm/boot/dts/st/stm32mp157a-dk1-scmi.dts | 8 +- + arch/arm/boot/dts/st/stm32mp157c-dk2-scmi.dts | 8 +- + arch/arm/boot/dts/st/stm32mp157c-ed1-scmi.dts | 8 +- + arch/arm/boot/dts/st/stm32mp157c-ev1-scmi.dts | 8 +- + arch/arm/boot/dts/st/stm32mp157c-ev1.dts | 9 +- + arch/arm/boot/dts/st/stm32mp157c-lxa-mc1.dts | 2 +- + arch/arm/boot/dts/st/stm32mp157f-dk2-scmi.dtsi | 1 + + arch/arm/boot/dts/st/stm32mp15xc-lxa-tac.dtsi | 2 +- + arch/arm/boot/dts/st/stm32mp15xx-dhcom-som.dtsi | 2 + + arch/arm/boot/dts/st/stm32mp15xx-dkx.dtsi | 35 +- + arch/arm64/boot/dts/st/Makefile | 23 + + arch/arm64/boot/dts/st/stm32mp231.dtsi | 105 ++- + arch/arm64/boot/dts/st/stm32mp235f-dk.dts | 75 ++ + arch/arm64/boot/dts/st/stm32mp23xc.dtsi | 7 + + arch/arm64/boot/dts/st/stm32mp23xx-dhcos-bb.dts | 15 + + arch/arm64/boot/dts/st/stm32mp23xx-dhcos-som.dtsi | 51 ++ + arch/arm64/boot/dts/st/stm32mp25-pinctrl.dtsi | 872 +++++++++++++++++++++ + arch/arm64/boot/dts/st/stm32mp251.dtsi | 67 +- + arch/arm64/boot/dts/st/stm32mp253.dtsi | 16 + + ...stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtso | 111 +++ + arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc.dts | 189 +++++ + .../boot/dts/st/stm32mp257-engicam-microgea.dtsi | 63 ++ + ...microgea-rmm-overlay-am-1280800w8tzqw-t00h.dtso | 67 ++ + ...cam-microgea-rmm-overlay-rk050hr345-ct106a.dtso | 68 ++ + .../dts/st/stm32mp257d-engicam-microgea-rmm.dts | 275 +++++++ + arch/arm64/boot/dts/st/stm32mp257f-dk.dts | 93 +++ + arch/arm64/boot/dts/st/stm32mp257f-ev1.dts | 101 ++- + arch/arm64/boot/dts/st/stm32mp25xc.dtsi | 7 + + arch/arm64/boot/dts/st/stm32mp25xx-dhcos-bb.dts | 15 + + arch/arm64/boot/dts/st/stm32mp25xx-dhcos-som.dtsi | 51 ++ + arch/arm64/boot/dts/st/stm32mp2xxx-dhcos-som.dtsi | 452 +++++++++++ + arch/arm64/configs/defconfig | 5 + + 40 files changed, 2862 insertions(+), 143 deletions(-) + create mode 100644 arch/arm64/boot/dts/st/stm32mp23xc.dtsi + create mode 100644 arch/arm64/boot/dts/st/stm32mp23xx-dhcos-bb.dts + create mode 100644 arch/arm64/boot/dts/st/stm32mp23xx-dhcos-som.dtsi + create mode 100644 arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc-overlay-imx219-x10.dtso + create mode 100644 arch/arm64/boot/dts/st/stm32mp255c-dhcos-dhsbc.dts + create mode 100644 arch/arm64/boot/dts/st/stm32mp257-engicam-microgea.dtsi + create mode 100644 arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm-overlay-am-1280800w8tzqw-t00h.dtso + create mode 100644 arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm-overlay-rk050hr345-ct106a.dtso + create mode 100644 arch/arm64/boot/dts/st/stm32mp257d-engicam-microgea-rmm.dts + create mode 100644 arch/arm64/boot/dts/st/stm32mp25xc.dtsi + create mode 100644 arch/arm64/boot/dts/st/stm32mp25xx-dhcos-bb.dts + create mode 100644 arch/arm64/boot/dts/st/stm32mp25xx-dhcos-som.dtsi + create mode 100644 arch/arm64/boot/dts/st/stm32mp2xxx-dhcos-som.dtsi +Merging sunxi/sunxi/for-next (afb622bddaa6f Merge branches 'sunxi/dt-for-7.4' and 'sunxi/clk-for-7.4' into sunxi/for-next) +$ git merge -m Merge branch 'sunxi/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sunxi/linux.git sunxi/sunxi/for-next +Auto-merging Documentation/devicetree/bindings/vendor-prefixes.yaml +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/sunxi.yaml | 10 + + .../devicetree/bindings/vendor-prefixes.yaml | 2 + + arch/arm/boot/dts/allwinner/sun7i-a20.dtsi | 2 +- + arch/arm64/boot/dts/allwinner/Makefile | 1 + + .../boot/dts/allwinner/sun50i-a133-teclast-p80.dts | 205 +++++++++++++++++++++ + drivers/clk/sunxi-ng/ccu-sun55i-a523-r.c | 37 ++-- + drivers/clk/sunxi-ng/ccu-sun55i-a523.c | 128 ++++++------- + drivers/clk/sunxi-ng/ccu_common.c | 2 +- + drivers/clk/sunxi-ng/ccu_div.h | 18 +- + 9 files changed, 308 insertions(+), 97 deletions(-) + create mode 100644 arch/arm64/boot/dts/allwinner/sun50i-a133-teclast-p80.dts +Merging tee/next (a4df08db64528 Merge branches 'qcomtee_fix_for_v7.3', 'qcomtee_for_v7.4', 'tee_for_v7.4' and 'otpee_for_v7.4' into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/jenswi/linux-tee.git tee/next +Merge made by the 'ort' strategy. + drivers/tee/optee/Kconfig | 1 + + drivers/tee/optee/Makefile | 4 ++-- + drivers/tee/optee/ffa_abi.c | 24 +++++++++------------- + drivers/tee/optee/notif.c | 1 - + drivers/tee/optee/optee_private.h | 39 ++++++++++++++++++++++++++++++++++- + drivers/tee/optee/smc_abi.c | 17 ++++++++------- + drivers/tee/qcomtee/async.c | 4 ++-- + drivers/tee/qcomtee/qcomtee_msg.h | 2 +- + drivers/tee/qcomtee/qcomtee_object.h | 1 + + include/uapi/linux/tee.h | 40 ++++++++++++++++++------------------ + 10 files changed, 83 insertions(+), 50 deletions(-) +Merging tegra/for-next (11430072551f5 Merge branch for-7.4/arm64/dt into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tegra/linux.git tegra/for-next +Auto-merging drivers/soc/tegra/pmc.c +Merge made by the 'ort' strategy. + .../bindings/fuse/nvidia,tegra20-fuse.yaml | 1 + + .../memory-controllers/nvidia,tegra124-emc.yaml | 10 +- + .../memory-controllers/nvidia,tegra124-mc.yaml | 1 + + arch/arm/boot/dts/nvidia/tegra124.dtsi | 16 +- + arch/arm64/boot/dts/nvidia/tegra132.dtsi | 16 +- + .../dts/nvidia/tegra186-p3509-0000+p3636-0001.dts | 8 +- + arch/arm64/boot/dts/nvidia/tegra186.dtsi | 74 +-- + arch/arm64/boot/dts/nvidia/tegra194-p2972-0000.dts | 8 +- + .../arm64/boot/dts/nvidia/tegra194-p3509-0000.dtsi | 8 +- + arch/arm64/boot/dts/nvidia/tegra194.dtsi | 78 +-- + arch/arm64/boot/dts/nvidia/tegra210-p3450-0000.dts | 10 +- + arch/arm64/boot/dts/nvidia/tegra210.dtsi | 6 +- + .../boot/dts/nvidia/tegra234-p3737-0000+p3701.dtsi | 9 +- + .../dts/nvidia/tegra234-p3740-0002+p3701-0008.dts | 45 +- + .../boot/dts/nvidia/tegra234-p3768-0000+p3767.dtsi | 2 +- + arch/arm64/boot/dts/nvidia/tegra234.dtsi | 74 +-- + arch/arm64/boot/dts/nvidia/tegra264-p3834.dtsi | 57 ++ + .../boot/dts/nvidia/tegra264-p4071-0000+p3834.dtsi | 93 ++++ + arch/arm64/boot/dts/nvidia/tegra264.dtsi | 580 ++++++++++++--------- + .../boot/dts/nvidia/thermal/nvidia,tegra264-bpmp.h | 20 + + drivers/soc/tegra/fuse/fuse-tegra.c | 11 +- + drivers/soc/tegra/fuse/fuse-tegra30.c | 168 +++++- + drivers/soc/tegra/fuse/fuse.h | 7 +- + drivers/soc/tegra/pmc.c | 98 ++++ + 24 files changed, 972 insertions(+), 428 deletions(-) + create mode 100644 arch/arm64/boot/dts/nvidia/thermal/nvidia,tegra264-bpmp.h +Merging tenstorrent-dt/tenstorrent-dt-for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'tenstorrent-dt-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git tenstorrent-dt/tenstorrent-dt-for-next +Already up to date. +Merging fustini-config/riscv-config-for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'riscv-config-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git fustini-config/riscv-config-for-next +Already up to date. +Merging thead-dt/thead-dt-for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'thead-dt-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git thead-dt/thead-dt-for-next +Already up to date. +Merging ti/ti-next (ba727be7fe4a6 Merge branches 'ti-drivers-soc-next' and 'ti-k3-dts-next' into ti-next) +$ git merge -m Merge branch 'ti-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ti/linux.git ti/ti-next +Auto-merging MAINTAINERS +Auto-merging arch/arm64/configs/defconfig +Auto-merging drivers/firmware/ti_sci.c +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/ti/k3.yaml | 7 + + .../bindings/clock/ti,am62-audio-refclk.yaml | 8 +- + .../soc/ti/ti,j721e-system-controller.yaml | 6 +- + MAINTAINERS | 1 + + arch/arm/boot/dts/ti/keystone/keystone-k2e.dtsi | 2 +- + arch/arm/boot/dts/ti/keystone/keystone-k2g.dtsi | 28 +- + arch/arm/boot/dts/ti/keystone/keystone-k2hk.dtsi | 16 +- + arch/arm/boot/dts/ti/keystone/keystone-k2l.dtsi | 16 +- + arch/arm/boot/dts/ti/keystone/keystone.dtsi | 4 +- + arch/arm64/boot/dts/ti/Makefile | 12 + + arch/arm64/boot/dts/ti/k3-am62-lp-sk.dts | 2 +- + arch/arm64/boot/dts/ti/k3-am62-phycore-som.dtsi | 2 +- + arch/arm64/boot/dts/ti/k3-am62-pocketbeagle2.dts | 2 +- + arch/arm64/boot/dts/ti/k3-am62-verdin-ivy.dtsi | 41 +- + arch/arm64/boot/dts/ti/k3-am62-verdin-zinnia.dtsi | 41 +- + arch/arm64/boot/dts/ti/k3-am62-verdin.dtsi | 49 +-- + arch/arm64/boot/dts/ti/k3-am625-beagleplay.dts | 6 +- + arch/arm64/boot/dts/ti/k3-am625-sk-common.dtsi | 3 +- + .../boot/dts/ti/k3-am625-tqma62xx-mba62xx.dts | 4 +- + arch/arm64/boot/dts/ti/k3-am625-tqma62xx.dtsi | 2 +- + .../boot/dts/ti/k3-am625-var-som-symphony.dts | 2 +- + arch/arm64/boot/dts/ti/k3-am625-var-som.dtsi | 4 +- + arch/arm64/boot/dts/ti/k3-am62a-main.dtsi | 1 + + arch/arm64/boot/dts/ti/k3-am62a-mcu.dtsi | 2 +- + arch/arm64/boot/dts/ti/k3-am62a-phycore-som.dtsi | 2 +- + arch/arm64/boot/dts/ti/k3-am62a7-sk.dts | 6 +- + arch/arm64/boot/dts/ti/k3-am62l3-evm.dts | 4 +- + .../boot/dts/ti/k3-am62p-j722s-common-main.dtsi | 12 + + .../boot/dts/ti/k3-am62p-j722s-common-mcu.dtsi | 2 +- + arch/arm64/boot/dts/ti/k3-am62x-phyboard-lyra.dtsi | 2 +- + arch/arm64/boot/dts/ti/k3-am62x-sk-common.dtsi | 2 +- + arch/arm64/boot/dts/ti/k3-am654-base-board.dts | 1 + + arch/arm64/boot/dts/ti/k3-am67-phycore-som.dtsi | 324 ++++++++++++++++ + .../arm64/boot/dts/ti/k3-am6754-phyboard-rigel.dts | 431 +++++++++++++++++++++ + ...uila-dsi-to-lvds-v2-panel-cap-touch-10inch.dtso | 161 ++++++++ + arch/arm64/boot/dts/ti/k3-am69-aquila.dtsi | 2 +- + .../boot/dts/ti/k3-j7200-common-proc-board.dts | 1 + + .../boot/dts/ti/k3-j721s2-common-proc-board.dts | 1 + + arch/arm64/boot/dts/ti/k3-j721s2-evm-audio.dtso | 157 ++++++++ + arch/arm64/boot/dts/ti/k3-j721s2-main.dtsi | 18 + + arch/arm64/boot/dts/ti/k3-j722s-evm.dts | 26 +- + .../boot/dts/ti/k3-j784s4-j742s2-evm-common.dtsi | 2 + + arch/arm64/boot/dts/ti/k3-pinctrl.h | 10 + + arch/arm64/configs/defconfig | 15 +- + drivers/firmware/ti_sci.c | 6 +- + drivers/soc/ti/knav_qmss_queue.c | 4 +- + include/linux/soc/ti/k3-ringacc.h | 31 +- + 47 files changed, 1260 insertions(+), 221 deletions(-) + create mode 100644 arch/arm64/boot/dts/ti/k3-am67-phycore-som.dtsi + create mode 100644 arch/arm64/boot/dts/ti/k3-am6754-phyboard-rigel.dts + create mode 100644 arch/arm64/boot/dts/ti/k3-am69-aquila-dsi-to-lvds-v2-panel-cap-touch-10inch.dtso + create mode 100644 arch/arm64/boot/dts/ti/k3-j721s2-evm-audio.dtso +Merging xilinx/for-next (bf126d9aa20f2 Merge branch 'zynqmp/soc' into for-next) +$ git merge -m Merge branch 'for-next' of https://github.com/Xilinx/linux-xlnx.git xilinx/for-next +Merge made by the 'ort' strategy. + arch/arm/boot/dts/xilinx/zynq-7000.dtsi | 8 ++-- + arch/arm/boot/dts/xilinx/zynq-parallella.dts | 4 +- + drivers/firmware/xilinx/zynqmp-ufs.c | 70 ++++++++++++++++++++++++++-- + drivers/firmware/xilinx/zynqmp.c | 44 ++++++++++++++--- + drivers/ufs/host/ufs-amd-versal2.c | 46 +++++------------- + include/linux/firmware/xlnx-zynqmp-ufs.h | 8 ++-- + include/linux/firmware/xlnx-zynqmp.h | 5 -- + 7 files changed, 126 insertions(+), 59 deletions(-) +Merging socfpga/for-next (04a1d21330501 arm64: dts: socfpga: agilex5: Add SoCDK TSN Config2 board) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/dinguyen/linux.git socfpga/for-next +Merge made by the 'ort' strategy. + Documentation/devicetree/bindings/arm/altera.yaml | 2 + + .../bindings/firmware/intel,stratix10-svc.yaml | 10 ++ + arch/arm64/boot/dts/altera/socfpga_stratix10.dtsi | 4 + + arch/arm64/boot/dts/intel/Makefile | 2 + + arch/arm64/boot/dts/intel/socfpga_agilex.dtsi | 2 + + arch/arm64/boot/dts/intel/socfpga_agilex5.dtsi | 61 ++++++++++ + .../arm64/boot/dts/intel/socfpga_agilex5_socdk.dts | 38 +++++- + .../boot/dts/intel/socfpga_agilex5_socdk_emmc.dts | 123 +++++++++++++++++++ + .../dts/intel/socfpga_agilex5_socdk_tsn_cfg2.dts | 130 +++++++++++++++++++++ + 9 files changed, 369 insertions(+), 3 deletions(-) + create mode 100644 arch/arm64/boot/dts/intel/socfpga_agilex5_socdk_emmc.dts + create mode 100644 arch/arm64/boot/dts/intel/socfpga_agilex5_socdk_tsn_cfg2.dts +Merging clk/clk-next (b1470947e92ae Merge branch 'clk-pile' into clk-next) +$ git merge -m Merge branch 'clk-next' of https://git.kernel.org/pub/scm/linux/kernel/git/clk/linux.git clk/clk-next +Auto-merging MAINTAINERS +Auto-merging drivers/clk/at91/clk-main.c +Auto-merging drivers/clk/clk-scpi.c +Auto-merging drivers/clk/renesas/r9a06g032-clocks.c +Auto-merging drivers/clk/renesas/renesas-cpg-mssr.c +Auto-merging drivers/clk/renesas/rzg2l-cpg.c +Auto-merging drivers/clk/renesas/rzv2h-cpg.c +Auto-merging drivers/clk/samsung/clk-pll.c +Merge made by the 'ort' strategy. + .../bindings/clock/airoha,en7523-scu.yaml | 24 +- + .../bindings/clock/anlogic,dr1v90-cru.yaml | 60 +++ + .../bindings/clock/clk-palmas-clk32kg-clocks.txt | 35 -- + .../bindings/clock/clock-nexus-node.yaml | 30 ++ + .../bindings/clock/ti,palmas-clk32kg.yaml | 56 +++ + .../devicetree/bindings/clock/ti/adpll.txt | 39 -- + .../devicetree/bindings/clock/ti/davinci/pll.txt | 96 ---- + .../bindings/clock/ti/davinci/ti,da850-pll.yaml | 164 ++++++ + .../devicetree/bindings/clock/ti/dra7-atl.txt | 94 ---- + .../devicetree/bindings/clock/ti/fapll.txt | 31 -- + .../bindings/clock/ti/ti,dm814-adpll-clock.yaml | 88 ++++ + .../bindings/clock/ti/ti,dm816-fapll-clock.yaml | 71 +++ + .../bindings/clock/ti/ti,dra7-atl-clock.yaml | 38 ++ + .../devicetree/bindings/clock/ti/ti,dra7-atl.yaml | 100 ++++ + MAINTAINERS | 22 +- + arch/riscv/boot/dts/anlogic/dr1v90-mlkpai-fs01.dts | 4 + + arch/riscv/boot/dts/anlogic/dr1v90.dtsi | 40 +- + drivers/clk/Kconfig | 5 + + drivers/clk/Makefile | 8 +- + drivers/clk/actions/owl-composite.h | 1 - + drivers/clk/actions/owl-fixed-factor.h | 2 - + drivers/clk/actions/owl-pll.c | 2 +- + drivers/clk/anlogic/Kconfig | 21 + + drivers/clk/anlogic/Makefile | 7 + + drivers/clk/anlogic/cru-dr1v90.c | 192 +++++++ + drivers/clk/anlogic/cru_dr1.c | 226 +++++++++ + drivers/clk/anlogic/cru_dr1.h | 117 +++++ + drivers/clk/aspeed/Kconfig | 1 + + drivers/clk/aspeed/clk-aspeed.c | 2 +- + drivers/clk/aspeed/clk-ast2600.c | 2 +- + drivers/clk/aspeed/clk-ast2700.c | 2 +- + drivers/clk/at91/clk-audio-pll.c | 4 +- + drivers/clk/at91/clk-h32mx.c | 2 +- + drivers/clk/at91/clk-main.c | 2 +- + drivers/clk/at91/clk-pll.c | 2 +- + drivers/clk/at91/clk-plldiv.c | 2 +- + drivers/clk/at91/clk-slow.c | 2 +- + drivers/clk/at91/clk-smd.c | 2 +- + drivers/clk/at91/clk-usb.c | 6 +- + drivers/clk/at91/sckc.c | 2 +- + drivers/clk/bcm/clk-iproc-armpll.c | 2 +- + drivers/clk/bcm/clk-iproc-asiu.c | 2 +- + drivers/clk/bcm/clk-iproc-pll.c | 2 +- + drivers/clk/bcm/clk-raspberrypi.c | 2 + + drivers/clk/berlin/berlin2-avpll.c | 4 +- + drivers/clk/berlin/berlin2-pll.c | 2 +- + drivers/clk/clk-axi-clkgen.c | 2 +- + drivers/clk/clk-cdce925.c | 2 +- + drivers/clk/clk-conf.c | 12 +- + drivers/clk/clk-cs2000-cp.c | 2 +- + drivers/clk/clk-en7523.c | 218 +++++++- + drivers/clk/clk-eyeq.c | 31 +- + drivers/clk/clk-fractional-divider.c | 2 +- + drivers/clk/clk-gemini.c | 2 +- + drivers/clk/clk-highbank.c | 2 +- + drivers/clk/clk-lmk04832.c | 6 +- + drivers/clk/clk-milbeaut.c | 14 +- + drivers/clk/clk-nomadik.c | 4 +- + drivers/clk/clk-npcm7xx.c | 2 +- + drivers/clk/clk-pwm.c | 2 +- + drivers/clk/clk-renesas-pcie.c | 43 +- + drivers/clk/clk-s2mps11.c | 12 +- + drivers/clk/clk-scpi.c | 2 +- + drivers/clk/clk-si514.c | 2 +- + drivers/clk/clk-si521xx.c | 43 +- + drivers/clk/clk-si5341.c | 2 +- + drivers/clk/clk-si5351.c | 20 +- + drivers/clk/clk-si544.c | 2 +- + drivers/clk/clk-si570.c | 2 +- + drivers/clk/clk-stm32f4.c | 4 +- + drivers/clk/clk-stm32h7.c | 2 +- + drivers/clk/clk-versaclock5.c | 12 +- + drivers/clk/clk-vt8500.c | 4 +- + drivers/clk/clk-xgene.c | 6 +- + drivers/clk/clk.c | 75 ++- + drivers/clk/clk.h | 5 +- + drivers/clk/clk_kunit_helpers.c | 31 ++ + drivers/clk/clk_test.c | 151 +++++- + drivers/clk/clkdev.c | 4 +- + drivers/clk/davinci/Kconfig | 10 + + drivers/clk/davinci/Makefile | 8 +- + drivers/clk/davinci/da8xx-cfgchip.c | 4 +- + drivers/clk/davinci/pll.c | 2 +- + drivers/clk/davinci/psc.c | 2 +- + drivers/clk/hisilicon/clk-hi3559a.c | 2 +- + drivers/clk/hisilicon/clk-hi3620.c | 2 +- + drivers/clk/hisilicon/clk-hi6220-stub.c | 2 +- + drivers/clk/hisilicon/clk-hisi-phase.c | 2 +- + drivers/clk/hisilicon/clk-hix5hd2.c | 2 +- + drivers/clk/hisilicon/clkdivider-hi6220.c | 2 +- + drivers/clk/hisilicon/clkgate-separated.c | 2 +- + drivers/clk/imx/clk-busy.c | 4 +- + drivers/clk/imx/clk-cpu.c | 2 +- + drivers/clk/imx/clk-divider-gate.c | 2 +- + drivers/clk/imx/clk-fixup-div.c | 2 +- + drivers/clk/imx/clk-fixup-mux.c | 2 +- + drivers/clk/imx/clk-frac-pll.c | 2 +- + drivers/clk/imx/clk-fracn-gppll.c | 2 +- + drivers/clk/imx/clk-gate-93.c | 2 +- + drivers/clk/imx/clk-gate-exclusive.c | 2 +- + drivers/clk/imx/clk-gate2.c | 2 +- + drivers/clk/imx/clk-lpcg-scu.c | 2 +- + drivers/clk/imx/clk-pfd.c | 2 +- + drivers/clk/imx/clk-pfdv2.c | 2 +- + drivers/clk/imx/clk-pll14xx.c | 2 +- + drivers/clk/imx/clk-pllv1.c | 2 +- + drivers/clk/imx/clk-pllv2.c | 2 +- + drivers/clk/imx/clk-pllv3.c | 2 +- + drivers/clk/imx/clk-pllv4.c | 2 +- + drivers/clk/imx/clk-scu.c | 4 +- + drivers/clk/imx/clk-sscg-pll.c | 2 +- + drivers/clk/ingenic/cgu.c | 2 +- + drivers/clk/keystone/gate.c | 2 +- + drivers/clk/keystone/pll.c | 2 +- + drivers/clk/keystone/syscon-clk.c | 2 +- + drivers/clk/kunit_clk_assigned_rates.h | 4 +- + drivers/clk/kunit_clk_parse_clkspec.dtso | 31 ++ + drivers/clk/mediatek/Kconfig | 1 + + drivers/clk/mediatek/clk-cpumux.c | 2 +- + drivers/clk/mediatek/reset.h | 6 +- + drivers/clk/mmp/Kconfig | 5 + + drivers/clk/mmp/clk-apbc.c | 2 +- + drivers/clk/mmp/clk-apmu.c | 2 +- + drivers/clk/mmp/clk-frac.c | 2 +- + drivers/clk/mmp/clk-gate.c | 2 +- + drivers/clk/mmp/clk-mix.c | 2 +- + drivers/clk/mmp/clk-pll.c | 2 +- + drivers/clk/mstar/clk-msc313-mpll.c | 2 +- + drivers/clk/mvebu/ap-cpu-clk.c | 6 +- + drivers/clk/mvebu/clk-corediv.c | 2 +- + drivers/clk/mvebu/clk-cpu.c | 2 +- + drivers/clk/mxs/clk-div.c | 2 +- + drivers/clk/mxs/clk-frac.c | 2 +- + drivers/clk/mxs/clk-pll.c | 2 +- + drivers/clk/mxs/clk-ref.c | 2 +- + drivers/clk/nxp/clk-lpc18xx-creg.c | 2 +- + drivers/clk/pistachio/clk-pll.c | 2 +- + drivers/clk/pistachio/clk.c | 15 +- + drivers/clk/pistachio/clk.h | 1 + + drivers/clk/renesas/Kconfig | 2 + + drivers/clk/renesas/r9a06g032-clocks.c | 2 +- + drivers/clk/renesas/r9a08g046-cpg.c | 92 +++- + drivers/clk/renesas/r9a09g077-cpg.c | 142 ++++++ + drivers/clk/renesas/rcar-gen3-cpg.c | 15 +- + drivers/clk/renesas/rcar-gen3-cpg.h | 3 +- + drivers/clk/renesas/renesas-cpg-mssr.c | 3 + + drivers/clk/renesas/renesas-cpg-mssr.h | 1 + + drivers/clk/renesas/rzg2l-cpg.c | 554 ++++++++++++++++++++- + drivers/clk/renesas/rzg2l-cpg.h | 51 +- + drivers/clk/renesas/rzv2h-cpg.c | 164 +++--- + drivers/clk/rockchip/clk-cpu.c | 2 +- + drivers/clk/rockchip/clk-ddr.c | 2 +- + drivers/clk/rockchip/clk-gate-grf.c | 2 +- + drivers/clk/rockchip/clk-inverter.c | 2 +- + drivers/clk/rockchip/clk-mmc-phase.c | 2 +- + drivers/clk/rockchip/clk-muxgrf.c | 2 +- + drivers/clk/rockchip/clk-pll.c | 2 +- + drivers/clk/rockchip/clk.c | 2 +- + drivers/clk/samsung/clk-cpu.c | 2 +- + drivers/clk/samsung/clk-exynos-clkout.c | 8 +- + drivers/clk/samsung/clk-pll.c | 2 +- + drivers/clk/socfpga/clk-gate-a10.c | 2 +- + drivers/clk/socfpga/clk-gate-s10.c | 6 +- + drivers/clk/socfpga/clk-gate.c | 2 +- + drivers/clk/socfpga/clk-periph-a10.c | 2 +- + drivers/clk/socfpga/clk-periph-s10.c | 8 +- + drivers/clk/socfpga/clk-periph.c | 2 +- + drivers/clk/socfpga/clk-pll-a10.c | 2 +- + drivers/clk/socfpga/clk-pll-s10.c | 8 +- + drivers/clk/socfpga/clk-pll.c | 2 +- + drivers/clk/spear/clk-aux-synth.c | 2 +- + drivers/clk/spear/clk-frac-synth.c | 2 +- + drivers/clk/spear/clk-gpt-synth.c | 2 +- + drivers/clk/spear/clk-vco-pll.c | 2 +- + drivers/clk/st/clk-flexgen.c | 2 +- + drivers/clk/st/clkgen-fsyn.c | 4 +- + drivers/clk/st/clkgen-pll.c | 2 +- + drivers/clk/starfive/clk-starfive-jh7110-isp.c | 2 +- + drivers/clk/starfive/clk-starfive-jh7110-sys.c | 3 + + drivers/clk/starfive/clk-starfive-jh7110-vout.c | 6 +- + drivers/clk/stm32/clk-stm32mp1.c | 4 +- + drivers/clk/sunxi/clk-sun4i-tcon-ch1.c | 2 +- + drivers/clk/tegra/clk-audio-sync.c | 2 +- + drivers/clk/tegra/clk-divider.c | 2 +- + drivers/clk/tegra/clk-periph-fixed.c | 2 +- + drivers/clk/tegra/clk-periph-gate.c | 2 +- + drivers/clk/tegra/clk-periph.c | 2 +- + drivers/clk/tegra/clk-pll-out.c | 2 +- + drivers/clk/tegra/clk-pll.c | 2 +- + drivers/clk/tegra/clk-sdmmc-mux.c | 2 +- + drivers/clk/tegra/clk-super.c | 4 +- + drivers/clk/tegra/clk-tegra-super-cclk.c | 2 +- + drivers/clk/tegra/clk-tegra124-emc.c | 2 +- + drivers/clk/tegra/clk-tegra20-emc.c | 2 +- + drivers/clk/tegra/clk-tegra210-emc.c | 2 +- + drivers/clk/tegra/clk.h | 2 +- + drivers/clk/ti/adpll.c | 1 + + drivers/clk/ti/clockdomain.c | 2 +- + drivers/clk/uniphier/clk-uniphier-cpugear.c | 2 +- + drivers/clk/uniphier/clk-uniphier-fixed-factor.c | 2 +- + drivers/clk/uniphier/clk-uniphier-fixed-rate.c | 2 +- + drivers/clk/uniphier/clk-uniphier-gate.c | 2 +- + drivers/clk/uniphier/clk-uniphier-mux.c | 2 +- + drivers/clk/ux500/clk-prcc.c | 2 +- + drivers/clk/ux500/clk-prcmu.c | 4 +- + drivers/clk/ux500/clk-sysctrl.c | 2 +- + drivers/clk/versatile/clk-icst.c | 6 +- + drivers/clk/versatile/clk-sp810.c | 2 +- + drivers/clk/versatile/clk-vexpress-osc.c | 2 +- + drivers/clk/x86/clk-pmc-atom.c | 2 +- + drivers/clk/xilinx/clk-xlnx-clock-wizard.c | 18 +- + drivers/clk/xilinx/xlnx_vcu.c | 2 +- + drivers/clk/zynqmp/clk-gate-zynqmp.c | 2 +- + drivers/clk/zynqmp/clk-mux-zynqmp.c | 2 +- + drivers/clk/zynqmp/divider.c | 2 +- + drivers/clk/zynqmp/pll.c | 2 +- + drivers/reset/Kconfig | 10 + + drivers/reset/Makefile | 1 + + drivers/reset/reset-dr1v90.c | 140 ++++++ + include/dt-bindings/clock/anlogic,dr1v90-cru.h | 46 ++ + include/dt-bindings/reset/anlogic,dr1v90-cru.h | 41 ++ + include/dt-bindings/soc/airoha,scu-ssr.h | 11 + + include/kunit/clk.h | 2 + + include/linux/clk-provider.h | 25 +- + include/linux/clk/tegra.h | 2 +- + include/linux/clk/ti.h | 2 +- + rust/kernel/clk.rs | 15 + + 227 files changed, 3273 insertions(+), 771 deletions(-) + create mode 100644 Documentation/devicetree/bindings/clock/anlogic,dr1v90-cru.yaml + delete mode 100644 Documentation/devicetree/bindings/clock/clk-palmas-clk32kg-clocks.txt + create mode 100644 Documentation/devicetree/bindings/clock/clock-nexus-node.yaml + create mode 100644 Documentation/devicetree/bindings/clock/ti,palmas-clk32kg.yaml + delete mode 100644 Documentation/devicetree/bindings/clock/ti/adpll.txt + delete mode 100644 Documentation/devicetree/bindings/clock/ti/davinci/pll.txt + create mode 100644 Documentation/devicetree/bindings/clock/ti/davinci/ti,da850-pll.yaml + delete mode 100644 Documentation/devicetree/bindings/clock/ti/dra7-atl.txt + delete mode 100644 Documentation/devicetree/bindings/clock/ti/fapll.txt + create mode 100644 Documentation/devicetree/bindings/clock/ti/ti,dm814-adpll-clock.yaml + create mode 100644 Documentation/devicetree/bindings/clock/ti/ti,dm816-fapll-clock.yaml + create mode 100644 Documentation/devicetree/bindings/clock/ti/ti,dra7-atl-clock.yaml + create mode 100644 Documentation/devicetree/bindings/clock/ti/ti,dra7-atl.yaml + create mode 100644 drivers/clk/anlogic/Kconfig + create mode 100644 drivers/clk/anlogic/Makefile + create mode 100644 drivers/clk/anlogic/cru-dr1v90.c + create mode 100644 drivers/clk/anlogic/cru_dr1.c + create mode 100644 drivers/clk/anlogic/cru_dr1.h + create mode 100644 drivers/clk/davinci/Kconfig + create mode 100644 drivers/clk/kunit_clk_parse_clkspec.dtso + create mode 100644 drivers/reset/reset-dr1v90.c + create mode 100644 include/dt-bindings/clock/anlogic,dr1v90-cru.h + create mode 100644 include/dt-bindings/reset/anlogic,dr1v90-cru.h + create mode 100644 include/dt-bindings/soc/airoha,scu-ssr.h +Merging clk-imx/for-next (39ec460b56b26 clk: imx95-blk-ctl: Fix REFCLK rise-fall mismatch on i.MX95) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/abelvesa/linux.git clk-imx/for-next +Already up to date. +Merging clk-renesas/renesas-clk (21cbd7ec29930 clk: renesas: r8a78000: Add SGASYNCD8_PERW_BUS) +$ git merge -m Merge branch 'renesas-clk' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-drivers.git clk-renesas/renesas-clk +Auto-merging drivers/clk/renesas/renesas-cpg-mssr.c +Merge made by the 'ort' strategy. + drivers/clk/renesas/Kconfig | 5 + + drivers/clk/renesas/Makefile | 1 + + drivers/clk/renesas/r8a774a3-cpg-mssr.c | 168 ++++++++++++++++++++++++++++++++ + drivers/clk/renesas/r8a779a0-cpg-mssr.c | 6 +- + drivers/clk/renesas/r8a779f0-cpg-mssr.c | 4 +- + drivers/clk/renesas/r8a779g0-cpg-mssr.c | 4 +- + drivers/clk/renesas/r8a779h0-cpg-mssr.c | 8 +- + drivers/clk/renesas/r8a78000-cpg.c | 3 + + drivers/clk/renesas/r9a07g043-cpg.c | 12 +-- + drivers/clk/renesas/r9a07g044-cpg.c | 12 +-- + drivers/clk/renesas/r9a08g045-cpg.c | 6 +- + drivers/clk/renesas/r9a08g046-cpg.c | 25 ++++- + drivers/clk/renesas/rcar-gen4-cpg.c | 25 +++-- + drivers/clk/renesas/rcar-gen4-cpg.h | 14 ++- + drivers/clk/renesas/renesas-cpg-mssr.c | 6 ++ + drivers/clk/renesas/renesas-cpg-mssr.h | 1 + + 16 files changed, 257 insertions(+), 43 deletions(-) + create mode 100644 drivers/clk/renesas/r8a774a3-cpg-mssr.c +Merging thead-clk/thead-clk-for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'thead-clk-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fustini/linux.git thead-clk/thead-clk-for-next +Already up to date. +Merging tenstorrent-clk/tenstorrent-clk-for-next (c64b54ddb692c MAINTAINERS: Update RISC-V Tenstorrent SoC entry) +$ git merge -m Merge branch 'tenstorrent-clk-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tenstorrent/linux.git tenstorrent-clk/tenstorrent-clk-for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 5 +++-- + drivers/clk/tenstorrent/atlantis-prcm.c | 35 +++++++++++++-------------------- + 2 files changed, 17 insertions(+), 23 deletions(-) +Merging csky/linux-next (abb81e5ce7d99 csky: Fix a4/a5 restoration in syscall trace path) +$ git merge -m Merge branch 'linux-next' of https://github.com/c-sky/csky-linux.git csky/linux-next +Already up to date. +Merging loongarch/loongarch-next (a2628ce4ddb68 perf build: Add clang and rust target flags for LoongArch) +$ git merge -m Merge branch 'loongarch-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chenhuacai/linux-loongson.git loongarch/loongarch-next +Already up to date. +Merging m68k/for-next (af32c3cb72529 m68k: defconfig: Enable ATARI_SVETHLANA and ETHOC) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/linux-m68k.git m68k/for-next +Merge made by the 'ort' strategy. + arch/m68k/Kconfig.devices | 12 ++++++ + arch/m68k/apollo/dn_ints.c | 6 +++ + arch/m68k/atari/config.c | 75 ++++++++++++++++++++++++++++++++++++ + arch/m68k/configs/amiga_defconfig | 4 ++ + arch/m68k/configs/apollo_defconfig | 3 ++ + arch/m68k/configs/atari_defconfig | 5 +++ + arch/m68k/configs/bvme6000_defconfig | 3 ++ + arch/m68k/configs/hp300_defconfig | 3 ++ + arch/m68k/configs/mac_defconfig | 3 ++ + arch/m68k/configs/multi_defconfig | 6 +++ + arch/m68k/configs/mvme147_defconfig | 3 ++ + arch/m68k/configs/mvme16x_defconfig | 3 ++ + arch/m68k/configs/q40_defconfig | 3 ++ + arch/m68k/configs/sun3_defconfig | 3 ++ + arch/m68k/configs/sun3x_defconfig | 3 ++ + arch/m68k/include/asm/atariints.h | 2 +- + arch/m68k/include/asm/irq.h | 6 +-- + arch/m68k/include/asm/serial.h | 21 +++------- + drivers/zorro/zorro.c | 5 +-- + 19 files changed, 145 insertions(+), 24 deletions(-) +Merging m68knommu/for-next (29036c5910df0 m68knommu: fix compile breakage for 5407 cleopatra board) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gerg/m68knommu.git m68knommu/for-next +Merge made by the 'ort' strategy. + arch/m68k/68000/entry.S | 3 +- + arch/m68k/coldfire/entry.S | 3 +- + arch/m68k/coldfire/m5407.c | 5 ++++ + arch/m68k/coldfire/reset.c | 2 +- + arch/m68k/configs/stmark2_defconfig | 2 ++ + arch/m68k/include/asm/m53xxacr.h | 2 +- + arch/m68k/include/asm/quicc_simple.h | 53 ------------------------------------ + 7 files changed, 13 insertions(+), 57 deletions(-) + delete mode 100644 arch/m68k/include/asm/quicc_simple.h +Merging microblaze/next (6b125c73aaa0d microblaze: Enable xilinx dmas and axi emac drivers) +$ git merge -m Merge branch 'next' of git://git.monstr.eu/linux-2.6-microblaze.git microblaze/next +Merge made by the 'ort' strategy. + arch/microblaze/Kconfig | 9 + + arch/microblaze/Makefile | 8 +- + arch/microblaze/configs/mmu_defconfig | 2 + + arch/microblaze/include/asm/entry.h | 13 + + arch/microblaze/include/asm/processor.h | 2 +- + arch/microblaze/include/asm/uaccess.h | 3 +- + arch/microblaze/kernel/cpu/cpuinfo-pvr-full.c | 2 +- + arch/microblaze/kernel/cpu/cpuinfo-static.c | 2 +- + arch/microblaze/kernel/cpu/cpuinfo.c | 1 + + arch/microblaze/kernel/entry.S | 348 ++++++++++++----------- + arch/microblaze/kernel/hw_exception_handler.S | 5 + + arch/microblaze/kernel/process.c | 5 +- + arch/microblaze/kernel/ptrace.c | 3 + + arch/microblaze/kernel/reset.c | 22 ++ + arch/microblaze/kernel/signal.c | 26 ++ + arch/microblaze/kernel/syscalls/syscall.tbl | 2 +- + tools/testing/kunit/qemu_configs/microblazebe.py | 24 ++ + tools/testing/kunit/qemu_configs/microblazeel.py | 27 ++ + 18 files changed, 328 insertions(+), 176 deletions(-) + create mode 100644 tools/testing/kunit/qemu_configs/microblazebe.py + create mode 100644 tools/testing/kunit/qemu_configs/microblazeel.py +Merging mips/mips-next (f552e14f77bfc mips: dts: pic32: pic32mzda: Remove unused address-cells, size-cells property) +$ git merge -m Merge branch 'mips-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mips/linux.git mips/mips-next +Merge made by the 'ort' strategy. + .../ABI/testing/sysfs-firmware-lefi-boardinfo | 21 +++--- + arch/mips/Makefile | 32 +++++---- + arch/mips/alchemy/common/clock.c | 8 +-- + arch/mips/boot/dts/pic32/pic32mzda.dtsi | 2 - + arch/mips/cavium-octeon/dma-octeon.c | 2 +- + arch/mips/cavium-octeon/executive/cvmx-helper.c | 20 ++++-- + arch/mips/cavium-octeon/executive/cvmx-l2c.c | 31 ++++---- + arch/mips/cavium-octeon/flash_setup.c | 2 +- + arch/mips/cavium-octeon/octeon-memcpy.S | 2 +- + arch/mips/cavium-octeon/setup.c | 22 +++++- + arch/mips/configs/decstation_64_defconfig | 44 +++++------- + arch/mips/dec/setup.c | 4 +- + arch/mips/generic/board-jaguar2.its.S | 2 - + arch/mips/generic/board-serval.its.S | 1 - + arch/mips/include/asm/amon.h | 12 ---- + arch/mips/include/asm/cmp.h | 10 --- + arch/mips/include/asm/dec/reset.h | 1 + + arch/mips/include/asm/mach-loongson64/boot_param.h | 2 +- + arch/mips/include/asm/mips-boards/sead3-addr.h | 83 ---------------------- + arch/mips/lib/memcpy.S | 2 +- + arch/mips/mm/mmap.c | 6 +- + arch/mips/mm/tlb-r4k.c | 21 ++++-- + arch/mips/n64/init.c | 2 +- + arch/mips/rb532/gpio.c | 7 +- + 24 files changed, 133 insertions(+), 206 deletions(-) + delete mode 100644 arch/mips/include/asm/amon.h + delete mode 100644 arch/mips/include/asm/cmp.h + delete mode 100644 arch/mips/include/asm/mips-boards/sead3-addr.h +Merging openrisc/for-next (6620f5e8c11c4 openrisc: drop unneeded semicolon) +$ git merge -m Merge branch 'for-next' of https://github.com/openrisc/linux.git openrisc/for-next +Already up to date. +Merging parisc-hd/for-next (a5f6df25261b3 parisc: fix typos in comments in parport.c) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/deller/parisc-linux.git parisc-hd/for-next +Merge made by the 'ort' strategy. + arch/parisc/include/asm/pdcpat.h | 4 ++-- + arch/parisc/include/asm/special_insns.h | 2 +- + arch/parisc/include/uapi/asm/mman.h | 2 +- + arch/parisc/lib/memcpy.c | 2 +- + drivers/parisc/ccio-dma.c | 2 +- + drivers/parisc/dino.c | 2 +- + drivers/parisc/pdc_stable.c | 2 +- + drivers/parisc/sba_iommu.c | 2 +- + drivers/parport/parport_gsc.c | 2 +- + 9 files changed, 10 insertions(+), 10 deletions(-) +Merging powerpc/next (12d238d584ba4 powerpc/sysfs: Remove redundant wait time clamps) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git powerpc/next +Merge made by the 'ort' strategy. + Documentation/ABI/testing/ppc-memtrace | 2 +- + arch/powerpc/boot/dts/ac14xx.dts | 13 --- + arch/powerpc/boot/dts/mpc5121.dtsi | 6 +- + arch/powerpc/boot/dts/pdm360ng.dts | 8 +- + arch/powerpc/include/asm/hardirq.h | 31 +++--- + arch/powerpc/kernel/dbell.c | 2 +- + arch/powerpc/kernel/irq.c | 131 +++++++++++++------------- + arch/powerpc/kernel/sysfs.c | 4 +- + arch/powerpc/kernel/time.c | 6 +- + arch/powerpc/kernel/trace/ftrace_entry.S | 6 ++ + arch/powerpc/kernel/traps.c | 11 +-- + arch/powerpc/kernel/watchdog.c | 2 +- + arch/powerpc/kvm/guest-state-buffer.c | 2 +- + arch/powerpc/lib/sstep.c | 111 +++++++++++----------- + arch/powerpc/platforms/512x/Kconfig | 3 +- + arch/powerpc/platforms/512x/Makefile | 1 - + arch/powerpc/platforms/512x/mpc512x_generic.c | 1 + + arch/powerpc/platforms/512x/pdm360ng.c | 126 ------------------------- + arch/powerpc/platforms/ps3/interrupt.c | 22 ++--- + 19 files changed, 182 insertions(+), 306 deletions(-) + delete mode 100644 arch/powerpc/platforms/512x/pdm360ng.c +Merging risc-v/for-next (a06776565b9e7 riscv: process: Use str_supported_unsupported() in compat_mode_detect()) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/riscv/linux.git risc-v/for-next +Merge made by the 'ort' strategy. + arch/riscv/Kconfig.errata | 1 + + arch/riscv/errata/mips/Makefile | 6 ++++++ + arch/riscv/errata/mips/errata.c | 38 +++++++++++++++++++++++--------------- + arch/riscv/kernel/process.c | 3 ++- + 4 files changed, 32 insertions(+), 16 deletions(-) +Merging riscv-dt/riscv-dt-for-next (f55a54f07a367 MAINTAINERS: update RISC-V Microchip maintainers) +$ git merge -m Merge branch 'riscv-dt-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git riscv-dt/riscv-dt-for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 3 ++- + 1 file changed, 2 insertions(+), 1 deletion(-) +Merging riscv-soc/riscv-soc-for-next (00bdd4d4daaae Merge branch 'riscv-firmware-for-next' into riscv-soc-for-next) +$ git merge -m Merge branch 'riscv-soc-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/conor/linux.git riscv-soc/riscv-soc-for-next +Merge made by the 'ort' strategy. + drivers/firmware/microchip/mpfs-auto-update.c | 4 ++-- + 1 file changed, 2 insertions(+), 2 deletions(-) +Merging s390/for-next (c57ae907d4f9a Merge branch 'features' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/s390/linux.git s390/for-next +Auto-merging arch/s390/kernel/traps.c +Merge made by the 'ort' strategy. + arch/s390/boot/ipl_parm.c | 13 +- + arch/s390/include/asm/ebcdic.h | 6 + + arch/s390/include/asm/entry-percpu.h | 71 ++----- + arch/s390/include/asm/idals.h | 9 +- + arch/s390/include/asm/lowcore.h | 5 +- + arch/s390/include/asm/percpu.h | 354 ++++++++++++++++++++++------------- + arch/s390/kernel/ebcdic.c | 40 +++- + arch/s390/kernel/hiperdispatch.c | 2 +- + arch/s390/kernel/irq.c | 10 +- + arch/s390/kernel/nmi.c | 4 +- + arch/s390/kernel/traps.c | 4 +- + arch/s390/lib/spinlock.c | 4 + + drivers/s390/char/keyboard.c | 6 +- + drivers/s390/char/sclp_vt220.c | 87 +++++++-- + drivers/s390/cio/chsc.c | 20 +- + drivers/s390/cio/chsc_sch.c | 300 +++++++++++------------------ + drivers/s390/cio/cmf.c | 11 +- + drivers/s390/cio/qdio.h | 2 +- + drivers/s390/cio/qdio_main.c | 23 +-- + drivers/s390/cio/qdio_setup.c | 14 +- + drivers/s390/cio/qdio_thinint.c | 2 +- + drivers/s390/cio/scm.c | 7 +- + 22 files changed, 538 insertions(+), 456 deletions(-) +Merging sh/for-next (be5a19d95030f sh: ecovec24: Use static device properties to describe the touchscreen) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/glaubitz/sh-linux.git sh/for-next +Merge made by the 'ort' strategy. + arch/sh/boards/mach-ecovec24/setup.c | 38 ++---- + arch/sh/boards/mach-x3proto/gpio.c | 15 ++- + arch/sh/boards/mach-x3proto/setup.c | 165 +++++++++++++-------------- + arch/sh/include/mach-x3proto/mach/hardware.h | 2 + + 4 files changed, 105 insertions(+), 115 deletions(-) +Merging sparc/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/alarsson/linux-sparc.git sparc/for-next +Already up to date. +Merging uml/next (2f88f5689de1a um: fix shutdown __inittext access) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git uml/next +Auto-merging arch/um/Kconfig +CONFLICT (content): Merge conflict in arch/um/Kconfig +Auto-merging lib/Kconfig.debug +Resolved 'arch/um/Kconfig' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 2e7c2602f0ee8] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/uml/linux.git +$ git diff -M --stat --summary HEAD^.. + Documentation/virt/index.rst | 1 + + Documentation/virt/uml/nommu-uml.rst | 54 ++++++++ + arch/um/Kconfig | 38 +++-- + arch/um/Makefile | 7 +- + arch/um/drivers/Kconfig | 3 + + arch/um/drivers/cow_user.c | 34 ++++- + arch/um/drivers/mconsole_kern.c | 14 +- + arch/um/drivers/vector_kern.c | 37 +++-- + arch/um/drivers/vector_user.c | 3 + + arch/um/drivers/virtio_pcidev.c | 19 ++- + arch/um/drivers/virtio_uml.c | 26 ++-- + arch/um/include/asm/common.lds.S | 2 + + arch/um/include/asm/futex.h | 53 +++++++ + arch/um/include/asm/mmu.h | 8 ++ + arch/um/include/asm/mmu_context.h | 2 + + arch/um/include/asm/perf_event.h | 7 + + arch/um/include/asm/pgtable.h | 25 ++-- + arch/um/include/asm/ptrace-generic.h | 6 + + arch/um/include/asm/tlbflush.h | 20 +++ + arch/um/include/asm/uaccess.h | 6 +- + arch/um/include/shared/as-layout.h | 3 +- + arch/um/include/shared/kern_util.h | 3 +- + arch/um/include/shared/longjmp.h | 3 +- + arch/um/include/shared/os.h | 6 +- + arch/um/include/shared/skas/stub-data.h | 9 ++ + arch/um/kernel/Makefile | 6 +- + arch/um/kernel/dyn.lds.S | 9 +- + arch/um/kernel/mem-pgtable.c | 55 ++++++++ + arch/um/kernel/mem.c | 47 ++----- + arch/um/kernel/physmem.c | 7 + + arch/um/kernel/process.c | 23 +++- + arch/um/kernel/skas/Makefile | 16 ++- + arch/um/kernel/skas/mmu.c | 25 ++++ + arch/um/kernel/skas/nommu.c | 91 ++++++++++++ + arch/um/kernel/skas/process.c | 32 +---- + arch/um/kernel/skas/syscall.c | 5 +- + arch/um/kernel/skas/uaccess.c | 134 +++++++----------- + arch/um/kernel/smp.c | 3 +- + arch/um/kernel/sysrq.c | 2 +- + arch/um/kernel/time.c | 3 +- + arch/um/kernel/tlb.c | 7 - + arch/um/kernel/trap-mmu.c | 237 ++++++++++++++++++++++++++++++++ + arch/um/kernel/trap-nommu.c | 13 ++ + arch/um/kernel/trap.c | 219 ----------------------------- + arch/um/kernel/um_arch.c | 11 +- + arch/um/kernel/uml.lds.S | 11 +- + arch/um/os-Linux/main.c | 17 ++- + arch/um/os-Linux/signal.c | 2 +- + arch/um/os-Linux/skas/process.c | 11 +- + arch/x86/um/Makefile | 3 + + arch/x86/um/asm/elf.h | 8 +- + arch/x86/um/asm/required-features.h | 9 -- + arch/x86/um/vdso/vma.c | 13 +- + fs/Kconfig.binfmt | 2 +- + lib/Kconfig.debug | 10 +- + lib/kunit/Kconfig | 2 +- + 56 files changed, 906 insertions(+), 516 deletions(-) + create mode 100644 Documentation/virt/uml/nommu-uml.rst + create mode 100644 arch/um/include/asm/perf_event.h + create mode 100644 arch/um/kernel/mem-pgtable.c + create mode 100644 arch/um/kernel/skas/nommu.c + create mode 100644 arch/um/kernel/trap-mmu.c + create mode 100644 arch/um/kernel/trap-nommu.c + delete mode 100644 arch/x86/um/asm/required-features.h +Merging xtensa/xtensa-for-next (28722ed2527aa xtensa: time: Fix clk reference leak in calibrate_ccount()) +$ git merge -m Merge branch 'xtensa-for-next' of https://github.com/jcmvbkbc/linux-xtensa.git xtensa/xtensa-for-next +Merge made by the 'ort' strategy. + arch/xtensa/include/asm/atomic.h | 13 ------------- + arch/xtensa/include/uapi/asm/mman.h | 2 +- + arch/xtensa/kernel/entry.S | 2 +- + arch/xtensa/kernel/time.c | 1 + + arch/xtensa/kernel/vectors.S | 2 +- + arch/xtensa/mm/init.c | 2 +- + 6 files changed, 5 insertions(+), 17 deletions(-) +Merging fs-next (bf234c28d9e24 Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/viro/vfs.git) +$ git merge -m Merge branch 'fs-next' of linux-next fs-next +Auto-merging CREDITS +Auto-merging Documentation/filesystems/vfs.rst +Auto-merging MAINTAINERS +Auto-merging arch/powerpc/Kconfig +Auto-merging drivers/android/binder/rust_binderfs.c +Auto-merging drivers/android/binderfs.c +Auto-merging fs/Kconfig +Auto-merging fs/buffer.c +Auto-merging fs/ceph/addr.c +Auto-merging fs/coredump.c +CONFLICT (content): Merge conflict in fs/coredump.c +Auto-merging fs/crypto/crypto.c +Auto-merging fs/f2fs/compress.c +Auto-merging fs/f2fs/data.c +Auto-merging fs/f2fs/f2fs.h +CONFLICT (content): Merge conflict in fs/f2fs/f2fs.h +Auto-merging fs/f2fs/segment.c +Auto-merging fs/fat/fat.h +Auto-merging fs/fat/file.c +Auto-merging fs/fat/misc.c +Auto-merging fs/fuse/dax.c +CONFLICT (content): Merge conflict in fs/fuse/dax.c +Auto-merging fs/hugetlbfs/inode.c +Auto-merging fs/iomap/buffered-io.c +Auto-merging fs/nfs/write.c +Auto-merging fs/ocfs2/journal.c +Auto-merging fs/ocfs2/quota_local.c +Auto-merging fs/ocfs2/xattr.c +Auto-merging fs/proc/base.c +Auto-merging fs/proc/internal.h +Auto-merging fs/proc/vmcore.c +Auto-merging fs/ubifs/file.c +Auto-merging fs/xfs/libxfs/xfs_btree.c +CONFLICT (content): Merge conflict in fs/xfs/libxfs/xfs_btree.c +Auto-merging include/linux/buffer_head.h +Auto-merging include/linux/sched.h +Auto-merging init/Kconfig +Auto-merging ipc/mqueue.c +Auto-merging kernel/Makefile +Auto-merging kernel/fork.c +Auto-merging mm/secretmem.c +Auto-merging mm/shmem.c +Auto-merging mm/userfaultfd.c +Auto-merging security/selinux/selinuxfs.c +Resolved 'fs/coredump.c' using previous resolution. +Resolved 'fs/f2fs/f2fs.h' using previous resolution. +Resolved 'fs/fuse/dax.c' using previous resolution. +Resolved 'fs/xfs/libxfs/xfs_btree.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master c0c20ac78811b] Merge branch 'fs-next' of linux-next +$ git diff -M --stat --summary HEAD^.. + CREDITS | 2 +- + Documentation/ABI/testing/sysfs-fs-f2fs | 7 + + Documentation/admin-guide/binfmt-misc.rst | 3 + + Documentation/admin-guide/nfs/pnfs-scsi-server.rst | 2 +- + Documentation/filesystems/befs.rst | 4 +- + Documentation/filesystems/bfs.rst | 60 - + Documentation/filesystems/exfat.rst | 117 ++ + Documentation/filesystems/index.rst | 2 +- + Documentation/filesystems/locking.rst | 24 +- + Documentation/filesystems/netfs_library.rst | 7 +- + Documentation/filesystems/ntfs3.rst | 23 + + Documentation/filesystems/porting.rst | 25 +- + Documentation/filesystems/proc.rst | 4 + + Documentation/filesystems/sharedsubtree.rst | 20 +- + Documentation/filesystems/smb/ksmbd.rst | 14 +- + Documentation/filesystems/squashfs.rst | 6 +- + Documentation/filesystems/vfs.rst | 30 +- + .../filesystems/xfs/xfs-online-fsck-design.rst | 13 +- + MAINTAINERS | 20 +- + arch/mips/configs/malta_defconfig | 1 - + arch/mips/configs/malta_kvm_defconfig | 1 - + arch/mips/configs/maltaup_xpa_defconfig | 1 - + arch/mips/configs/rm200_defconfig | 1 - + arch/powerpc/Kconfig | 1 - + arch/powerpc/configs/fsl-emb-nonhw.config | 1 - + arch/powerpc/configs/ppc6xx_defconfig | 1 - + arch/powerpc/include/asm/elf.h | 6 - + arch/powerpc/include/asm/spu.h | 3 - + arch/powerpc/platforms/cell/Kconfig | 1 - + arch/powerpc/platforms/cell/spu_syscalls.c | 20 - + arch/powerpc/platforms/cell/spufs/Makefile | 1 - + arch/powerpc/platforms/cell/spufs/coredump.c | 183 -- + arch/powerpc/platforms/cell/spufs/file.c | 114 -- + arch/powerpc/platforms/cell/spufs/inode.c | 14 +- + arch/powerpc/platforms/cell/spufs/spufs.h | 12 - + arch/powerpc/platforms/cell/spufs/syscalls.c | 4 - + block/bio-integrity-fs.c | 15 +- + block/bio-integrity.c | 1 + + block/bio.c | 219 +-- + block/blk-map.c | 2 +- + block/blk-settings.c | 6 + + block/fops.c | 3 +- + block/partitions/efi.h | 8 +- + drivers/acpi/acpi_configfs.c | 2 +- + drivers/android/binder/rust_binderfs.c | 2 +- + drivers/android/binderfs.c | 2 +- + drivers/base/devtmpfs.c | 108 +- + drivers/gpio/gpiolib-cdev.c | 18 +- + drivers/gpu/drm/msm/msm_perfcntr.c | 4 +- + drivers/gpu/drm/xe/xe_configfs.c | 4 +- + drivers/media/mc/mc-request.c | 8 +- + drivers/misc/ntsync.c | 6 +- + drivers/virt/coco/guest/report.c | 6 +- + fs/9p/acl.c | 4 +- + fs/9p/acl.h | 4 +- + fs/9p/v9fs.h | 2 +- + fs/9p/v9fs_vfs.h | 2 +- + fs/9p/vfs_addr.c | 1 - + fs/9p/vfs_inode.c | 14 +- + fs/9p/vfs_inode_dotl.c | 14 +- + fs/9p/xattr.c | 2 +- + fs/Kconfig | 9 +- + fs/Makefile | 1 - + fs/adfs/adfs.h | 2 +- + fs/adfs/dir.c | 2 +- + fs/adfs/inode.c | 2 +- + fs/affs/affs.h | 10 +- + fs/affs/inode.c | 2 +- + fs/affs/namei.c | 8 +- + fs/afs/dir.c | 16 +- + fs/afs/file.c | 8 +- + fs/afs/inode.c | 4 +- + fs/afs/internal.h | 8 +- + fs/afs/security.c | 2 +- + fs/afs/xattr.c | 4 +- + fs/aio.c | 11 +- + fs/anon_inodes.c | 4 +- + fs/attr.c | 16 +- + fs/autofs/root.c | 12 +- + fs/backing-file.c | 2 +- + fs/bad_inode.c | 20 +- + fs/bfs/Kconfig | 21 - + fs/bfs/Makefile | 8 - + fs/bfs/bfs.h | 69 - + fs/bfs/dir.c | 4 +- + fs/bfs/file.c | 203 -- + fs/bfs/inode.c | 538 ------ + fs/binfmt_elf.c | 18 +- + fs/binfmt_elf_fdpic.c | 14 +- + fs/binfmt_misc.c | 12 +- + fs/bpf_fs_kfuncs.c | 4 +- + fs/btrfs/Kconfig | 1 + + fs/btrfs/acl.c | 2 +- + fs/btrfs/acl.h | 2 +- + fs/btrfs/bio.c | 163 +- + fs/btrfs/bio.h | 9 +- + fs/btrfs/block-group.c | 498 +---- + fs/btrfs/block-group.h | 16 +- + fs/btrfs/btrfs_inode.h | 24 +- + fs/btrfs/compression.c | 61 +- + fs/btrfs/ctree.c | 4 +- + fs/btrfs/delalloc-space.c | 13 +- + fs/btrfs/delayed-inode.c | 115 +- + fs/btrfs/delayed-inode.h | 22 +- + fs/btrfs/dev-replace.c | 10 +- + fs/btrfs/dir-item.c | 41 +- + fs/btrfs/dir-item.h | 5 +- + fs/btrfs/disk-io.c | 126 +- + fs/btrfs/extent-io-tree.c | 2 +- + fs/btrfs/extent-tree.c | 10 + + fs/btrfs/extent_io.c | 247 ++- + fs/btrfs/extent_map.c | 8 + + fs/btrfs/file-item.c | 46 +- + fs/btrfs/free-space-cache.c | 1344 +------------ + fs/btrfs/free-space-cache.h | 29 +- + fs/btrfs/fs.c | 2 +- + fs/btrfs/fs.h | 4 +- + fs/btrfs/inode.c | 366 ++-- + fs/btrfs/ioctl.c | 57 +- + fs/btrfs/ioctl.h | 2 +- + fs/btrfs/ordered-data.c | 27 +- + fs/btrfs/qgroup.c | 126 +- + fs/btrfs/qgroup.h | 16 +- + fs/btrfs/raid56.c | 57 +- + fs/btrfs/relocation.c | 2 +- + fs/btrfs/send.c | 2 +- + fs/btrfs/space-info.c | 2 - + fs/btrfs/space-info.h | 4 - + fs/btrfs/super.c | 48 +- + fs/btrfs/sysfs.c | 129 +- + fs/btrfs/tests/extent-io-tests.c | 129 +- + fs/btrfs/transaction.c | 64 +- + fs/btrfs/transaction.h | 31 +- + fs/btrfs/tree-checker.c | 83 +- + fs/btrfs/tree-log.c | 11 +- + fs/btrfs/verity.c | 31 +- + fs/btrfs/volumes.c | 55 +- + fs/btrfs/volumes.h | 1 + + fs/btrfs/xattr.c | 6 +- + fs/btrfs/zlib.c | 10 +- + fs/btrfs/zoned.c | 21 +- + fs/btrfs/zstd.c | 73 +- + fs/buffer.c | 131 +- + fs/cachefiles/Kconfig | 2 +- + fs/cachefiles/interface.c | 96 +- + fs/cachefiles/internal.h | 18 +- + fs/cachefiles/io.c | 472 +++-- + fs/cachefiles/namei.c | 34 +- + fs/cachefiles/xattr.c | 82 +- + fs/ceph/Kconfig | 1 + + fs/ceph/acl.c | 2 +- + fs/ceph/addr.c | 4 +- + fs/ceph/dir.c | 10 +- + fs/ceph/file.c | 2 +- + fs/ceph/inode.c | 10 +- + fs/ceph/mds_client.h | 2 +- + fs/ceph/super.h | 10 +- + fs/ceph/xattr.c | 2 +- + fs/char_dev.c | 4 +- + fs/coda/coda_linux.h | 6 +- + fs/coda/dir.c | 10 +- + fs/coda/inode.c | 4 +- + fs/coda/pioctl.c | 4 +- + fs/configfs/configfs_internal.h | 14 +- + fs/configfs/dir.c | 12 +- + fs/configfs/file.c | 6 +- + fs/configfs/inode.c | 4 +- + fs/configfs/symlink.c | 2 +- + fs/coredump.c | 558 ++++-- + fs/crypto/Kconfig | 2 +- + fs/crypto/crypto.c | 5 + + fs/crypto/fscrypt_private.h | 27 - + fs/crypto/hooks.c | 9 + + fs/crypto/keyring.c | 13 + + fs/crypto/keysetup_v1.c | 5 +- + fs/dax.c | 26 +- + fs/dcache.c | 206 +- + fs/debugfs/inode.c | 2 +- + fs/devpts/inode.c | 2 + + fs/dlm/config.c | 30 +- + fs/dlm/dlm_internal.h | 5 + + fs/dlm/lock.c | 18 +- + fs/dlm/lock.h | 4 +- + fs/dlm/lowcomms.c | 1 + + fs/dlm/midcomms.c | 1 + + fs/dlm/plock.c | 14 +- + fs/dlm/user.c | 23 + + fs/ecryptfs/inode.c | 26 +- + fs/efivarfs/inode.c | 6 +- + fs/erofs/inode.c | 2 +- + fs/erofs/internal.h | 2 +- + fs/eventfd.c | 4 +- + fs/eventpoll.c | 6 +- + fs/exec.c | 28 +- + fs/exfat/dir.c | 52 +- + fs/exfat/exfat_fs.h | 5 +- + fs/exfat/fatent.c | 19 +- + fs/exfat/file.c | 6 +- + fs/exfat/iomap.c | 28 +- + fs/exfat/misc.c | 2 +- + fs/exfat/namei.c | 13 +- + fs/exfat/super.c | 30 +- + fs/ext2/Makefile | 2 + + fs/ext2/acl.c | 2 +- + fs/ext2/acl.h | 2 +- + fs/ext2/balloc.c | 4 + + fs/ext2/ext2.h | 25 +- + fs/ext2/file.c | 7 + + fs/ext2/inode.c | 11 +- + fs/ext2/ioctl.c | 2 +- + fs/ext2/namei.c | 12 +- + fs/ext2/super.c | 33 +- + fs/ext2/xattr.c | 2 +- + fs/ext2/xattr_security.c | 2 +- + fs/ext2/xattr_trusted.c | 2 +- + fs/ext2/xattr_user.c | 2 +- + fs/ext4/acl.c | 2 +- + fs/ext4/acl.h | 2 +- + fs/ext4/ext4.h | 10 +- + fs/ext4/ext4_jbd2.c | 2 +- + fs/ext4/ialloc.c | 2 +- + fs/ext4/inode.c | 6 +- + fs/ext4/ioctl.c | 6 +- + fs/ext4/mmp.c | 2 +- + fs/ext4/namei.c | 16 +- + fs/ext4/symlink.c | 2 +- + fs/ext4/xattr_hurd.c | 2 +- + fs/ext4/xattr_security.c | 2 +- + fs/ext4/xattr_trusted.c | 2 +- + fs/ext4/xattr_user.c | 2 +- + fs/f2fs/Makefile | 2 +- + fs/f2fs/acl.c | 32 +- + fs/f2fs/acl.h | 10 +- + fs/f2fs/cache.c | 720 +++++++ + fs/f2fs/cache.h | 240 +++ + fs/f2fs/checkpoint.c | 558 +++--- + fs/f2fs/compress.c | 183 +- + fs/f2fs/data.c | 648 ++++--- + fs/f2fs/debug.c | 94 +- + fs/f2fs/dir.c | 221 ++- + fs/f2fs/extent_cache.c | 25 +- + fs/f2fs/f2fs.h | 516 +++-- + fs/f2fs/file.c | 195 +- + fs/f2fs/gc.c | 222 ++- + fs/f2fs/inline.c | 293 +-- + fs/f2fs/inode.c | 242 +-- + fs/f2fs/iostat.h | 11 + + fs/f2fs/namei.c | 138 +- + fs/f2fs/node.c | 1277 ++++++------- + fs/f2fs/node.h | 191 +- + fs/f2fs/recovery.c | 255 +-- + fs/f2fs/segment.c | 429 +++-- + fs/f2fs/segment.h | 87 +- + fs/f2fs/shrinker.c | 16 +- + fs/f2fs/super.c | 280 +-- + fs/f2fs/sysfs.c | 32 +- + fs/f2fs/verity.c | 6 +- + fs/f2fs/xattr.c | 137 +- + fs/f2fs/xattr.h | 23 +- + fs/failfs.c | 4 +- + fs/fat/fat.h | 4 +- + fs/fat/file.c | 6 +- + fs/fat/misc.c | 2 +- + fs/fat/namei_msdos.c | 6 +- + fs/fat/namei_vfat.c | 6 +- + fs/fhandle.c | 2 +- + fs/file.c | 321 +++- + fs/file_attr.c | 6 +- + fs/fs-writeback.c | 23 +- + fs/fs_pin.c | 9 +- + fs/fuse/Kconfig | 8 +- + fs/fuse/Makefile | 2 +- + fs/fuse/acl.c | 4 +- + fs/fuse/backing.c | 2 +- + fs/fuse/dax.c | 215 +-- + fs/fuse/dev.c | 50 +- + fs/fuse/dev.h | 2 +- + fs/fuse/dev_uring.c | 6 + + fs/fuse/dir.c | 55 +- + fs/fuse/file.c | 131 +- + fs/fuse/fuse_i.h | 108 +- + fs/fuse/inode.c | 82 +- + fs/fuse/ioctl.c | 5 +- + fs/fuse/iomode.c | 4 +- + fs/fuse/notify.c | 4 +- + fs/fuse/passthrough.c | 15 +- + fs/fuse/req.c | 15 +- + fs/fuse/virtio_fs.c | 28 +- + fs/fuse/xattr.c | 2 +- + fs/gfs2/acl.c | 13 +- + fs/gfs2/acl.h | 2 +- + fs/gfs2/aops.c | 17 +- + fs/gfs2/bmap.c | 93 +- + fs/gfs2/dentry.c | 6 +- + fs/gfs2/dir.c | 61 +- + fs/gfs2/export.c | 7 +- + fs/gfs2/file.c | 74 +- + fs/gfs2/glock.c | 60 +- + fs/gfs2/glops.c | 13 +- + fs/gfs2/incore.h | 8 +- + fs/gfs2/inode.c | 148 +- + fs/gfs2/inode.h | 4 +- + fs/gfs2/lock_dlm.c | 4 +- + fs/gfs2/log.c | 7 +- + fs/gfs2/lops.c | 22 +- + fs/gfs2/meta_io.c | 7 +- + fs/gfs2/ops_fstype.c | 26 +- + fs/gfs2/quota.c | 57 +- + fs/gfs2/recovery.c | 19 +- + fs/gfs2/rgrp.c | 129 +- + fs/gfs2/super.c | 98 +- + fs/gfs2/trace_gfs2.h | 6 +- + fs/gfs2/util.c | 8 +- + fs/gfs2/xattr.c | 71 +- + fs/hfs/attr.c | 2 +- + fs/hfs/dir.c | 6 +- + fs/hfs/hfs_fs.h | 2 +- + fs/hfs/inode.c | 2 +- + fs/hfsplus/dir.c | 10 +- + fs/hfsplus/hfsplus_fs.h | 4 +- + fs/hfsplus/inode.c | 6 +- + fs/hfsplus/xattr.c | 2 +- + fs/hfsplus/xattr_security.c | 2 +- + fs/hfsplus/xattr_trusted.c | 2 +- + fs/hfsplus/xattr_user.c | 2 +- + fs/hostfs/hostfs_kern.c | 14 +- + fs/hpfs/hpfs_fn.h | 2 +- + fs/hpfs/inode.c | 2 +- + fs/hpfs/namei.c | 10 +- + fs/hugetlbfs/inode.c | 14 +- + fs/inode.c | 14 +- + fs/internal.h | 30 +- + fs/iomap/bio.c | 4 +- + fs/iomap/buffered-io.c | 8 +- + fs/iomap/direct-io.c | 46 +- + fs/iomap/ioend.c | 122 +- + fs/isofs/compress.c | 77 +- + fs/isofs/inode.c | 46 +- + fs/isofs/isofs.h | 2 + + fs/isofs/rock.c | 11 +- + fs/jbd2/commit.c | 22 +- + fs/jbd2/journal.c | 31 +- + fs/jbd2/transaction.c | 2 +- + fs/jffs2/acl.c | 2 +- + fs/jffs2/acl.h | 2 +- + fs/jffs2/dir.c | 20 +- + fs/jffs2/fs.c | 2 +- + fs/jffs2/os-linux.h | 2 +- + fs/jffs2/security.c | 2 +- + fs/jffs2/xattr_trusted.c | 2 +- + fs/jffs2/xattr_user.c | 2 +- + fs/jfs/acl.c | 2 +- + fs/jfs/file.c | 2 +- + fs/jfs/ioctl.c | 2 +- + fs/jfs/jfs_acl.h | 2 +- + fs/jfs/jfs_inode.h | 4 +- + fs/jfs/namei.c | 10 +- + fs/jfs/xattr.c | 4 +- + fs/kernfs/dir.c | 231 ++- + fs/kernfs/file.c | 47 +- + fs/kernfs/inode.c | 10 +- + fs/kernfs/kernfs-internal.h | 24 +- + fs/kernfs/mount.c | 34 +- + fs/kernfs/symlink.c | 17 +- + fs/libfs.c | 8 +- + fs/lockd/svc.c | 1 - + fs/lockd/svclock.c | 33 +- + fs/lockd/trace.h | 1 - + fs/lockd/xdr.h | 2 +- + fs/minix/file.c | 2 +- + fs/minix/inode.c | 2 +- + fs/minix/minix.h | 2 +- + fs/minix/namei.c | 12 +- + fs/mnt_idmapping.c | 33 +- + fs/mount.h | 3 +- + fs/namei.c | 155 +- + fs/namespace.c | 75 +- + fs/netfs/Kconfig | 3 + + fs/netfs/Makefile | 2 +- + fs/netfs/buffered_read.c | 217 ++- + fs/netfs/buffered_write.c | 69 +- + fs/netfs/direct_read.c | 14 +- + fs/netfs/direct_write.c | 8 +- + fs/netfs/fscache_cookie.c | 8 +- + fs/netfs/fscache_internal.h | 14 - + fs/netfs/fscache_io.c | 10 +- + fs/netfs/internal.h | 70 +- + fs/netfs/iterator.c | 2 +- + fs/netfs/main.c | 1 - + fs/netfs/misc.c | 10 +- + fs/netfs/objects.c | 7 +- + fs/netfs/read_collect.c | 14 +- + fs/netfs/read_pgpriv2.c | 13 +- + fs/netfs/read_retry.c | 7 +- + fs/netfs/read_single.c | 50 +- + fs/netfs/stats.c | 4 +- + fs/netfs/write_collect.c | 157 +- + fs/netfs/write_issue.c | 141 +- + fs/netfs/write_retry.c | 8 +- + fs/nfs/Kconfig | 18 +- + fs/nfs/blocklayout/Makefile | 3 +- + fs/nfs/blocklayout/blocklayout.c | 119 +- + fs/nfs/blocklayout/dev.c | 32 +- + fs/nfs/callback_proc.c | 23 +- + fs/nfs/callback_xdr.c | 9 +- + fs/nfs/client.c | 6 +- + fs/nfs/dir.c | 12 +- + fs/nfs/direct.c | 139 +- + fs/nfs/filelayout/filelayout.c | 17 +- + fs/nfs/filelayout/filelayoutdev.c | 19 +- + fs/nfs/flexfilelayout/flexfilelayout.c | 413 ++-- + fs/nfs/flexfilelayout/flexfilelayout.h | 48 +- + fs/nfs/flexfilelayout/flexfilelayoutdev.c | 212 ++- + fs/nfs/inode.c | 4 +- + fs/nfs/internal.h | 13 +- + fs/nfs/namespace.c | 4 +- + fs/nfs/netns.h | 5 +- + fs/nfs/nfs3_fs.h | 2 +- + fs/nfs/nfs3acl.c | 2 +- + fs/nfs/nfs3client.c | 9 +- + fs/nfs/nfs4_fs.h | 3 + + fs/nfs/nfs4client.c | 8 +- + fs/nfs/nfs4file.c | 1 + + fs/nfs/nfs4proc.c | 146 +- + fs/nfs/nfs4state.c | 7 +- + fs/nfs/nfs4xdr.c | 2 +- + fs/nfs/pagelist.c | 66 +- + fs/nfs/pnfs.c | 376 +++- + fs/nfs/pnfs.h | 102 +- + fs/nfs/pnfs_dev.c | 30 +- + fs/nfs/pnfs_nfs.c | 108 +- + fs/nfs/read.c | 2 +- + fs/nfs/super.c | 25 - + fs/nfs/unlink.c | 3 + + fs/nfs/write.c | 2 +- + fs/nfs_common/nfs_ssc.c | 126 +- + fs/nfsd/blocklayout.c | 1 + + fs/nfsd/blocklayoutxdr.c | 11 + + fs/nfsd/export.c | 8 +- + fs/nfsd/export.h | 3 +- + fs/nfsd/filecache.c | 1 + + fs/nfsd/flexfilelayout.c | 24 +- + fs/nfsd/flexfilelayoutxdr.c | 9 +- + fs/nfsd/flexfilelayoutxdr.h | 8 +- + fs/nfsd/localio.c | 10 +- + fs/nfsd/lockd.c | 5 +- + fs/nfsd/netns.h | 17 +- + fs/nfsd/nfs2acl.c | 48 +- + fs/nfsd/nfs3acl.c | 1 + + fs/nfsd/nfs3proc.c | 79 +- + fs/nfsd/nfs3xdr.c | 4 + + fs/nfsd/nfs4acl.c | 1 + + fs/nfsd/nfs4callback.c | 8 +- + fs/nfsd/nfs4ctl.h | 83 + + fs/nfsd/nfs4idmap.c | 1 + + fs/nfsd/nfs4layouts.c | 1 + + fs/nfsd/nfs4proc.c | 465 +++-- + fs/nfsd/nfs4recover.c | 1 + + fs/nfsd/nfs4state.c | 587 +++++- + fs/nfsd/nfs4xdr.c | 188 +- + fs/nfsd/nfscache.c | 1 + + fs/nfsd/nfsctl.c | 36 +- + fs/nfsd/nfsd.h | 237 +-- + fs/nfsd/nfserr.h | 159 ++ + fs/nfsd/nfsfh.c | 73 +- + fs/nfsd/nfsfh.h | 28 +- + fs/nfsd/nfsproc.c | 39 +- + fs/nfsd/nfssvc.c | 8 + + fs/nfsd/nfsxdr.c | 1 + + fs/nfsd/state.h | 46 +- + fs/nfsd/trace.h | 45 +- + fs/nfsd/vfs.c | 311 ++- + fs/nfsd/vfs.h | 39 +- + fs/nfsd/xdr.h | 6 +- + fs/nfsd/xdr3.h | 2 +- + fs/nfsd/xdr4.h | 156 +- + fs/nfsd/xdr4cb.h | 20 +- + fs/nilfs2/inode.c | 4 +- + fs/nilfs2/ioctl.c | 2 +- + fs/nilfs2/namei.c | 10 +- + fs/nilfs2/nilfs.h | 6 +- + fs/nls/nls_iso8859-14.c | 26 +- + fs/notify/fanotify/fanotify_user.c | 2 +- + fs/nsfs.c | 4 +- + fs/ntfs/attrib.c | 76 +- + fs/ntfs/attrlist.c | 14 +- + fs/ntfs/bitmap.c | 19 +- + fs/ntfs/collate.c | 8 +- + fs/ntfs/dir.c | 23 +- + fs/ntfs/ea.c | 10 +- + fs/ntfs/ea.h | 6 +- + fs/ntfs/file.c | 60 +- + fs/ntfs/index.c | 2 +- + fs/ntfs/index.h | 2 +- + fs/ntfs/inode.c | 131 +- + fs/ntfs/inode.h | 4 +- + fs/ntfs/iomap.c | 16 +- + fs/ntfs/layout.h | 19 +- + fs/ntfs/mft.c | 921 ++++++--- + fs/ntfs/mft.h | 19 +- + fs/ntfs/mst.c | 4 +- + fs/ntfs/namei.c | 43 +- + fs/ntfs/ntfs.h | 7 +- + fs/ntfs/reparse.c | 2 +- + fs/ntfs/runlist.c | 5 +- + fs/ntfs/super.c | 391 +++- + fs/ntfs/unistr.c | 29 +- + fs/ntfs/volume.h | 15 + + fs/ntfs3/attrib.c | 47 +- + fs/ntfs3/dir.c | 3 + + fs/ntfs3/file.c | 9 +- + fs/ntfs3/frecord.c | 5 + + fs/ntfs3/fslog.c | 4 +- + fs/ntfs3/fsntfs.c | 3 + + fs/ntfs3/index.c | 13 +- + fs/ntfs3/inode.c | 21 +- + fs/ntfs3/namei.c | 10 +- + fs/ntfs3/ntfs_fs.h | 16 +- + fs/ntfs3/record.c | 16 +- + fs/ntfs3/super.c | 2 +- + fs/ntfs3/xattr.c | 26 +- + fs/ocfs2/acl.c | 2 +- + fs/ocfs2/acl.h | 2 +- + fs/ocfs2/buffer_head_io.c | 12 +- + fs/ocfs2/dlmfs/dlmfs.c | 6 +- + fs/ocfs2/file.c | 6 +- + fs/ocfs2/file.h | 6 +- + fs/ocfs2/ioctl.c | 2 +- + fs/ocfs2/ioctl.h | 2 +- + fs/ocfs2/journal.c | 25 +- + fs/ocfs2/namei.c | 10 +- + fs/ocfs2/quota_local.c | 4 + + fs/ocfs2/xattr.c | 6 +- + fs/omfs/dir.c | 6 +- + fs/omfs/file.c | 2 +- + fs/omfs/inode.c | 4 +- + fs/open.c | 50 +- + fs/orangefs/acl.c | 2 +- + fs/orangefs/inode.c | 29 +- + fs/orangefs/namei.c | 8 +- + fs/orangefs/orangefs-debugfs.c | 10 +- + fs/orangefs/orangefs-kernel.h | 8 +- + fs/orangefs/super.c | 18 + + fs/orangefs/xattr.c | 2 +- + fs/overlayfs/dir.c | 14 +- + fs/overlayfs/file.c | 2 +- + fs/overlayfs/inode.c | 16 +- + fs/overlayfs/overlayfs.h | 14 +- + fs/overlayfs/ovl_entry.h | 2 +- + fs/overlayfs/util.c | 4 +- + fs/overlayfs/xattrs.c | 4 +- + fs/pidfs.c | 6 +- + fs/pipe.c | 2 +- + fs/pnode.c | 108 +- + fs/posix_acl.c | 26 +- + fs/proc/base.c | 53 +- + fs/proc/fd.c | 6 +- + fs/proc/fd.h | 2 +- + fs/proc/generic.c | 4 +- + fs/proc/internal.h | 4 +- + fs/proc/proc_net.c | 2 +- + fs/proc/proc_sysctl.c | 6 +- + fs/proc/root.c | 2 +- + fs/proc/vmcore.c | 20 + + fs/quota/dquot.c | 18 +- + fs/ramfs/file-nommu.c | 4 +- + fs/ramfs/inode.c | 10 +- + fs/read_write.c | 8 +- + fs/remap_range.c | 2 +- + fs/smb/client/cifsacl.c | 4 +- + fs/smb/client/cifsfs.c | 2 +- + fs/smb/client/cifsfs.h | 16 +- + fs/smb/client/cifsproto.h | 4 +- + fs/smb/client/dir.c | 6 +- + fs/smb/client/inode.c | 8 +- + fs/smb/client/link.c | 2 +- + fs/smb/client/transport.c | 13 +- + fs/smb/client/xattr.c | 2 +- + fs/smb/common/compress/compress.c | 16 +- + fs/smb/common/fscc.h | 5 +- + fs/smb/common/smbglob.h | 1 + + fs/smb/server/Kconfig | 2 + + fs/smb/server/Makefile | 1 + + fs/smb/server/compress.c | 1 + + fs/smb/server/connection.c | 3 +- + fs/smb/server/connection.h | 3 + + fs/smb/server/ksmbd_work.c | 3 - + fs/smb/server/ksmbd_work.h | 5 +- + fs/smb/server/mgmt/user_session.c | 4 +- + fs/smb/server/ndr.c | 2 +- + fs/smb/server/ndr.h | 2 +- + fs/smb/server/oplock.c | 2 +- + fs/smb/server/smb2ops.c | 33 + + fs/smb/server/smb2pdu.c | 710 +++---- + fs/smb/server/smb2pdu.h | 22 +- + fs/smb/server/smb_common.c | 6 +- + fs/smb/server/smbacl.c | 27 +- + fs/smb/server/smbacl.h | 8 +- + fs/smb/server/tests/Kconfig | 15 + + fs/smb/server/tests/Makefile | 4 + + fs/smb/server/tests/smbacl_kunit.c | 301 +++ + fs/smb/server/transport_tcp.c | 8 +- + fs/smb/server/vfs.c | 52 +- + fs/smb/server/vfs.h | 33 +- + fs/smb/server/vfs_cache.c | 66 +- + fs/smb/server/vfs_cache.h | 6 - + fs/splice.c | 70 +- + fs/stat.c | 4 +- + fs/super.c | 19 +- + fs/tests/.kunitconfig | 2 + + fs/tests/fdtable_kunit.c | 72 + + fs/tracefs/event_inode.c | 2 +- + fs/tracefs/inode.c | 8 +- + fs/ubifs/dir.c | 14 +- + fs/ubifs/file.c | 4 +- + fs/ubifs/ioctl.c | 2 +- + fs/ubifs/ubifs.h | 6 +- + fs/ubifs/xattr.c | 2 +- + fs/udf/file.c | 2 +- + fs/udf/inode.c | 59 +- + fs/udf/misc.c | 15 +- + fs/udf/namei.c | 12 +- + fs/udf/super.c | 5 +- + fs/udf/symlink.c | 2 +- + fs/ufs/dir.c | 2 +- + fs/ufs/inode.c | 2 +- + fs/ufs/namei.c | 10 +- + fs/ufs/ufs.h | 2 +- + fs/vboxsf/dir.c | 8 +- + fs/vboxsf/utils.c | 4 +- + fs/vboxsf/vfsmod.h | 4 +- + fs/xattr.c | 30 +- + fs/xfs/libxfs/xfs_attr.c | 14 +- + fs/xfs/libxfs/xfs_bmap.c | 5 +- + fs/xfs/libxfs/xfs_bmap.h | 2 +- + fs/xfs/libxfs/xfs_btree_mem.c | 6 +- + fs/xfs/libxfs/xfs_dquot_buf.c | 13 +- + fs/xfs/libxfs/xfs_errortag.h | 6 +- + fs/xfs/libxfs/xfs_exchmaps.c | 1 - + fs/xfs/libxfs/xfs_exchmaps.h | 3 +- + fs/xfs/libxfs/xfs_ialloc.c | 2 +- + fs/xfs/libxfs/xfs_ialloc_btree.c | 1 - + fs/xfs/libxfs/xfs_ialloc_btree.h | 5 +- + fs/xfs/libxfs/xfs_inode_fork.c | 4 +- + fs/xfs/libxfs/xfs_inode_util.h | 2 +- + fs/xfs/libxfs/xfs_metadir.c | 32 +- + fs/xfs/libxfs/xfs_metadir.h | 5 +- + fs/xfs/libxfs/xfs_parent.h | 1 - + fs/xfs/libxfs/xfs_refcount_btree.c | 16 +- + fs/xfs/libxfs/xfs_refcount_btree.h | 26 + + fs/xfs/libxfs/xfs_rmap.c | 13 +- + fs/xfs/libxfs/xfs_rmap_btree.c | 16 +- + fs/xfs/libxfs/xfs_rmap_btree.h | 26 + + fs/xfs/libxfs/xfs_rtrefcount_btree.c | 121 +- + fs/xfs/libxfs/xfs_rtrefcount_btree.h | 5 +- + fs/xfs/libxfs/xfs_rtrmap_btree.c | 236 +-- + fs/xfs/libxfs/xfs_rtrmap_btree.h | 3 +- + fs/xfs/libxfs/xfs_sb.c | 40 - + fs/xfs/libxfs/xfs_sb.h | 1 - + fs/xfs/libxfs/xfs_symlink_remote.c | 8 +- + fs/xfs/libxfs/xfs_symlink_remote.h | 2 +- + fs/xfs/libxfs/xfs_trans_resv.c | 45 +- + fs/xfs/libxfs/xfs_trans_space.c | 6 +- + fs/xfs/libxfs/xfs_types.c | 7 +- + fs/xfs/libxfs/xfs_types.h | 7 +- + fs/xfs/scrub/alloc_repair.c | 15 +- + fs/xfs/scrub/bmap.c | 11 +- + fs/xfs/scrub/bmap_repair.c | 4 +- + fs/xfs/scrub/inode_repair.c | 32 +- + fs/xfs/scrub/orphanage.c | 2 +- + fs/xfs/scrub/quota.c | 2 +- + fs/xfs/scrub/quota_repair.c | 9 +- + fs/xfs/scrub/quotacheck_repair.c | 54 +- + fs/xfs/scrub/rtrefcount_repair.c | 3 +- + fs/xfs/scrub/rtrmap_repair.c | 2 +- + fs/xfs/scrub/xfarray.c | 201 +- + fs/xfs/scrub/xfarray.h | 9 +- + fs/xfs/xfs_acl.c | 2 +- + fs/xfs/xfs_acl.h | 2 +- + fs/xfs/xfs_aops.c | 13 +- + fs/xfs/xfs_bmap_item.c | 2 +- + fs/xfs/xfs_buf.c | 11 + + fs/xfs/xfs_dquot.c | 20 +- + fs/xfs/xfs_dquot.h | 2 +- + fs/xfs/xfs_exchmaps_item.c | 4 +- + fs/xfs/xfs_exchrange.c | 2 +- + fs/xfs/xfs_file.c | 33 +- + fs/xfs/xfs_handle.c | 8 +- + fs/xfs/xfs_inode.c | 24 +- + fs/xfs/xfs_inode.h | 2 +- + fs/xfs/xfs_ioctl.c | 423 +++-- + fs/xfs/xfs_ioctl.h | 6 +- + fs/xfs/xfs_ioctl32.c | 188 +- + fs/xfs/xfs_ioend.c | 137 +- + fs/xfs/xfs_ioend.h | 2 + + fs/xfs/xfs_iops.c | 24 +- + fs/xfs/xfs_iops.h | 2 +- + fs/xfs/xfs_itable.c | 2 +- + fs/xfs/xfs_itable.h | 2 +- + fs/xfs/xfs_mount.h | 8 + + fs/xfs/xfs_qm.c | 6 +- + fs/xfs/xfs_rmap_item.c | 2 +- + fs/xfs/xfs_rtalloc.h | 49 +- + fs/xfs/xfs_super.c | 3 +- + fs/xfs/xfs_symlink.c | 6 +- + fs/xfs/xfs_symlink.h | 2 +- + fs/xfs/xfs_sysfs.c | 78 +- + fs/xfs/xfs_trace.h | 29 +- + fs/xfs/xfs_trans_dquot.c | 6 +- + fs/xfs/xfs_xattr.c | 2 +- + fs/xfs/xfs_zone_alloc.c | 4 + + fs/zonefs/super.c | 2 +- + include/linux/binfmts.h | 3 +- + include/linux/bio-integrity.h | 3 +- + include/linux/bio.h | 11 +- + include/linux/blkdev.h | 8 +- + include/linux/buffer_head.h | 86 +- + include/linux/capability.h | 8 +- + include/linux/cleanup.h | 7 - + include/linux/configfs.h | 73 +- + include/linux/coredump.h | 37 +- + include/linux/dax.h | 12 - + include/linux/dcache.h | 38 +- + include/linux/f2fs_fs.h | 140 +- + include/linux/fdtable.h | 15 +- + include/linux/file.h | 130 +- + include/linux/fileattr.h | 2 +- + include/linux/fs.h | 127 +- + include/linux/fs_context.h | 4 + + include/linux/fscache-cache.h | 2 +- + include/linux/fscache.h | 53 +- + include/linux/iomap.h | 38 +- + include/linux/lsm_hook_defs.h | 25 +- + include/linux/mnt_idmapping.h | 24 +- + include/linux/mount.h | 4 +- + include/linux/namei.h | 19 +- + include/linux/netfs.h | 112 +- + include/linux/nfs.h | 55 +- + include/linux/nfs3.h | 43 + + include/linux/nfs4.h | 6 + + include/linux/nfs_fh.h | 63 + + include/linux/nfs_fs.h | 6 +- + include/linux/nfs_fs_sb.h | 8 + + include/linux/nfs_page.h | 8 +- + include/linux/nfs_ssc.h | 69 +- + include/linux/nfs_xdr.h | 2 + + include/linux/nfsd_ssc.h | 38 + + include/linux/nfslocalio.h | 11 +- + include/linux/posix_acl.h | 24 +- + include/linux/quotaops.h | 6 +- + include/linux/sched.h | 2 +- + include/linux/sched/signal.h | 33 +- + include/linux/security.h | 61 +- + include/linux/splice.h | 4 +- + include/linux/sunrpc/svc_xprt.h | 5 +- + include/linux/uidgid.h | 12 +- + include/linux/user_namespace.h | 11 +- + include/linux/wait_bit.h | 26 + + include/linux/xattr.h | 20 +- + include/trace/events/cachefiles.h | 81 +- + include/trace/events/f2fs.h | 71 + + include/trace/events/fscache.h | 10 +- + include/trace/events/netfs.h | 145 +- + include/trace/misc/nfs.h | 13 +- + include/uapi/linux/btrfs_tree.h | 25 +- + include/uapi/linux/close_range.h | 31 +- + include/uapi/linux/coredump.h | 149 +- + include/uapi/linux/fs.h | 2 +- + include/uapi/linux/fuse.h | 12 +- + init/Kconfig | 11 + + init/initramfs.c | 11 +- + io_uring/io-wq.c | 2 + + io_uring/mock_file.c | 8 +- + ipc/mqueue.c | 2 +- + kernel/Makefile | 1 + + kernel/bpf/bpf_iter.c | 6 +- + kernel/bpf/inode.c | 6 +- + kernel/bpf/token.c | 6 +- + kernel/capability.c | 4 +- + kernel/exit.c | 15 +- + kernel/fork.c | 68 +- + kernel/kthread.c | 2 +- + kernel/pid_namespace.c | 3 +- + kernel/ptrace.c | 6 + + kernel/signal.c | 14 + + kernel/tests/.kunitconfig | 4 + + kernel/tests/user_ns_map_kunit.c | 98 + + kernel/user_namespace.c | 60 +- + kernel/utsname.c | 1 - + mm/secretmem.c | 2 +- + mm/shmem.c | 30 +- + mm/shmem_quota.c | 5 + + mm/userfaultfd.c | 6 +- + net/core/scm.c | 8 +- + net/handshake/netlink.c | 8 +- + net/kcm/kcmsock.c | 6 +- + net/socket.c | 6 +- + net/sunrpc/svc_xprt.c | 36 +- + net/sunrpc/svcauth_unix.c | 9 + + net/sunrpc/svcsock.c | 490 +++-- + net/sunrpc/xprtrdma/svc_rdma_recvfrom.c | 18 +- + net/sunrpc/xprtrdma/svc_rdma_transport.c | 3 +- + net/sunrpc/xprtsock.c | 69 +- + net/unix/af_unix.c | 2 +- + rust/kernel/configfs.rs | 127 +- + samples/configfs/configfs_sample.c | 137 +- + security/apparmor/apparmorfs.c | 2 +- + security/apparmor/lsm.c | 4 +- + security/commoncap.c | 10 +- + security/integrity/evm/evm_main.c | 26 +- + security/integrity/ima/ima.h | 10 +- + security/integrity/ima/ima_api.c | 2 +- + security/integrity/ima/ima_appraise.c | 12 +- + security/integrity/ima/ima_main.c | 6 +- + security/integrity/ima/ima_policy.c | 4 +- + security/security.c | 49 +- + security/selinux/hooks.c | 45 +- + security/selinux/selinuxfs.c | 2 +- + security/smack/smack_lsm.c | 14 +- + tools/include/uapi/linux/coredump.h | 149 +- + tools/testing/selftests/Makefile | 4 + + .../selftests/clone3/clone3_clear_sighand.c | 6 +- + tools/testing/selftests/core/close_range_test.c | 951 ++++++++++ + tools/testing/selftests/coredump/.gitignore | 2 + + tools/testing/selftests/coredump/Makefile | 11 +- + .../selftests/coredump/coredump_notify_signal.h | 29 + + .../coredump/coredump_notify_signal_helper.c | 46 + + .../coredump/coredump_notify_signal_test.c | 245 +++ + .../selftests/coredump/coredump_signal_test.c | 238 +++ + .../coredump/coredump_socket_protocol_test.c | 1983 +++++++++++++++----- + tools/testing/selftests/coredump/coredump_test.h | 32 +- + .../selftests/coredump/coredump_test_helpers.c | 1742 ++++++++++++++++- + .../selftests/coredump/coredump_test_helpers.h | 79 + + .../selftests/coredump/coredump_worker_test.c | 447 +++++ + tools/testing/selftests/exec/Makefile | 4 + + tools/testing/selftests/exec/binfmt_misc_delim.c | 127 ++ + tools/testing/selftests/filesystems/.gitignore | 1 - + tools/testing/selftests/filesystems/Makefile | 2 +- + tools/testing/selftests/filesystems/config | 8 + + .../selftests/filesystems/configfs/.gitignore | 2 + + .../selftests/filesystems/configfs/Makefile | 8 + + .../testing/selftests/filesystems/configfs/config | 5 + + .../selftests/filesystems/configfs/configfs_test.c | 481 +++++ + .../selftests/filesystems/file_stressor/.gitignore | 2 + + .../selftests/filesystems/file_stressor/Makefile | 6 + + .../{ => file_stressor}/file_stressor.c | 0 + .../selftests/filesystems/file_stressor/settings | 3 + + .../selftests/filesystems/fscontext_ns/.gitignore | 2 + + .../testing/selftests/filesystems/fuse/.gitignore | 1 + + tools/testing/selftests/filesystems/fuse/Makefile | 2 +- + tools/testing/selftests/filesystems/kernfs_test.c | 1203 +++++++++++- + .../filesystems/mntns_unbindable/Makefile | 6 + + .../mntns_unbindable/mntns_unbindable_test.c | 227 +++ + .../selftests/filesystems/openat2/openat2_test.c | 10 +- + .../selftests/filesystems/openat2/resolve_test.c | 7 +- + .../filesystems/statmount/statmount_test.c | 2 +- + .../filesystems/umount_propagation/Makefile | 6 + + .../umount_propagation/umount_propagation_test.c | 226 +++ + .../move_mount_set_group_test.c | 74 +- + tools/testing/selftests/pidfd/pidfd_open_test.c | 2 +- + virt/kvm/guest_memfd.c | 2 +- + 861 files changed, 27304 insertions(+), 15917 deletions(-) + delete mode 100644 Documentation/filesystems/bfs.rst + create mode 100644 Documentation/filesystems/exfat.rst + delete mode 100644 arch/powerpc/platforms/cell/spufs/coredump.c + delete mode 100644 fs/bfs/Kconfig + delete mode 100644 fs/bfs/Makefile + delete mode 100644 fs/bfs/bfs.h + delete mode 100644 fs/bfs/file.c + delete mode 100644 fs/bfs/inode.c + create mode 100644 fs/f2fs/cache.c + create mode 100644 fs/f2fs/cache.h + delete mode 100644 fs/netfs/fscache_internal.h + create mode 100644 fs/nfsd/nfs4ctl.h + create mode 100644 fs/nfsd/nfserr.h + create mode 100644 fs/smb/server/tests/Kconfig + create mode 100644 fs/smb/server/tests/Makefile + create mode 100644 fs/smb/server/tests/smbacl_kunit.c + create mode 100644 fs/tests/.kunitconfig + create mode 100644 fs/tests/fdtable_kunit.c + create mode 100644 include/linux/nfs_fh.h + create mode 100644 include/linux/nfsd_ssc.h + create mode 100644 kernel/tests/.kunitconfig + create mode 100644 kernel/tests/user_ns_map_kunit.c + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal.h + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal_helper.c + create mode 100644 tools/testing/selftests/coredump/coredump_notify_signal_test.c + create mode 100644 tools/testing/selftests/coredump/coredump_signal_test.c + create mode 100644 tools/testing/selftests/coredump/coredump_test_helpers.h + create mode 100644 tools/testing/selftests/coredump/coredump_worker_test.c + create mode 100644 tools/testing/selftests/exec/binfmt_misc_delim.c + create mode 100644 tools/testing/selftests/filesystems/config + create mode 100644 tools/testing/selftests/filesystems/configfs/.gitignore + create mode 100644 tools/testing/selftests/filesystems/configfs/Makefile + create mode 100644 tools/testing/selftests/filesystems/configfs/config + create mode 100644 tools/testing/selftests/filesystems/configfs/configfs_test.c + create mode 100644 tools/testing/selftests/filesystems/file_stressor/.gitignore + create mode 100644 tools/testing/selftests/filesystems/file_stressor/Makefile + rename tools/testing/selftests/filesystems/{ => file_stressor}/file_stressor.c (100%) + create mode 100644 tools/testing/selftests/filesystems/file_stressor/settings + create mode 100644 tools/testing/selftests/filesystems/fscontext_ns/.gitignore + create mode 100644 tools/testing/selftests/filesystems/mntns_unbindable/Makefile + create mode 100644 tools/testing/selftests/filesystems/mntns_unbindable/mntns_unbindable_test.c + create mode 100644 tools/testing/selftests/filesystems/umount_propagation/Makefile + create mode 100644 tools/testing/selftests/filesystems/umount_propagation/umount_propagation_test.c +Merging printk/for-next (3fa6f22c1dc83 Merge branch 'for-7.4' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/printk/linux.git printk/for-next +Auto-merging lib/vsprintf.c +Merge made by the 'ort' strategy. + kernel/printk/printk.c | 94 +++++++++++++++++++++------------------ + kernel/printk/printk_ringbuffer.c | 17 +++---- + lib/vsprintf.c | 11 ++--- + 3 files changed, 62 insertions(+), 60 deletions(-) +Merging pci/next (81bebadfa0f7f Merge branch 'pci/misc') +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pci/pci.git pci/next +Auto-merging MAINTAINERS +Auto-merging arch/arm64/boot/dts/qcom/qcs6490-rb3gen2.dts +Auto-merging drivers/pci/controller/dwc/pci-imx6.c +Auto-merging drivers/pci/endpoint/pci-ep-msi.c +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-bus-pci | 14 +- + Documentation/PCI/controller/index.rst | 1 - + .../PCI/controller/rcar-pcie-firmware.rst | 32 - + .../bindings/pci/nvidia,tegra20-pcie.txt | 670 --------------------- + .../bindings/pci/nvidia,tegra20-pcie.yaml | 532 ++++++++++++++++ + .../devicetree/bindings/pci/rcar-gen4-pci-ep.yaml | 9 +- + .../bindings/pci/rcar-gen4-pci-host.yaml | 81 ++- + .../bindings/pci/renesas,r9a08g045-pcie.yaml | 33 +- + .../devicetree/bindings/pci/toshiba,tc9563.yaml | 14 +- + .../devicetree/bindings/pci/xilinx-versal-cpm.yaml | 38 ++ + Documentation/driver-api/pci/p2pdma.rst | 6 +- + MAINTAINERS | 1 - + arch/arm64/boot/dts/qcom/qcs6490-rb3gen2.dts | 7 +- + drivers/gpio/Kconfig | 11 + + drivers/gpio/Makefile | 1 + + drivers/gpio/gpio-tc9563.c | 99 +++ + drivers/net/wireless/ath/ath10k/Kconfig | 2 +- + drivers/net/wireless/ath/ath10k/pci.c | 11 +- + drivers/net/wireless/ath/ath10k/pci.h | 5 +- + drivers/net/wireless/ath/ath11k/Kconfig | 2 +- + drivers/net/wireless/ath/ath11k/pci.c | 19 +- + drivers/net/wireless/ath/ath11k/pci.h | 3 +- + drivers/net/wireless/ath/ath12k/Kconfig | 2 +- + drivers/net/wireless/ath/ath12k/pci.c | 19 +- + drivers/net/wireless/ath/ath12k/pci.h | 4 +- + .../pci/controller/cadence/pcie-cadence-debugfs.c | 15 +- + drivers/pci/controller/cadence/pcie-cadence-plat.c | 6 +- + drivers/pci/controller/cadence/pcie-cadence.c | 16 +- + drivers/pci/controller/cadence/pcie-cadence.h | 2 - + drivers/pci/controller/dwc/pci-dra7xx.c | 22 +- + drivers/pci/controller/dwc/pci-imx6.c | 94 ++- + drivers/pci/controller/dwc/pci-keystone.c | 32 +- + drivers/pci/controller/dwc/pcie-designware-ep.c | 8 +- + drivers/pci/controller/dwc/pcie-designware.h | 1 + + drivers/pci/controller/dwc/pcie-dw-rockchip.c | 1 + + drivers/pci/controller/dwc/pcie-histb.c | 1 + + drivers/pci/controller/dwc/pcie-qcom-ep.c | 1 + + drivers/pci/controller/dwc/pcie-qcom.c | 176 +++++- + drivers/pci/controller/dwc/pcie-rcar-gen4.c | 325 ++++++++-- + drivers/pci/controller/dwc/pcie-spacemit-k1.c | 3 + + drivers/pci/controller/dwc/pcie-tegra194.c | 1 + + drivers/pci/controller/pci-tegra.c | 1 + + drivers/pci/controller/pci-xgene.c | 11 +- + drivers/pci/controller/pcie-aspeed.c | 9 +- + drivers/pci/controller/pcie-mediatek-gen3.c | 15 +- + drivers/pci/controller/pcie-mediatek.c | 4 +- + drivers/pci/controller/pcie-rockchip-host.c | 1 + + drivers/pci/controller/pcie-rzg3s-host.c | 76 ++- + drivers/pci/controller/pcie-xilinx-cpm.c | 77 ++- + drivers/pci/controller/plda/pcie-starfive.c | 1 + + drivers/pci/controller/vmd.c | 57 +- + drivers/pci/endpoint/functions/pci-epf-vntb.c | 93 ++- + drivers/pci/endpoint/pci-ep-msi.c | 40 +- + drivers/pci/hotplug/pciehp_core.c | 2 +- + drivers/pci/iov.c | 25 - + drivers/pci/p2pdma.c | 39 +- + drivers/pci/pci-label.c | 94 ++- + drivers/pci/pci.h | 13 +- + drivers/pci/pcie/aspm.c | 140 +++-- + drivers/pci/probe.c | 15 +- + drivers/pci/pwrctrl/Kconfig | 2 + + drivers/pci/pwrctrl/pci-pwrctrl-tc9563.c | 321 ++++++---- + drivers/pci/quirks.c | 126 +++- + drivers/pci/tph.c | 6 +- + include/linux/pci-epf.h | 3 +- + include/linux/pci.h | 7 +- + include/linux/soc/qcom/tc9563.h | 19 + + 67 files changed, 2244 insertions(+), 1273 deletions(-) + delete mode 100644 Documentation/PCI/controller/rcar-pcie-firmware.rst + delete mode 100644 Documentation/devicetree/bindings/pci/nvidia,tegra20-pcie.txt + create mode 100644 Documentation/devicetree/bindings/pci/nvidia,tegra20-pcie.yaml + create mode 100644 drivers/gpio/gpio-tc9563.c + create mode 100644 include/linux/soc/qcom/tc9563.h +Merging pstore/for-next/pstore (7c756181175d5 pstore: publish big_oops_buf after max_compressed_size) +$ git merge -m Merge branch 'for-next/pstore' of https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git pstore/for-next/pstore +Merge made by the 'ort' strategy. + fs/pstore/platform.c | 12 +++++++++--- + fs/pstore/ram.c | 1 + + fs/pstore/ram_core.c | 6 ++++++ + 3 files changed, 16 insertions(+), 3 deletions(-) +Merging hid/for-next (145c2b2e9a5c0 Merge branch 'for-7.3/upstream-fixes' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/hid/hid.git hid/for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 6 + + drivers/hid/Kconfig | 21 +- + drivers/hid/Makefile | 2 +- + drivers/hid/amd-sfh-hid/amd_sfh_common.h | 2 + + drivers/hid/amd-sfh-hid/amd_sfh_pcie.c | 10 + + drivers/hid/amd-sfh-hid/sfh1_1/amd_sfh_desc.c | 13 +- + drivers/hid/amd-sfh-hid/sfh1_1/amd_sfh_interface.c | 37 +- + drivers/hid/amd-sfh-hid/sfh1_1/amd_sfh_interface.h | 7 + + drivers/hid/bpf/hid_bpf_dispatch.c | 2 +- + .../hid/bpf/progs/Huion__Inspiroy-Frego-M.bpf.c | 11 +- + drivers/hid/hid-appletb-kbd.c | 95 ++- + drivers/hid/hid-asus.c | 14 +- + drivers/hid/hid-google-hammer.c | 2 +- + drivers/hid/hid-ids.h | 15 + + drivers/hid/hid-kysona.c | 292 -------- + drivers/hid/hid-lenovo.c | 14 + + drivers/hid/hid-logitech-dj.c | 48 +- + drivers/hid/hid-logitech-hidpp.c | 51 +- + drivers/hid/hid-multitouch.c | 2 +- + drivers/hid/hid-nintendo.c | 14 +- + drivers/hid/hid-pulsar.c | 765 +++++++++++++++++++++ + drivers/hid/hid-universal-pidff.c | 1 + + drivers/hid/i2c-hid/i2c-hid-acpi.c | 8 +- + drivers/hid/i2c-hid/i2c-hid-core.c | 11 +- + .../intel-thc-hid/intel-quicki2c/pci-quicki2c.c | 10 +- + .../intel-thc-hid/intel-quicki2c/quicki2c-dev.h | 2 + + .../intel-thc-hid/intel-quicki2c/quicki2c-hid.c | 6 +- + .../intel-thc-hid/intel-quicki2c/quicki2c-hid.h | 2 +- + .../intel-thc-hid/intel-quickspi/pci-quickspi.c | 5 +- + .../intel-thc-hid/intel-quickspi/quickspi-dev.h | 2 + + .../intel-thc-hid/intel-quickspi/quickspi-hid.c | 6 +- + .../intel-thc-hid/intel-quickspi/quickspi-hid.h | 2 +- + .../intel-quickspi/quickspi-protocol.c | 2 +- + drivers/hid/surface-hid/surface_kbd.c | 4 +- + drivers/hid/usbhid/hiddev.c | 45 +- + drivers/hid/wacom.h | 2 +- + drivers/hid/wacom_sys.c | 196 ++++-- + drivers/hid/wacom_wac.c | 72 +- + drivers/hid/wacom_wac.h | 6 +- + include/linux/hiddev.h | 2 + + tools/testing/selftests/hid/hid_bpf.c | 31 + + 41 files changed, 1270 insertions(+), 568 deletions(-) + delete mode 100644 drivers/hid/hid-kysona.c + create mode 100644 drivers/hid/hid-pulsar.c +Merging i2c/i2c/for-next (8cd9520d35a6c Linux 7.1) +$ git merge -m Merge branch 'i2c/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/wsa/linux.git i2c/i2c/for-next +Already up to date. +$ git am -3 ../patches/0001-i2c-Fix-up-the-rest-of-the-merge.patch +Applying: i2c: Fix up the rest of the merge +Using index info to reconstruct a base tree... +M drivers/i2c/i2c-core-base.c +Falling back to patching base and 3-way merge... +Auto-merging drivers/i2c/i2c-core-base.c +No changes -- Patch already applied. +Merging i2c-andi/i2c/i2c-next (e33eb6c27c095 Merge branch 'i2c/i2c' into i2c/i2c-next) +$ git merge -m Merge branch 'i2c/i2c-next' of https://git.kernel.org/pub/scm/linux/kernel/git/andi.shyti/linux.git i2c-andi/i2c/i2c-next +Merge made by the 'ort' strategy. + .../bindings/i2c/marvell,mv64xxx-i2c.yaml | 1 + + .../devicetree/bindings/i2c/renesas,rcar-i2c.yaml | 4 +- + .../bindings/i2c/snps,designware-i2c.yaml | 15 ++++ + .../bindings/i2c/xlnx,xps-iic-2.00.a.yaml | 2 +- + drivers/i2c/Kconfig | 13 +++ + drivers/i2c/busses/i2c-bcm2835.c | 2 +- + drivers/i2c/busses/i2c-ljca.c | 9 -- + drivers/i2c/busses/i2c-qcom-geni.c | 77 +++++++++++------ + drivers/i2c/i2c-core-acpi.c | 1 + + drivers/i2c/i2c-core-base.c | 43 ++++++++++ + drivers/i2c/i2c-core-smbus.c | 24 ++++-- + drivers/i2c/i2c-dev.c | 3 + + drivers/i2c/i2c-mux.c | 98 ++++++++++++---------- + drivers/media/i2c/isl7998x.c | 3 +- + include/linux/i2c.h | 36 +++++++- + 15 files changed, 232 insertions(+), 99 deletions(-) +Merging i2c-rust/rust-i2c-next (61ddec70c9bcc i2c: rust: mark I2cAdapter methods as inline) +$ git merge -m Merge branch 'rust-i2c-next' of https://github.com/ikrtn/rust-for-linux i2c-rust/rust-i2c-next +Auto-merging drivers/i2c/busses/i2c-imx-lpi2c.c +CONFLICT (content): Merge conflict in drivers/i2c/busses/i2c-imx-lpi2c.c +Auto-merging drivers/i2c/busses/i2c-imx.c +Auto-merging drivers/i2c/busses/i2c-qcom-geni.c +Auto-merging rust/kernel/i2c.rs +Resolved 'drivers/i2c/busses/i2c-imx-lpi2c.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 8b36d0d274658] Merge branch 'rust-i2c-next' of https://github.com/ikrtn/rust-for-linux +$ git diff -M --stat --summary HEAD^.. + drivers/i2c/busses/i2c-imx-lpi2c.c | 18 ++++-------------- + 1 file changed, 4 insertions(+), 14 deletions(-) +Merging i3c/i3c/next (bf9ec06c90dcd i3c: master: amd: Report direct CCC read length in payload.actual_len) +$ git merge -m Merge branch 'i3c/next' of https://git.kernel.org/pub/scm/linux/kernel/git/i3c/linux.git i3c/i3c/next +Auto-merging drivers/i3c/master/amd-i3c-master.c +Merge made by the 'ort' strategy. + drivers/i3c/device.c | 15 ++- + drivers/i3c/internals.h | 2 + + drivers/i3c/master.c | 91 ++++++++++++-- + drivers/i3c/master/amd-i3c-master.c | 4 +- + drivers/i3c/master/mipi-i3c-hci/cmd.h | 4 +- + drivers/i3c/master/mipi-i3c-hci/cmd_v1.c | 28 ++++- + drivers/i3c/master/mipi-i3c-hci/cmd_v2.c | 4 +- + drivers/i3c/master/mipi-i3c-hci/core.c | 106 +++++++++++++--- + drivers/i3c/master/mipi-i3c-hci/dat.h | 2 + + drivers/i3c/master/mipi-i3c-hci/dat_v1.c | 13 +- + drivers/i3c/master/mipi-i3c-hci/dma.c | 136 +++++++++++++++------ + drivers/i3c/master/mipi-i3c-hci/hci.h | 9 +- + drivers/i3c/master/mipi-i3c-hci/mipi-i3c-hci-pci.c | 5 +- + include/linux/i3c/device.h | 4 + + include/linux/i3c/master.h | 3 + + include/linux/platform_data/mipi-i3c-hci.h | 4 + + 16 files changed, 339 insertions(+), 91 deletions(-) +Merging dmi/dmi-for-next (1afafbaf749d8 firmware/dmi: Include product_family info to modalias) +$ git merge -m Merge branch 'dmi-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jdelvare/staging.git dmi/dmi-for-next +Already up to date. +Merging hwmon-staging/hwmon-next (4781ca52761e6 hwmon:(pmbus/xdpe1a2g7b) Add support for xdpe1a2g7c controller) +$ git merge -m Merge branch 'hwmon-next' of https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git hwmon-staging/hwmon-next +Auto-merging Documentation/devicetree/bindings/vendor-prefixes.yaml +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-class-hwmon | 36 + + .../bindings/hwmon/axiado,ax3000-pwm-fan.yaml | 72 ++ + .../bindings/hwmon/axiado,ax3000-tsadc.yaml | 45 + + .../devicetree/bindings/hwmon/national,lm63.yaml | 55 ++ + .../devicetree/bindings/hwmon/national,lm90.yaml | 28 +- + .../bindings/hwmon/pmbus/adi,max20826.yaml | 102 ++ + .../bindings/hwmon/pmbus/infineon,tda38740.yaml | 55 ++ + .../bindings/hwmon/pmbus/isil,isl68137.yaml | 2 + + .../bindings/hwmon/pmbus/ti,tps25990.yaml | 8 +- + .../devicetree/bindings/hwmon/ti,tmp102.yaml | 16 +- + .../devicetree/bindings/hwmon/ti,tmp401.yaml | 3 + + .../devicetree/bindings/trivial-devices.yaml | 17 +- + .../devicetree/bindings/vendor-prefixes.yaml | 2 + + Documentation/hwmon/arctic_fan_controller.rst | 26 +- + Documentation/hwmon/asus_ec_sensors.rst | 2 + + Documentation/hwmon/asus_rog_ryujin.rst | 1 + + Documentation/hwmon/axiado-pwm-fan.rst | 38 + + Documentation/hwmon/axiado-tsadc.rst | 41 + + Documentation/hwmon/honor-fmi.rst | 32 + + Documentation/hwmon/index.rst | 6 + + Documentation/hwmon/it87.rst | 8 + + Documentation/hwmon/lm63.rst | 22 + + Documentation/hwmon/max20826.rst | 139 +++ + Documentation/hwmon/max34440.rst | 28 +- + Documentation/hwmon/minisforum-um780xtx.rst | 68 ++ + Documentation/hwmon/nct6683.rst | 4 + + Documentation/hwmon/nct6775.rst | 32 + + Documentation/hwmon/sht4x.rst | 17 +- + Documentation/hwmon/socfpga-hwmon.rst | 52 + + Documentation/hwmon/sysfs-interface.rst | 14 +- + Documentation/hwmon/tda38740.rst | 65 ++ + Documentation/hwmon/tps25990.rst | 15 +- + Documentation/hwmon/tps53679.rst | 39 +- + Documentation/hwmon/vexpress.rst | 2 +- + Documentation/hwmon/yogafan.rst | 46 +- + MAINTAINERS | 45 +- + drivers/hwmon/Kconfig | 57 +- + drivers/hwmon/Makefile | 4 + + drivers/hwmon/abituguru.c | 4 +- + drivers/hwmon/acpi_power_meter.c | 4 +- + drivers/hwmon/adm1177.c | 2 +- + drivers/hwmon/adt7x10.c | 2 +- + drivers/hwmon/arctic_fan_controller.c | 22 +- + drivers/hwmon/asus-ec-sensors.c | 4 + + drivers/hwmon/asus_atk0110.c | 4 +- + drivers/hwmon/asus_rog_ryujin.c | 3 + + drivers/hwmon/asus_wmi_sensors.c | 2 +- + drivers/hwmon/axiado-pwm-fan.c | 416 ++++++++ + drivers/hwmon/axiado-tsadc.c | 275 ++++++ + drivers/hwmon/coretemp.c | 2 +- + drivers/hwmon/dell-smm-hwmon.c | 32 + + drivers/hwmon/emc2103.c | 2 +- + drivers/hwmon/hih6130.c | 4 +- + drivers/hwmon/honor-fmi.c | 178 ++++ + drivers/hwmon/hp-wmi-sensors.c | 2 +- + drivers/hwmon/hwmon.c | 6 + + drivers/hwmon/it87.c | 308 +++++- + drivers/hwmon/k10temp.c | 2 + + drivers/hwmon/lm63.c | 426 ++++++-- + drivers/hwmon/lm85.c | 2 +- + drivers/hwmon/lm90.c | 5 + + drivers/hwmon/minisforum-um780xtx.c | 501 ++++++++++ + drivers/hwmon/nct6775-core.c | 116 +++ + drivers/hwmon/nct6775-platform.c | 33 +- + drivers/hwmon/nct6775.h | 4 +- + drivers/hwmon/pmbus/Kconfig | 41 +- + drivers/hwmon/pmbus/Makefile | 2 + + drivers/hwmon/pmbus/ibm-cffps.c | 4 +- + drivers/hwmon/pmbus/isl68137.c | 4 + + drivers/hwmon/pmbus/ltc4286.c | 96 +- + drivers/hwmon/pmbus/max20826.c | 1044 ++++++++++++++++++++ + drivers/hwmon/pmbus/max34440.c | 2 + + drivers/hwmon/pmbus/mp2888.c | 4 +- + drivers/hwmon/pmbus/pmbus.h | 7 + + drivers/hwmon/pmbus/pmbus_core.c | 29 +- + drivers/hwmon/pmbus/tda38740.c | 64 ++ + drivers/hwmon/pmbus/tps25990.c | 249 +++-- + drivers/hwmon/pmbus/tps53679.c | 146 ++- + drivers/hwmon/pmbus/xdpe1a2g7b.c | 10 +- + drivers/hwmon/pt5161l.c | 4 +- + drivers/hwmon/pwm-fan.c | 2 +- + drivers/hwmon/scmi-hwmon.c | 3 +- + drivers/hwmon/sht4x.c | 66 +- + drivers/hwmon/socfpga-hwmon.c | 22 + + drivers/hwmon/spd5118.c | 67 +- + drivers/hwmon/xgene-hwmon.c | 6 +- + drivers/hwmon/yogafan.c | 48 + + include/linux/hwmon.h | 12 + + 88 files changed, 5146 insertions(+), 391 deletions(-) + create mode 100644 Documentation/devicetree/bindings/hwmon/axiado,ax3000-pwm-fan.yaml + create mode 100644 Documentation/devicetree/bindings/hwmon/axiado,ax3000-tsadc.yaml + create mode 100644 Documentation/devicetree/bindings/hwmon/national,lm63.yaml + create mode 100644 Documentation/devicetree/bindings/hwmon/pmbus/adi,max20826.yaml + create mode 100644 Documentation/devicetree/bindings/hwmon/pmbus/infineon,tda38740.yaml + create mode 100644 Documentation/hwmon/axiado-pwm-fan.rst + create mode 100644 Documentation/hwmon/axiado-tsadc.rst + create mode 100644 Documentation/hwmon/honor-fmi.rst + create mode 100644 Documentation/hwmon/max20826.rst + create mode 100644 Documentation/hwmon/minisforum-um780xtx.rst + create mode 100644 Documentation/hwmon/tda38740.rst + create mode 100644 drivers/hwmon/axiado-pwm-fan.c + create mode 100644 drivers/hwmon/axiado-tsadc.c + create mode 100644 drivers/hwmon/honor-fmi.c + create mode 100644 drivers/hwmon/minisforum-um780xtx.c + create mode 100644 drivers/hwmon/pmbus/max20826.c + create mode 100644 drivers/hwmon/pmbus/tda38740.c +$ git am -3 ../patches/0001-drivers-hwmon-sht4x.c-Fix-up-merge-issues.patch +Applying: drivers/hwmon/sht4x.c: Fix up merge issues +Using index info to reconstruct a base tree... +M drivers/hwmon/sht4x.c +Falling back to patching base and 3-way merge... +Auto-merging drivers/hwmon/sht4x.c +CONFLICT (content): Merge conflict in drivers/hwmon/sht4x.c +Recorded preimage for 'drivers/hwmon/sht4x.c' +error: Failed to merge in the changes. +hint: Use 'git am --show-current-patch=diff' to see the failed patch +hint: When you have resolved this problem, run "git am --continue". +hint: If you prefer to skip this patch, run "git am --skip" instead. +hint: To restore the original branch and stop patching, run "git am --abort". +hint: Disable this message with "git config advice.mergeConflict false" +Patch failed at 0001 drivers/hwmon/sht4x.c: Fix up merge issues +Merging jc_docs/docs-next (2d72a4c09867a docs: kdoc: parse context_lock_struct() as struct declaration) +$ git merge -m Merge branch 'docs-next' of git://git.lwn.net/linux.git jc_docs/docs-next +Auto-merging Documentation/ABI/testing/sysfs-bus-pci +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging Documentation/admin-guide/sysctl/kernel.rst +Auto-merging Documentation/filesystems/locking.rst +Auto-merging Documentation/filesystems/porting.rst +Auto-merging Documentation/filesystems/proc.rst +Auto-merging Makefile +Merge made by the 'ort' strategy. + .../ABI/stable/sysfs-driver-firmware-zynqmp | 70 +- + Documentation/ABI/testing/sysfs-bus-pci | 14 +- + Documentation/ABI/testing/sysfs-bus-usb | 3 +- + .../ABI/testing/sysfs-class-firmware-attributes | 21 +- + .../ABI/testing/sysfs-driver-aspeed-uart-routing | 7 +- + Documentation/ABI/testing/sysfs-driver-xdata | 20 +- + .../ABI/testing/sysfs-driver-xilinx-tmr-manager | 14 +- + Documentation/ABI/testing/sysfs-platform-intel-ifs | 6 +- + Documentation/Makefile | 5 + + Documentation/admin-guide/LSM/LoadPin.rst | 10 +- + Documentation/admin-guide/RAS/main.rst | 12 +- + Documentation/admin-guide/cpu-isolation.rst | 4 +- + Documentation/admin-guide/kernel-parameters.txt | 58 +- + Documentation/admin-guide/parport.rst | 2 +- + Documentation/admin-guide/sysctl/kernel.rst | 19 +- + Documentation/admin-guide/sysctl/vm.rst | 2 +- + .../verify-bugs-and-bisect-regressions.rst | 26 +- + Documentation/arch/powerpc/vas-api.rst | 4 +- + Documentation/core-api/cpu_hotplug.rst | 4 +- + Documentation/core-api/debug-objects.rst | 2 +- + Documentation/core-api/dma-attributes.rst | 2 +- + Documentation/core-api/dma-isa-lpc.rst | 11 +- + Documentation/core-api/housekeeping.rst | 2 +- + Documentation/core-api/irq/irq-affinity.rst | 2 +- + Documentation/core-api/irq/irqflags-tracing.rst | 8 +- + Documentation/core-api/maple_tree.rst | 9 +- + Documentation/core-api/real-time/differences.rst | 10 +- + Documentation/core-api/swiotlb.rst | 2 +- + Documentation/core-api/this_cpu_ops.rst | 6 +- + Documentation/core-api/xarray.rst | 4 +- + Documentation/doc-guide/kernel-doc.rst | 7 +- + Documentation/doc-guide/parse-headers.rst | 15 +- + Documentation/driver-api/reset.rst | 2 +- + Documentation/driver-api/serial/serial-rs485.rst | 5 +- + .../locking/cmpxchg-local/arch-support.txt | 2 +- + Documentation/filesystems/fuse/fuse.rst | 8 +- + Documentation/filesystems/locking.rst | 2 +- + Documentation/filesystems/porting.rst | 2 +- + Documentation/filesystems/proc.rst | 10 +- + Documentation/input/devices/yealink.rst | 7 +- + Documentation/input/notifier.rst | 2 +- + Documentation/livepatch/api.rst | 5 +- + Documentation/misc-devices/spear-pcie-gadget.rst | 10 +- + Documentation/process/1.Intro.rst | 2 +- + Documentation/process/2.Process.rst | 20 +- + Documentation/process/3.Early-stage.rst | 2 +- + Documentation/process/5.Posting.rst | 12 +- + Documentation/process/6.Followthrough.rst | 2 +- + Documentation/process/7.AdvancedTopics.rst | 36 +- + Documentation/process/backporting.rst | 18 +- + .../process/embargoed-hardware-issues.rst | 2 +- + Documentation/process/handling-regressions.rst | 2 +- + Documentation/process/howto.rst | 2 +- + Documentation/process/maintainer-pgp-guide.rst | 38 +- + Documentation/process/submitting-patches.rst | 2 +- + Documentation/scheduler/index.rst | 1 + + Documentation/scheduler/sched-preemption.rst | 87 ++ + Documentation/timers/hpet.rst | 33 +- + Documentation/timers/no_hz.rst | 4 +- + .../translations/pt_BR/admin-guide/README.rst | 382 ++++++ + .../translations/pt_BR/admin-guide/devices.rst | 278 +++++ + .../translations/pt_BR/admin-guide/index.rst | 190 +++ + Documentation/translations/pt_BR/index.rst | 10 + + .../translations/pt_BR/process/2.Process.rst | 14 +- + .../translations/pt_BR/process/3.Early-stage.rst | 10 +- + .../translations/pt_BR/process/4.Coding.rst | 6 +- + .../translations/pt_BR/process/5.Posting.rst | 12 +- + .../translations/pt_BR/process/6.Followthrough.rst | 14 +- + .../translations/pt_BR/process/8.Conclusion.rst | 3 +- + .../translations/pt_BR/process/adding-syscalls.rst | 56 +- + .../pt_BR/process/applying-patches.rst | 6 +- + .../translations/pt_BR/process/backporting.rst | 4 +- + .../process/code-of-conduct-interpretation.rst | 2 + + .../translations/pt_BR/process/code-of-conduct.rst | 4 +- + .../pt_BR/process/coding-assistants.rst | 60 + + .../translations/pt_BR/process/coding-style.rst | 1320 ++++++++++++++++++++ + Documentation/translations/pt_BR/process/cve.rst | 4 +- + .../translations/pt_BR/process/debugging/index.rst | 73 ++ + .../pt_BR/process/development-process.rst | 2 + + .../pt_BR/process/embargoed-hardware-issues.rst | 362 ++++++ + Documentation/translations/pt_BR/process/howto.rst | 19 +- + Documentation/translations/pt_BR/process/index.rst | 10 + + .../translations/pt_BR/process/kernel-docs.rst | 2 + + .../pt_BR/process/kernel-enforcement-statement.rst | 163 +++ + .../translations/pt_BR/process/license-rules.rst | 2 + + .../pt_BR/process/maintainer-devicetree.rst | 76 ++ + .../pt_BR/process/maintainer-handbooks.rst | 4 +- + .../pt_BR/process/maintainer-kvm-x86.rst | 10 +- + .../pt_BR/process/maintainer-soc-clean-dts.rst | 5 +- + .../translations/pt_BR/process/maintainer-tip.rst | 847 +++++++++++++ + .../pt_BR/process/management-style.rst | 9 +- + .../translations/pt_BR/process/security-bugs.rst | 25 +- + .../pt_BR/process/stable-api-nonsense.rst | 208 +++ + .../pt_BR/process/submit-checklist.rst | 4 +- + .../pt_BR/process/submitting-patches.rst | 963 ++++++++++++++ + .../pt_BR/process/volatile-considered-harmful.rst | 130 ++ + .../translations/zh_CN/admin-guide/README.rst | 2 +- + .../zh_CN/admin-guide/mm/damon/index.rst | 15 +- + .../zh_CN/admin-guide/mm/damon/lru_sort.rst | 68 +- + .../zh_CN/admin-guide/mm/damon/reclaim.rst | 82 +- + .../zh_CN/admin-guide/mm/damon/start.rst | 63 +- + .../zh_CN/admin-guide/mm/damon/stat.rst | 94 ++ + .../zh_CN/admin-guide/mm/damon/usage.rst | 521 ++++++-- + .../translations/zh_CN/mm/damon/design.rst | 667 +++++++++- + .../translations/zh_CN/networking/driver.rst | 138 ++ + .../translations/zh_CN/networking/index.rst | 10 +- + .../translations/zh_CN/networking/ipv6.rst | 76 ++ + .../translations/zh_CN/networking/secid.rst | 24 + + .../translations/zh_CN/networking/sriov.rst | 34 + + .../translations/zh_CN/networking/team.rst | 16 + + .../zh_CN/process/applying-patches.rst | 390 ++++++ + Documentation/translations/zh_CN/process/howto.rst | 2 +- + Documentation/translations/zh_CN/process/index.rst | 2 +- + Documentation/translations/zh_TW/glossary.rst | 168 +++ + Documentation/translations/zh_TW/index.rst | 5 +- + .../translations/zh_TW/process/1.Intro.rst | 227 ++-- + .../translations/zh_TW/process/2.Process.rst | 325 +++-- + .../translations/zh_TW/process/5.Posting.rst | 235 ++-- + .../zh_TW/process/7.AdvancedTopics.rst | 106 +- + .../translations/zh_TW/process/8.Conclusion.rst | 58 +- + .../process/code-of-conduct-interpretation.rst | 162 ++- + .../translations/zh_TW/process/coding-style.rst | 610 ++++----- + .../translations/zh_TW/process/email-clients.rst | 185 +-- + .../zh_TW/process/embargoed-hardware-issues.rst | 209 ++-- + Documentation/translations/zh_TW/process/howto.rst | 344 ++--- + Documentation/translations/zh_TW/process/index.rst | 105 +- + .../translations/zh_TW/process/license-rules.rst | 201 +-- + .../zh_TW/process/programming-language.rst | 92 +- + .../zh_TW/process/stable-kernel-rules.rst | 244 +++- + .../zh_TW/process/submitting-patches.rst | 568 +++++---- + Documentation/userspace-api/dma-buf-heaps.rst | 2 +- + Documentation/userspace-api/futex2.rst | 4 +- + Documentation/userspace-api/ioctl/ioctl-number.rst | 7 +- + Documentation/userspace-api/iommufd.rst | 2 +- + Documentation/userspace-api/vduse.rst | 33 +- + Documentation/virt/coco/sev-guest.rst | 2 +- + Makefile | 2 +- + include/uapi/linux/gpio.h | 2 +- + security/landlock/syscalls.c | 2 +- + tools/lib/python/kdoc/c_lex.py | 2 +- + tools/lib/python/kdoc/kdoc_parser.py | 11 +- + tools/lib/python/kdoc/xforms_lists.py | 3 +- + tools/unittests/test_kdoc_parser.py | 26 + + 143 files changed, 9947 insertions(+), 2187 deletions(-) + create mode 100644 Documentation/scheduler/sched-preemption.rst + create mode 100644 Documentation/translations/pt_BR/admin-guide/README.rst + create mode 100644 Documentation/translations/pt_BR/admin-guide/devices.rst + create mode 100644 Documentation/translations/pt_BR/admin-guide/index.rst + create mode 100644 Documentation/translations/pt_BR/process/coding-assistants.rst + create mode 100644 Documentation/translations/pt_BR/process/coding-style.rst + create mode 100644 Documentation/translations/pt_BR/process/debugging/index.rst + create mode 100644 Documentation/translations/pt_BR/process/embargoed-hardware-issues.rst + create mode 100644 Documentation/translations/pt_BR/process/kernel-enforcement-statement.rst + create mode 100644 Documentation/translations/pt_BR/process/maintainer-devicetree.rst + create mode 100644 Documentation/translations/pt_BR/process/maintainer-tip.rst + create mode 100644 Documentation/translations/pt_BR/process/stable-api-nonsense.rst + create mode 100644 Documentation/translations/pt_BR/process/submitting-patches.rst + create mode 100644 Documentation/translations/pt_BR/process/volatile-considered-harmful.rst + create mode 100644 Documentation/translations/zh_CN/admin-guide/mm/damon/stat.rst + create mode 100644 Documentation/translations/zh_CN/networking/driver.rst + create mode 100644 Documentation/translations/zh_CN/networking/ipv6.rst + create mode 100644 Documentation/translations/zh_CN/networking/secid.rst + create mode 100644 Documentation/translations/zh_CN/networking/sriov.rst + create mode 100644 Documentation/translations/zh_CN/networking/team.rst + create mode 100644 Documentation/translations/zh_CN/process/applying-patches.rst + create mode 100644 Documentation/translations/zh_TW/glossary.rst +Merging v4l-dvb/next (58348f64125e9 media: ipu6: Support upstream sub-devices without get_frame_desc()) +$ git merge -m Merge branch 'next' of git://linuxtv.org/media-ci/media-pending.git v4l-dvb/next +Auto-merging MAINTAINERS +Auto-merging arch/arm64/boot/dts/freescale/imx8mq-librem5.dtsi +Auto-merging arch/arm64/boot/dts/qcom/qcm6490-fairphone-fp5.dts +Auto-merging arch/arm64/boot/dts/qcom/sc8280xp-lenovo-thinkpad-x13s.dts +Auto-merging arch/arm64/boot/dts/qcom/sdm670-google-common.dtsi +Auto-merging drivers/media/i2c/isl7998x.c +Auto-merging drivers/media/pci/intel/ipu-bridge.c +Auto-merging drivers/media/platform/nxp/imx7-media-csi.c +Auto-merging drivers/media/usb/em28xx/em28xx-video.c +Auto-merging drivers/media/v4l2-core/v4l2-ctrls-core.c +Merge made by the 'ort' strategy. + Documentation/admin-guide/media/vivid.rst | 4 +- + .../bindings/clock/nxp,imx95-blk-ctl.yaml | 71 + + .../bindings/media/fsl,imx95-csi-formatter.yaml | 88 + + .../bindings/media/i2c/himax,hm1246.yaml | 121 + + .../devicetree/bindings/media/i2c/hynix,hi846.yaml | 3 +- + .../devicetree/bindings/media/i2c/ite,it6625.yaml | 175 ++ + .../bindings/media/i2c/ovti,os02g10.yaml | 94 + + .../bindings/media/i2c/ovti,ov08d10.yaml | 3 +- + .../devicetree/bindings/media/i2c/ovti,ov4689.yaml | 3 +- + .../devicetree/bindings/media/i2c/ovti,ov5675.yaml | 3 +- + .../devicetree/bindings/media/i2c/ovti,ov5693.yaml | 5 +- + .../bindings/media/i2c/ovti,ov64a40.yaml | 3 +- + .../devicetree/bindings/media/i2c/sony,imx111.yaml | 3 +- + .../devicetree/bindings/media/i2c/sony,imx355.yaml | 3 +- + .../devicetree/bindings/media/i2c/sony,imx415.yaml | 3 +- + .../devicetree/bindings/media/i2c/st,vd55g1.yaml | 3 +- + .../devicetree/bindings/media/i2c/st,vd56g3.yaml | 3 +- + .../bindings/media/i2c/thine,thp7312.yaml | 3 +- + .../bindings/media/i2c/toshiba,tc358743.txt | 48 - + .../bindings/media/i2c/toshiba,tc358743.yaml | 93 + + .../bindings/media/nxp,imx8mq-mipi-csi2.yaml | 4 +- + .../devicetree/bindings/media/snps,dw-hdmi-rx.yaml | 13 +- + .../bindings/media/video-interface-devices.yaml | 17 +- + Documentation/driver-api/media/tx-rx.rst | 6 +- + .../userspace-api/media/drivers/dcmipp.rst | 14 + + .../userspace-api/media/drivers/index.rst | 1 + + MAINTAINERS | 51 +- + .../nvidia/tegra30-asus-nexus7-grouper-common.dtsi | 3 +- + .../nvidia/tegra30-asus-transformer-common.dtsi | 3 +- + arch/arm/boot/dts/nvidia/tegra30-lg-p895.dts | 4 +- + arch/arm/boot/dts/nvidia/tegra30-lg-x3.dtsi | 3 +- + .../imx8mp-tqma8mpql-mba8mp-ras314-imx219.dtso | 3 +- + arch/arm64/boot/dts/freescale/imx8mq-librem5.dtsi | 3 +- + arch/arm64/boot/dts/qcom/qcm6490-fairphone-fp5.dts | 3 +- + .../dts/qcom/sc8280xp-lenovo-thinkpad-x13s.dts | 3 +- + arch/arm64/boot/dts/qcom/sdm670-google-common.dtsi | 3 +- + .../r8a779g3-sparrow-hawk-camera-j1-imx219.dtso | 3 +- + .../r8a779g3-sparrow-hawk-camera-j1-imx462.dtso | 3 +- + .../r8a779g3-sparrow-hawk-camera-j2-imx219.dtso | 3 +- + .../r8a779g3-sparrow-hawk-camera-j2-imx462.dtso | 3 +- + arch/arm64/boot/dts/rockchip/px30-pp1516.dtsi | 3 +- + .../rockchip/px30-ringneck-haikou-video-demo.dtso | 3 +- + .../boot/dts/rockchip/rk3399-pinephone-pro.dts | 5 +- + .../rk3588-rock-5b-plus-radxa-cam4k-cam0.dtso | 3 +- + .../rk3588-rock-5b-plus-radxa-cam4k-cam1.dtso | 3 +- + drivers/media/cec/platform/meson/ao-cec-g12a.c | 2 +- + .../extron-da-hd-4k-plus/extron-da-hd-4k-plus.c | 4 +- + drivers/media/common/cypress_firmware.c | 2 + + drivers/media/common/saa7146/saa7146_core.c | 12 +- + drivers/media/common/saa7146/saa7146_video.c | 11 + + drivers/media/dvb-frontends/drxd_map_firm.h | 4 +- + drivers/media/dvb-frontends/stv0900_core.c | 2 + + drivers/media/dvb-frontends/stv090x.c | 2 + + drivers/media/i2c/Kconfig | 51 +- + drivers/media/i2c/Makefile | 3 + + drivers/media/i2c/adv7170.c | 1 + + drivers/media/i2c/adv7175.c | 3 +- + drivers/media/i2c/adv7180.c | 5 +- + drivers/media/i2c/adv7183.c | 3 +- + drivers/media/i2c/adv7343.c | 10 +- + drivers/media/i2c/adv7393.c | 10 +- + drivers/media/i2c/adv748x/adv748x-afe.c | 1 + + drivers/media/i2c/adv748x/adv748x-core.c | 13 +- + drivers/media/i2c/adv748x/adv748x-csi2.c | 1 + + drivers/media/i2c/adv748x/adv748x-hdmi.c | 1 + + drivers/media/i2c/adv7511-v4l2.c | 1 + + drivers/media/i2c/adv7604.c | 7 +- + drivers/media/i2c/adv7842.c | 11 +- + drivers/media/i2c/ak881x.c | 2 +- + drivers/media/i2c/alvium-csi2.c | 4 + + drivers/media/i2c/ar0521.c | 2 +- + drivers/media/i2c/ccs/ccs-core.c | 12 +- + drivers/media/i2c/ccs/ccs-data.h | 4 +- + drivers/media/i2c/cvs/Kconfig | 1 + + drivers/media/i2c/cvs/core.c | 30 +- + drivers/media/i2c/cvs/v4l2.c | 96 +- + drivers/media/i2c/cx25840/cx25840-core.c | 39 +- + drivers/media/i2c/ds90ub913.c | 1 + + drivers/media/i2c/ds90ub953.c | 1 + + drivers/media/i2c/ds90ub960.c | 1 + + drivers/media/i2c/et8ek8/et8ek8_driver.c | 1 + + drivers/media/i2c/gc0308.c | 1 + + drivers/media/i2c/gc0310.c | 2 +- + drivers/media/i2c/gc05a2.c | 4 +- + drivers/media/i2c/gc08a3.c | 4 +- + drivers/media/i2c/gc2145.c | 2 + + drivers/media/i2c/hi556.c | 2 + + drivers/media/i2c/hi846.c | 2 + + drivers/media/i2c/hi847.c | 3 +- + drivers/media/i2c/hm1246.c | 1287 +++++++++++ + drivers/media/i2c/imx111.c | 1 + + drivers/media/i2c/imx208.c | 1 + + drivers/media/i2c/imx214.c | 4 +- + drivers/media/i2c/imx219.c | 42 +- + drivers/media/i2c/imx258.c | 2 + + drivers/media/i2c/imx274.c | 3 + + drivers/media/i2c/imx283.c | 2 + + drivers/media/i2c/imx290.c | 4 +- + drivers/media/i2c/imx296.c | 7 +- + drivers/media/i2c/imx319.c | 1 + + drivers/media/i2c/imx334.c | 135 +- + drivers/media/i2c/imx335.c | 4 +- + drivers/media/i2c/imx355.c | 4 +- + drivers/media/i2c/imx412.c | 3 +- + drivers/media/i2c/imx415.c | 4 +- + drivers/media/i2c/imx471.c | 6 +- + drivers/media/i2c/imx678.c | 2 +- + drivers/media/i2c/isl7998x.c | 1 + + drivers/media/i2c/it6625.c | 2391 ++++++++++++++++++++ + drivers/media/i2c/lt6911uxe.c | 5 +- + drivers/media/i2c/max9286.c | 1 + + drivers/media/i2c/max96714.c | 1 + + drivers/media/i2c/max96717.c | 1 + + drivers/media/i2c/ml86v7667.c | 1 - + drivers/media/i2c/mt9m001.c | 9 +- + drivers/media/i2c/mt9m111.c | 3 + + drivers/media/i2c/mt9m114.c | 6 + + drivers/media/i2c/mt9p031.c | 3 + + drivers/media/i2c/mt9t112.c | 3 + + drivers/media/i2c/mt9v011.c | 1 + + drivers/media/i2c/mt9v032.c | 3 + + drivers/media/i2c/mt9v111.c | 1 + + drivers/media/i2c/og01a1b.c | 8 +- + drivers/media/i2c/og0ve1b.c | 3 +- + drivers/media/i2c/os02g10.c | 934 ++++++++ + drivers/media/i2c/os05b10.c | 2 + + drivers/media/i2c/ov01a10.c | 3 + + drivers/media/i2c/ov02a10.c | 3 +- + drivers/media/i2c/ov02c10.c | 1 + + drivers/media/i2c/ov02e10.c | 1 + + drivers/media/i2c/ov08d10.c | 1 + + drivers/media/i2c/ov08x40.c | 1 + + drivers/media/i2c/ov13858.c | 1 + + drivers/media/i2c/ov13b10.c | 1 + + drivers/media/i2c/ov2640.c | 2 + + drivers/media/i2c/ov2659.c | 1 + + drivers/media/i2c/ov2680.c | 3 + + drivers/media/i2c/ov2685.c | 2 + + drivers/media/i2c/ov2732.c | 4 +- + drivers/media/i2c/ov2735.c | 5 +- + drivers/media/i2c/ov2740.c | 1 + + drivers/media/i2c/ov4689.c | 2 + + drivers/media/i2c/ov5640.c | 2 + + drivers/media/i2c/ov5645.c | 4 +- + drivers/media/i2c/ov5647.c | 6 + + drivers/media/i2c/ov5648.c | 5 +- + drivers/media/i2c/ov5670.c | 2 + + drivers/media/i2c/ov5675.c | 2 + + drivers/media/i2c/ov5693.c | 28 + + drivers/media/i2c/ov5695.c | 1 + + drivers/media/i2c/ov6211.c | 3 +- + drivers/media/i2c/ov64a40.c | 2 + + drivers/media/i2c/ov7251.c | 4 +- + drivers/media/i2c/ov7670.c | 1 + + drivers/media/i2c/ov772x.c | 4 +- + drivers/media/i2c/ov7740.c | 1 + + drivers/media/i2c/ov8856.c | 33 +- + drivers/media/i2c/ov8858.c | 3 +- + drivers/media/i2c/ov8865.c | 2 + + drivers/media/i2c/ov9282.c | 4 +- + drivers/media/i2c/ov9640.c | 2 + + drivers/media/i2c/ov9650.c | 1 + + drivers/media/i2c/ov9734.c | 1 + + drivers/media/i2c/rdacm20.c | 1 - + drivers/media/i2c/rdacm21.c | 1 - + drivers/media/i2c/rj54n1cb0c.c | 3 + + drivers/media/i2c/s5c73m3/s5c73m3-core.c | 2 + + drivers/media/i2c/s5k3m5.c | 4 +- + drivers/media/i2c/s5k5baf.c | 3 + + drivers/media/i2c/s5k6a3.c | 1 + + drivers/media/i2c/s5kjn1.c | 4 +- + drivers/media/i2c/saa6752hs.c | 1 + + drivers/media/i2c/saa7115.c | 1 + + drivers/media/i2c/saa717x.c | 1 + + drivers/media/i2c/st-mipid02.c | 1 + + drivers/media/i2c/t4ka3.c | 4 +- + drivers/media/i2c/tc358743.c | 6 +- + drivers/media/i2c/tc358746.c | 1 + + drivers/media/i2c/tda1997x.c | 3 +- + drivers/media/i2c/thp7312.c | 1 + + drivers/media/i2c/ths7303.c | 10 +- + drivers/media/i2c/ths8200.c | 14 +- + drivers/media/i2c/ths8200_regs.h | 14 +- + drivers/media/i2c/tvp514x.c | 1 + + drivers/media/i2c/tvp5150.c | 3 +- + drivers/media/i2c/tvp7002.c | 1 + + drivers/media/i2c/tw9900.c | 1 + + drivers/media/i2c/tw9910.c | 2 + + drivers/media/i2c/vd55g1.c | 4 +- + drivers/media/i2c/vd56g3.c | 4 +- + drivers/media/i2c/vgxy61.c | 4 +- + drivers/media/i2c/wm8739.c | 2 +- + drivers/media/pci/bt8xx/bttv-driver.c | 8 +- + drivers/media/pci/cobalt/cobalt-driver.c | 8 +- + drivers/media/pci/cobalt/cobalt-v4l2.c | 10 +- + drivers/media/pci/cx18/cx18-av-core.c | 1 + + drivers/media/pci/cx18/cx18-controls.c | 2 +- + drivers/media/pci/cx18/cx18-driver.c | 9 +- + drivers/media/pci/cx18/cx18-ioctl.c | 2 +- + drivers/media/pci/cx18/cx18-queue.c | 5 +- + drivers/media/pci/cx18/cx23418.h | 2 +- + drivers/media/pci/cx23885/cx23885-core.c | 4 + + drivers/media/pci/cx23885/cx23885-dvb.c | 14 +- + drivers/media/pci/cx23885/cx23885-video.c | 4 +- + drivers/media/pci/cx88/cx88-input.c | 22 +- + drivers/media/pci/cx88/cx88-mpeg.c | 9 +- + drivers/media/pci/cx88/cx88-video.c | 8 +- + drivers/media/pci/hws/hws.h | 4 +- + drivers/media/pci/hws/hws_pci.c | 6 +- + drivers/media/pci/hws/hws_reg.h | 10 +- + drivers/media/pci/hws/hws_v4l2_ioctl.c | 5 +- + drivers/media/pci/hws/hws_video.c | 6 +- + drivers/media/pci/intel/ipu-bridge.c | 148 +- + drivers/media/pci/intel/ipu3/ipu3-cio2.c | 1 + + drivers/media/pci/intel/ipu6/Kconfig | 9 + + drivers/media/pci/intel/ipu6/Makefile | 10 +- + drivers/media/pci/intel/ipu6/ipu6-bus.h | 6 +- + drivers/media/pci/intel/ipu6/ipu6-buttress.c | 588 +++-- + drivers/media/pci/intel/ipu6/ipu6-buttress.h | 50 +- + drivers/media/pci/intel/ipu6/ipu6-cpd.c | 201 +- + drivers/media/pci/intel/ipu6/ipu6-cpd.h | 43 + + drivers/media/pci/intel/ipu6/ipu6-dma.c | 24 +- + drivers/media/pci/intel/ipu6/ipu6-dma.h | 2 + + drivers/media/pci/intel/ipu6/ipu6-fw-isys.c | 586 ++++- + drivers/media/pci/intel/ipu6/ipu6-fw-isys.h | 48 +- + drivers/media/pci/intel/ipu6/ipu6-isys-csi2.c | 584 ++++- + drivers/media/pci/intel/ipu6/ipu6-isys-csi2.h | 18 +- + drivers/media/pci/intel/ipu6/ipu6-isys-queue.c | 213 +- + drivers/media/pci/intel/ipu6/ipu6-isys-queue.h | 6 +- + drivers/media/pci/intel/ipu6/ipu6-isys-subdev.c | 23 +- + drivers/media/pci/intel/ipu6/ipu6-isys-subdev.h | 3 + + drivers/media/pci/intel/ipu6/ipu6-isys-video.c | 772 ++----- + drivers/media/pci/intel/ipu6/ipu6-isys-video.h | 62 +- + drivers/media/pci/intel/ipu6/ipu6-isys.c | 607 ++--- + drivers/media/pci/intel/ipu6/ipu6-isys.h | 105 +- + drivers/media/pci/intel/ipu6/ipu6-mmu-hw.c | 296 +++ + drivers/media/pci/intel/ipu6/ipu6-mmu.c | 133 +- + drivers/media/pci/intel/ipu6/ipu6-mmu.h | 158 +- + .../pci/intel/ipu6/ipu6-platform-buttress-regs.h | 111 +- + drivers/media/pci/intel/ipu6/ipu6.c | 574 +++-- + drivers/media/pci/intel/ipu6/ipu6.h | 196 +- + drivers/media/pci/intel/ipu6/ipu7-boot.c | 405 ++++ + drivers/media/pci/intel/ipu6/ipu7-boot.h | 46 + + drivers/media/pci/intel/ipu6/ipu7-fw-com.c | 74 + + drivers/media/pci/intel/ipu6/ipu7-fw-com.h | 53 + + drivers/media/pci/intel/ipu6/ipu7-fw-isys.c | 796 +++++++ + drivers/media/pci/intel/ipu6/ipu7-fw-isys.h | 296 +++ + drivers/media/pci/intel/ipu6/ipu7-isys-csi-phy.c | 1072 +++++++++ + drivers/media/pci/intel/ipu6/ipu7-isys-csi-phy.h | 16 + + drivers/media/pci/intel/ipu6/ipu7-isys-csi2-regs.h | 1189 ++++++++++ + drivers/media/pci/intel/ipu6/ipu7-mmu-hw.c | 858 +++++++ + drivers/media/pci/intel/ipu6/ipu7-mmu-hw.h | 254 +++ + drivers/media/pci/intel/ipu6/ipu7-platform-regs.h | 32 + + drivers/media/pci/intel/ivsc/mei_ace.c | 4 +- + drivers/media/pci/intel/ivsc/mei_csi.c | 8 +- + drivers/media/pci/ivtv/ivtv-controls.c | 2 +- + drivers/media/pci/ivtv/ivtv-ioctl.c | 2 +- + drivers/media/pci/saa7134/saa7134-core.c | 8 +- + drivers/media/pci/saa7134/saa7134-empress.c | 4 +- + drivers/media/pci/saa7134/saa7134-input.c | 7 +- + drivers/media/pci/saa7146/mxb.c | 4 +- + drivers/media/pci/saa7164/saa7164-core.c | 32 +- + drivers/media/pci/saa7164/saa7164.h | 1 - + drivers/media/pci/tw68/tw68-core.c | 10 +- + drivers/media/platform/amd/isp4/isp4_subdev.c | 1 + + drivers/media/platform/amd/isp4/isp4_video.c | 2 +- + .../media/platform/amlogic/c3/isp/c3-isp-core.c | 1 + + .../media/platform/amlogic/c3/isp/c3-isp-resizer.c | 3 + + .../amlogic/c3/mipi-adapter/c3-mipi-adap.c | 1 + + .../platform/amlogic/c3/mipi-csi2/c3-mipi-csi2.c | 1 + + drivers/media/platform/arm/mali-c55/mali-c55-isp.c | 3 + + .../media/platform/arm/mali-c55/mali-c55-resizer.c | 15 +- + drivers/media/platform/arm/mali-c55/mali-c55-tpg.c | 1 + + drivers/media/platform/aspeed/aspeed-video.c | 12 +- + drivers/media/platform/atmel/atmel-isi.c | 11 +- + drivers/media/platform/broadcom/bcm2835-unicam.c | 1 + + drivers/media/platform/cadence/cdns-csi2rx.c | 5 + + drivers/media/platform/cadence/cdns-csi2tx.c | 1 + + drivers/media/platform/intel/pxa_camera.c | 6 +- + drivers/media/platform/m2m-deinterlace.c | 2 +- + drivers/media/platform/marvell/Kconfig | 1 + + drivers/media/platform/marvell/cafe-driver.c | 2 +- + drivers/media/platform/marvell/mcam-core.c | 24 +- + drivers/media/platform/marvell/mmp-driver.c | 5 +- + drivers/media/platform/mediatek/mdp/mtk_mdp_ipi.h | 2 +- + .../mediatek/vcodec/decoder/vdec_ipi_msg.h | 4 +- + .../media/platform/microchip/microchip-csi2dc.c | 2 + + .../media/platform/microchip/microchip-isc-base.c | 82 +- + .../media/platform/microchip/microchip-isc-clk.c | 7 +- + .../media/platform/microchip/microchip-isc-regs.h | 16 +- + .../platform/microchip/microchip-isc-scaler.c | 2 + + drivers/media/platform/microchip/microchip-isc.h | 5 +- + .../platform/microchip/microchip-sama5d2-isc.c | 48 +- + .../platform/microchip/microchip-sama7g5-isc.c | 48 +- + drivers/media/platform/nuvoton/npcm-video.c | 18 +- + drivers/media/platform/nxp/Kconfig | 16 + + drivers/media/platform/nxp/Makefile | 1 + + drivers/media/platform/nxp/imx-jpeg/mxc-jpeg.c | 23 +- + drivers/media/platform/nxp/imx-mipi-csis.c | 3 +- + drivers/media/platform/nxp/imx7-media-csi.c | 1 + + .../platform/nxp/imx8-isi/imx8-isi-crossbar.c | 1 + + .../media/platform/nxp/imx8-isi/imx8-isi-pipe.c | 3 + + drivers/media/platform/nxp/imx8mq-mipi-csi2.c | 1 + + drivers/media/platform/nxp/imx95-csi-formatter.c | 759 +++++++ + drivers/media/platform/qcom/camss/camss-csid.c | 3 +- + drivers/media/platform/qcom/camss/camss-csiphy.c | 3 +- + drivers/media/platform/qcom/camss/camss-ispif.c | 3 +- + drivers/media/platform/qcom/camss/camss-tpg.c | 3 +- + drivers/media/platform/qcom/camss/camss-vfe.c | 12 +- + drivers/media/platform/qcom/camss/camss.c | 17 +- + drivers/media/platform/raspberrypi/rp1-cfe/csi2.c | 1 + + .../media/platform/raspberrypi/rp1-cfe/pisp-fe.c | 1 + + drivers/media/platform/renesas/rcar-csi2.c | 385 +++- + drivers/media/platform/renesas/rcar-isp/csisp.c | 228 +- + .../media/platform/renesas/rcar-vin/rcar-core.c | 27 +- + drivers/media/platform/renesas/rcar-vin/rcar-dma.c | 2 +- + drivers/media/platform/renesas/rcar_drif.c | 2 +- + drivers/media/platform/renesas/renesas-ceu.c | 7 +- + .../media/platform/renesas/rzg2l-cru/rzg2l-core.c | 3 +- + .../media/platform/renesas/rzg2l-cru/rzg2l-cru.h | 2 +- + .../media/platform/renesas/rzg2l-cru/rzg2l-csi2.c | 3 +- + .../media/platform/renesas/rzg2l-cru/rzg2l-ip.c | 3 +- + .../media/platform/renesas/rzg2l-cru/rzg2l-video.c | 13 +- + .../platform/renesas/rzv2h-ivc/rzv2h-ivc-subdev.c | 1 + + drivers/media/platform/renesas/sh_vou.c | 6 +- + drivers/media/platform/renesas/vsp1/vsp1_brx.c | 3 + + drivers/media/platform/renesas/vsp1/vsp1_dl.c | 2 +- + drivers/media/platform/renesas/vsp1/vsp1_drm.c | 18 +- + drivers/media/platform/renesas/vsp1/vsp1_entity.c | 4 +- + drivers/media/platform/renesas/vsp1/vsp1_entity.h | 1 + + drivers/media/platform/renesas/vsp1/vsp1_histo.c | 3 + + drivers/media/platform/renesas/vsp1/vsp1_hsit.c | 1 + + drivers/media/platform/renesas/vsp1/vsp1_rwpf.c | 3 + + drivers/media/platform/renesas/vsp1/vsp1_sru.c | 1 + + drivers/media/platform/renesas/vsp1/vsp1_uds.c | 1 + + drivers/media/platform/renesas/vsp1/vsp1_uif.c | 2 + + drivers/media/platform/renesas/vsp1/vsp1_vspx.c | 3 +- + .../platform/rockchip/rkcif/rkcif-interface.c | 3 + + .../media/platform/rockchip/rkisp1/rkisp1-csi.c | 1 + + .../media/platform/rockchip/rkisp1/rkisp1-dev.c | 4 +- + .../media/platform/rockchip/rkisp1/rkisp1-isp.c | 3 + + .../platform/rockchip/rkisp1/rkisp1-resizer.c | 3 + + .../platform/samsung/exynos4-is/fimc-capture.c | 8 +- + .../media/platform/samsung/exynos4-is/fimc-isp.c | 1 + + .../media/platform/samsung/exynos4-is/fimc-lite.c | 3 + + .../media/platform/samsung/exynos4-is/mipi-csis.c | 1 + + .../platform/samsung/s3c-camif/camif-capture.c | 3 + + .../media/platform/samsung/s3c-camif/camif-core.c | 2 +- + drivers/media/platform/st/stm32/stm32-csi.c | 6 +- + drivers/media/platform/st/stm32/stm32-dcmi.c | 31 +- + .../media/platform/st/stm32/stm32-dcmipp/Makefile | 3 +- + .../st/stm32/stm32-dcmipp/dcmipp-byteproc.c | 30 +- + .../{dcmipp-bytecap.c => dcmipp-capture.c} | 600 +++-- + .../platform/st/stm32/stm32-dcmipp/dcmipp-common.h | 99 +- + .../platform/st/stm32/stm32-dcmipp/dcmipp-core.c | 124 +- + .../platform/st/stm32/stm32-dcmipp/dcmipp-input.c | 127 +- + .../platform/st/stm32/stm32-dcmipp/dcmipp-isp.c | 493 ++++ + .../st/stm32/stm32-dcmipp/dcmipp-pixelcommon.c | 181 ++ + .../st/stm32/stm32-dcmipp/dcmipp-pixelcommon.h | 42 + + .../st/stm32/stm32-dcmipp/dcmipp-pixelproc.c | 942 ++++++++ + .../media/platform/sunxi/sun4i-csi/sun4i_v4l2.c | 1 + + .../platform/sunxi/sun6i-csi/sun6i_csi_bridge.c | 156 +- + .../platform/sunxi/sun6i-csi/sun6i_csi_bridge.h | 9 - + .../platform/sunxi/sun6i-csi/sun6i_csi_capture.c | 27 +- + .../sunxi/sun6i-mipi-csi2/sun6i_mipi_csi2.c | 108 +- + .../sunxi/sun6i-mipi-csi2/sun6i_mipi_csi2.h | 2 - + .../sun8i-a83t-mipi-csi2/sun8i_a83t_mipi_csi2.c | 113 +- + .../sun8i-a83t-mipi-csi2/sun8i_a83t_mipi_csi2.h | 2 - + drivers/media/platform/synopsys/dw-mipi-csi2rx.c | 1 + + .../media/platform/synopsys/hdmirx/snps_hdmirx.c | 423 +++- + .../media/platform/synopsys/hdmirx/snps_hdmirx.h | 8 + + drivers/media/platform/ti/Kconfig | 11 - + drivers/media/platform/ti/am437x/am437x-vpfe.c | 2 +- + drivers/media/platform/ti/cal/cal-camerarx.c | 1 + + drivers/media/platform/ti/cal/cal-video.c | 19 +- + drivers/media/platform/ti/cal/cal.c | 10 +- + drivers/media/platform/ti/davinci/vpif_capture.c | 5 +- + drivers/media/platform/ti/davinci/vpif_display.c | 3 + + .../media/platform/ti/j721e-csi2rx/j721e-csi2rx.c | 25 + + drivers/media/platform/ti/omap3isp/ispccdc.c | 5 +- + drivers/media/platform/ti/omap3isp/ispccp2.c | 3 +- + drivers/media/platform/ti/omap3isp/ispcsi2.c | 3 +- + drivers/media/platform/ti/omap3isp/isppreview.c | 5 +- + drivers/media/platform/ti/omap3isp/ispresizer.c | 5 +- + drivers/media/platform/ti/omap3isp/ispvideo.c | 4 +- + drivers/media/platform/ti/vpe/vip.c | 4 +- + drivers/media/platform/via/via-camera.c | 4 +- + drivers/media/platform/video-mux.c | 1 + + drivers/media/platform/xilinx/xilinx-csi2rxss.c | 1 + + drivers/media/platform/xilinx/xilinx-tpg.c | 1 + + drivers/media/radio/si4713/radio-usb-si4713.c | 4 +- + drivers/media/radio/si4713/si4713.c | 2 +- + drivers/media/rc/bpf-lirc.c | 18 +- + drivers/media/rc/ene_ir.c | 4 +- + drivers/media/rc/fintek-cir.c | 13 - + drivers/media/rc/fintek-cir.h | 22 - + drivers/media/rc/imon.c | 175 +- + drivers/media/rc/ir-hix5hd2.c | 5 +- + drivers/media/rc/ir-mce_kbd-decoder.c | 5 +- + drivers/media/rc/ir_toy.c | 4 +- + drivers/media/rc/ite-cir.c | 1 - + drivers/media/rc/ite-cir.h | 1 - + drivers/media/rc/lirc_dev.c | 5 + + drivers/media/rc/mceusb.c | 1 - + drivers/media/rc/meson-ir-tx.c | 20 +- + drivers/media/rc/nuvoton-cir.h | 3 - + drivers/media/rc/rc-ir-raw.c | 86 +- + drivers/media/rc/rc-loopback.c | 5 - + drivers/media/rc/rc-main.c | 206 +- + drivers/media/rc/redrat3.c | 32 +- + drivers/media/rc/serial_ir.c | 2 +- + drivers/media/rc/streamzap.c | 1 + + drivers/media/rc/sunxi-cir.c | 2 +- + drivers/media/spi/Kconfig | 4 - + drivers/media/test-drivers/vicodec/codec-fwht.h | 2 +- + drivers/media/test-drivers/vicodec/vicodec-core.c | 14 +- + drivers/media/test-drivers/vidtv/vidtv_bridge.c | 75 +- + drivers/media/test-drivers/vidtv/vidtv_demod.c | 9 - + drivers/media/test-drivers/vim2m.c | 33 +- + drivers/media/test-drivers/vimc/vimc-debayer.c | 1 + + drivers/media/test-drivers/vimc/vimc-scaler.c | 3 + + drivers/media/test-drivers/vimc/vimc-sensor.c | 9 +- + drivers/media/test-drivers/vivid/vivid-cec.c | 10 +- + drivers/media/test-drivers/vivid/vivid-core.c | 1 + + drivers/media/test-drivers/vivid/vivid-vid-cap.c | 13 + + drivers/media/tuners/tda18250.c | 3 +- + drivers/media/usb/au0828/au0828-core.c | 4 + + drivers/media/usb/au0828/au0828-dvb.c | 20 +- + drivers/media/usb/cx231xx/cx231xx-417.c | 2 +- + drivers/media/usb/cx231xx/cx231xx-audio.c | 15 +- + drivers/media/usb/cx231xx/cx231xx-cards.c | 2 +- + drivers/media/usb/cx231xx/cx231xx-video.c | 4 +- + drivers/media/usb/cx231xx/cx231xx.h | 2 +- + drivers/media/usb/dvb-usb-v2/mxl111sf-i2c.c | 4 +- + drivers/media/usb/dvb-usb/cxusb-analog.c | 6 +- + drivers/media/usb/dvb-usb/dib0700_core.c | 2 +- + drivers/media/usb/dvb-usb/dvb-usb-firmware.c | 2 + + drivers/media/usb/em28xx/em28xx-camera.c | 2 +- + drivers/media/usb/em28xx/em28xx-video.c | 4 + + drivers/media/usb/go7007/go7007-driver.c | 5 +- + drivers/media/usb/go7007/go7007-usb.c | 8 + + drivers/media/usb/go7007/go7007-v4l2.c | 2 +- + drivers/media/usb/go7007/s2250-board.c | 1 + + drivers/media/usb/gspca/gspca.c | 3 +- + drivers/media/usb/gspca/ov519.c | 2 +- + drivers/media/usb/gspca/w996Xcf.c | 2 +- + drivers/media/usb/hackrf/hackrf.c | 7 +- + drivers/media/usb/pvrusb2/pvrusb2-hdw.c | 14 +- + drivers/media/usb/usbtv/usbtv-core.c | 2 - + drivers/media/usb/usbtv/usbtv-video.c | 4 + + drivers/media/v4l2-core/v4l2-common.c | 21 +- + drivers/media/v4l2-core/v4l2-ctrls-core.c | 19 +- + drivers/media/v4l2-core/v4l2-mc.c | 5 +- + drivers/media/v4l2-core/v4l2-subdev.c | 163 +- + drivers/platform/x86/intel/int3472/discrete.c | 81 +- + drivers/platform/x86/intel/int3472/tps68470.c | 2 +- + drivers/staging/media/atomisp/i2c/atomisp-gc2235.c | 1 + + drivers/staging/media/atomisp/i2c/atomisp-ov2722.c | 1 + + drivers/staging/media/atomisp/pci/atomisp_cmd.c | 16 +- + drivers/staging/media/atomisp/pci/atomisp_csi2.c | 1 + + drivers/staging/media/atomisp/pci/atomisp_subdev.c | 3 + + drivers/staging/media/atomisp/pci/atomisp_v4l2.c | 8 +- + .../pci/isp/kernels/s3a/s3a_1.0/ia_css_s3a_types.h | 4 +- + drivers/staging/media/av7110/av7110.c | 2 +- + drivers/staging/media/av7110/av7110_ir.c | 2 - + drivers/staging/media/av7110/sp8870.c | 9 +- + drivers/staging/media/imx/imx-ic-prp.c | 2 +- + drivers/staging/media/imx/imx-ic-prpencvf.c | 4 +- + drivers/staging/media/imx/imx-media-capture.c | 1 - + drivers/staging/media/imx/imx-media-csc-scaler.c | 3 +- + drivers/staging/media/imx/imx-media-csi.c | 3 + + drivers/staging/media/imx/imx-media-dev.c | 1 - + drivers/staging/media/imx/imx-media-vdic.c | 1 + + drivers/staging/media/imx/imx6-mipi-csi2.c | 1 + + drivers/staging/media/ipu3/ipu3-css.c | 1 - + drivers/staging/media/ipu3/ipu3-v4l2.c | 3 + + drivers/staging/media/ipu7/TODO | 28 +- + drivers/staging/media/ipu7/ipu7-isys-csi2.c | 2 + + drivers/staging/media/ipu7/ipu7-isys-subdev.c | 1 + + drivers/staging/media/ipu7/ipu7-isys-subdev.h | 1 + + drivers/staging/media/ipu7/ipu7-isys.c | 1 + + drivers/staging/media/ipu7/ipu7.c | 7 + + drivers/staging/media/max96712/max96712.c | 1 - + .../staging/media/sunxi/sun6i-isp/sun6i_isp_proc.c | 1 + + drivers/staging/media/tegra-video/csi.c | 1 + + drivers/staging/media/tegra-video/vi.c | 102 +- + .../dt-bindings/media/video-interface-devices.h | 13 + + include/linux/platform_data/x86/int3472.h | 2 + + include/linux/property.h | 5 + + include/media/ipu-bridge.h | 52 +- + include/media/ipu6-pci-table.h | 4 + + include/media/rc-map.h | 2 - + include/media/v4l2-common.h | 75 +- + include/media/v4l2-ctrls.h | 6 + + include/media/v4l2-dv-timings.h | 14 + + include/media/v4l2-subdev.h | 34 +- + include/uapi/linux/it6625.h | 25 + + include/uapi/linux/media/st/dcmipp_config.h | 16 + + include/uapi/linux/v4l2-controls.h | 21 + + 499 files changed, 20234 insertions(+), 4318 deletions(-) + create mode 100644 Documentation/devicetree/bindings/media/fsl,imx95-csi-formatter.yaml + create mode 100644 Documentation/devicetree/bindings/media/i2c/himax,hm1246.yaml + create mode 100644 Documentation/devicetree/bindings/media/i2c/ite,it6625.yaml + create mode 100644 Documentation/devicetree/bindings/media/i2c/ovti,os02g10.yaml + delete mode 100644 Documentation/devicetree/bindings/media/i2c/toshiba,tc358743.txt + create mode 100644 Documentation/devicetree/bindings/media/i2c/toshiba,tc358743.yaml + create mode 100644 Documentation/userspace-api/media/drivers/dcmipp.rst + create mode 100644 drivers/media/i2c/hm1246.c + create mode 100644 drivers/media/i2c/it6625.c + create mode 100644 drivers/media/i2c/os02g10.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu6-mmu-hw.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-boot.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-boot.h + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-fw-com.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-fw-com.h + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-fw-isys.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-fw-isys.h + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-isys-csi-phy.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-isys-csi-phy.h + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-isys-csi2-regs.h + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-mmu-hw.c + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-mmu-hw.h + create mode 100644 drivers/media/pci/intel/ipu6/ipu7-platform-regs.h + create mode 100644 drivers/media/platform/nxp/imx95-csi-formatter.c + rename drivers/media/platform/st/stm32/stm32-dcmipp/{dcmipp-bytecap.c => dcmipp-capture.c} (53%) + create mode 100644 drivers/media/platform/st/stm32/stm32-dcmipp/dcmipp-isp.c + create mode 100644 drivers/media/platform/st/stm32/stm32-dcmipp/dcmipp-pixelcommon.c + create mode 100644 drivers/media/platform/st/stm32/stm32-dcmipp/dcmipp-pixelcommon.h + create mode 100644 drivers/media/platform/st/stm32/stm32-dcmipp/dcmipp-pixelproc.c + create mode 100644 include/dt-bindings/media/video-interface-devices.h + create mode 100644 include/uapi/linux/it6625.h + create mode 100644 include/uapi/linux/media/st/dcmipp_config.h +Merging v4l-dvb-next/master (adc218676eef2 Linux 6.12) +$ git merge -m Merge branch 'master' of git://linuxtv.org/mchehab/media-next.git v4l-dvb-next/master +Already up to date. +Merging pm/linux-next (114920a3e8292 Merge branch 'thermal-drivers' into linux-next) +$ git merge -m Merge branch 'linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rafael/linux-pm.git pm/linux-next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging arch/arm64/kernel/topology.c +Auto-merging sound/soc/intel/boards/bytcr_rt5651.c +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-devices-system-cpu | 14 +- + Documentation/admin-guide/kernel-parameters.txt | 7 - + Documentation/admin-guide/pm/amd-pstate.rst | 6 - + Documentation/admin-guide/pm/cpuidle.rst | 2 +- + .../driver-api/thermal/cpu-idle-cooling.rst | 2 +- + Documentation/firmware-guide/acpi/apei/einj.rst | 6 +- + Documentation/power/runtime_pm.rst | 379 +--- + Documentation/power/userland-swsusp.rst | 4 +- + arch/arm64/kernel/topology.c | 14 +- + arch/x86/kernel/acpi/cppc.c | 14 + + drivers/acpi/acpi_extlog.c | 64 +- + drivers/acpi/acpi_mrrm.c | 14 +- + drivers/acpi/acpi_pcc.c | 55 +- + drivers/acpi/acpi_video.c | 43 +- + drivers/acpi/acpica/dsfield.c | 3 +- + drivers/acpi/acpica/exfield.c | 11 +- + drivers/acpi/apei/ghes.c | 92 +- + drivers/acpi/apei/ghes_helpers.c | 18 +- + drivers/acpi/arm64/amba.c | 3 +- + drivers/acpi/battery.c | 131 +- + drivers/acpi/bus.c | 8 +- + drivers/acpi/button.c | 12 + + drivers/acpi/cppc_acpi.c | 2223 +++++++++++++++++--- + drivers/acpi/device_pm.c | 77 +- + drivers/acpi/fan.h | 2 +- + drivers/acpi/fan_core.c | 147 +- + drivers/acpi/fan_hwmon.c | 2 +- + drivers/acpi/glue.c | 145 +- + drivers/acpi/internal.h | 1 + + drivers/acpi/numa/hmat.c | 2 +- + drivers/acpi/numa/srat.c | 2 +- + drivers/acpi/osl.c | 2 +- + drivers/acpi/pfr_update.c | 2 +- + drivers/acpi/power.c | 13 + + drivers/acpi/processor_driver.c | 6 +- + drivers/acpi/processor_thermal.c | 80 +- + drivers/acpi/riscv/cppc.c | 26 +- + drivers/acpi/sbs.c | 8 +- + drivers/acpi/scan.c | 54 +- + drivers/acpi/sysfs.c | 5 +- + drivers/acpi/tables.c | 3 + + drivers/acpi/thermal.c | 19 +- + drivers/acpi/utils.c | 13 +- + drivers/acpi/video_detect.c | 8 + + drivers/acpi/x86/s2idle.c | 29 + + drivers/base/base.h | 2 + + drivers/base/bus.c | 67 +- + drivers/base/power/clock_ops.c | 2 +- + drivers/base/power/main.c | 31 +- + drivers/base/power/runtime-test.c | 57 + + drivers/base/power/runtime.c | 99 +- + drivers/clocksource/timer-ti-dm.c | 9 +- + drivers/cpufreq/amd-pstate-ut.c | 23 +- + drivers/cpufreq/amd-pstate.c | 257 ++- + drivers/cpufreq/amd-pstate.h | 16 + + drivers/cpufreq/cppc_cpufreq.c | 73 +- + drivers/cpufreq/cpufreq.c | 4 +- + drivers/cpufreq/cpufreq_conservative.c | 8 +- + drivers/cpufreq/cpufreq_governor.c | 29 +- + drivers/cpufreq/cpufreq_governor.h | 2 + + drivers/cpufreq/intel_pstate.c | 10 + + drivers/cpuidle/cpuidle-tegra.c | 2 +- + drivers/cpuidle/governors/menu.c | 8 +- + drivers/cpuidle/governors/teo.c | 11 + + drivers/cxl/core/ras.c | 3 +- + drivers/firmware/efi/cper.c | 51 +- + drivers/idle/intel_idle.c | 12 +- + drivers/platform/x86/lenovo/yogabook.c | 4 +- + drivers/platform/x86/serdev_helpers.h | 3 +- + drivers/platform/x86/x86-android-tablets/core.c | 4 +- + drivers/pnp/driver.c | 4 +- + drivers/powercap/intel_rapl_msr.c | 5 +- + drivers/thermal/cpufreq_cooling.c | 14 +- + drivers/thermal/devfreq_cooling.c | 4 +- + drivers/thermal/gov_power_allocator.c | 7 +- + .../int340x_thermal/processor_thermal_device.c | 14 +- + .../int340x_thermal/processor_thermal_soc_slider.c | 28 +- + drivers/thermal/intel/intel_powerclamp.c | 64 +- + drivers/thermal/intel/intel_tcc_cooling.c | 1 + + drivers/thermal/qcom/qcom-spmi-adc-tm5.c | 4 +- + drivers/thermal/thermal_core.c | 30 +- + drivers/thermal/thermal_core.h | 3 +- + drivers/thermal/thermal_of.c | 17 +- + drivers/thermal/ti-soc-thermal/ti-bandgap.c | 12 +- + drivers/thunderbolt/acpi.c | 5 +- + include/acpi/acpi_bus.h | 43 +- + include/acpi/cppc_acpi.h | 18 +- + include/acpi/ghes.h | 4 + + include/acpi/processor.h | 6 +- + include/cxl/event.h | 6 +- + include/linux/acpi.h | 30 +- + include/linux/apple-gmux.h | 2 +- + include/linux/cpufreq.h | 3 + + include/linux/device/bus.h | 1 + + include/linux/pm.h | 93 + + include/linux/pm_runtime.h | 409 ++-- + include/linux/thermal.h | 19 +- + include/uapi/linux/thermal.h | 8 +- + kernel/cpu_pm.c | 9 +- + kernel/power/em_netlink.c | 126 +- + kernel/power/em_netlink.h | 8 +- + kernel/power/energy_model.c | 8 +- + sound/hda/codecs/side-codecs/aw88399_hda.c | 3 +- + sound/soc/amd/acp-es8336.c | 2 +- + sound/soc/amd/acp/acp3x-es83xx/acp3x-es83xx.c | 2 +- + sound/soc/intel/boards/bytcht_es8316.c | 4 +- + sound/soc/intel/boards/bytcr_rt5640.c | 4 +- + sound/soc/intel/boards/bytcr_rt5651.c | 4 +- + sound/soc/intel/boards/cht_bsw_rt5645.c | 5 +- + sound/soc/intel/boards/sof_cirrus_common.c | 2 +- + sound/soc/intel/boards/sof_es8336.c | 4 +- + sound/soc/loongson/loongson_card.c | 3 +- + tools/power/pm-graph/sleepgraph.py | 2 +- + 113 files changed, 3834 insertions(+), 1764 deletions(-) +Merging cpufreq-arm/cpufreq/arm/linux-next (d82d896f00e7b cpufreq: Use %pe to print error pointers symbolically) +$ git merge -m Merge branch 'cpufreq/arm/linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git cpufreq-arm/cpufreq/arm/linux-next +Auto-merging drivers/cpufreq/cppc_cpufreq.c +CONFLICT (content): Merge conflict in drivers/cpufreq/cppc_cpufreq.c +Resolved 'drivers/cpufreq/cppc_cpufreq.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 05f0c876ba59d] Merge branch 'cpufreq/arm/linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git +$ git diff -M --stat --summary HEAD^.. + drivers/cpufreq/airoha-cpufreq.c | 2 +- + drivers/cpufreq/bmips-cpufreq.c | 4 ++-- + drivers/cpufreq/cppc_cpufreq.c | 4 ++-- + drivers/cpufreq/qoriq-cpufreq.c | 3 +-- + drivers/cpufreq/rcpufreq_dt.rs | 1 + + drivers/cpufreq/s3c64xx-cpufreq.c | 5 ++--- + drivers/cpufreq/sparc-us2e-cpufreq.c | 11 +++++------ + drivers/cpufreq/sti-cpufreq.c | 4 ++-- + drivers/cpufreq/tegra194-cpufreq.c | 2 +- + rust/kernel/cpufreq.rs | 18 ++++++++++++++---- + 10 files changed, 31 insertions(+), 23 deletions(-) +Merging cpupower/cpupower (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'cpupower' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux.git cpupower/cpupower +Already up to date. +Merging devfreq/devfreq-next (9a222650d9e70 PM / devfreq: Fix governor_store() failing when device has no current governor) +$ git merge -m Merge branch 'devfreq-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chanwoo/linux.git devfreq/devfreq-next +Auto-merging drivers/devfreq/event/rockchip-dfi.c +Merge made by the 'ort' strategy. + drivers/devfreq/devfreq.c | 50 +++++++----------------------------- + drivers/devfreq/event/rockchip-dfi.c | 4 ++- + 2 files changed, 12 insertions(+), 42 deletions(-) +Merging pmdomain/next (d1a93cf1d3bed pmdomain: Merge branch dt into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/linux-pm.git pmdomain/next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../devicetree/bindings/power/qcom,rpmpd.yaml | 1 + + MAINTAINERS | 8 + + drivers/cpuidle/cpuidle-psci-domain.c | 7 +- + drivers/cpuidle/cpuidle-psci.c | 2 +- + drivers/pmdomain/Kconfig | 1 + + drivers/pmdomain/Makefile | 1 + + drivers/pmdomain/core.c | 163 ++++++- + drivers/pmdomain/core.h | 18 + + drivers/pmdomain/governor.c | 45 ++ + drivers/pmdomain/imx/gpcv2.c | 16 +- + drivers/pmdomain/imx/imx8m-blk-ctrl.c | 15 +- + drivers/pmdomain/imx/imx93-blk-ctrl.c | 3 +- + drivers/pmdomain/imx/scu-pd.c | 10 +- + drivers/pmdomain/qcom/rpmhpd.c | 41 ++ + drivers/pmdomain/renesas/Kconfig | 4 + + drivers/pmdomain/renesas/Makefile | 1 + + drivers/pmdomain/renesas/r8a774a3-sysc.c | 45 ++ + drivers/pmdomain/renesas/r8a7795-sysc.c | 2 +- + drivers/pmdomain/renesas/r8a779a0-sysc.c | 2 +- + drivers/pmdomain/renesas/r8a779f0-sysc.c | 2 +- + drivers/pmdomain/renesas/r8a779g0-sysc.c | 2 +- + drivers/pmdomain/renesas/r8a779h0-sysc.c | 2 +- + drivers/pmdomain/renesas/r8a78000-mdlc.c | 6 +- + drivers/pmdomain/renesas/rcar-sysc.c | 3 + + drivers/pmdomain/renesas/rcar-sysc.h | 13 +- + drivers/pmdomain/riscv/Kconfig | 15 + + drivers/pmdomain/riscv/Makefile | 3 + + drivers/pmdomain/riscv/riscv-rpmi-device-power.c | 485 +++++++++++++++++++++ + drivers/pmdomain/rockchip/pm-domains.c | 4 +- + include/linux/mailbox/riscv-rpmi-message.h | 11 + + include/linux/pm_domain.h | 9 + + include/linux/pm_qos.h | 9 + + kernel/power/qos.c | 4 +- + 33 files changed, 886 insertions(+), 67 deletions(-) + create mode 100644 drivers/pmdomain/core.h + create mode 100644 drivers/pmdomain/renesas/r8a774a3-sysc.c + create mode 100644 drivers/pmdomain/riscv/Kconfig + create mode 100644 drivers/pmdomain/riscv/Makefile + create mode 100644 drivers/pmdomain/riscv/riscv-rpmi-device-power.c +Merging opp/opp/linux-next (1f6de65e33145 opp: fix use after free in _update_opp_table_clk()) +$ git merge -m Merge branch 'opp/linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/vireshk/pm.git opp/opp/linux-next +Already up to date. +Merging thermal/thermal/linux-next (e856ca3013b30 thermal/drivers/renesas/rzg3s: Add RZ/G3L TSU support) +$ git merge -m Merge branch 'thermal/linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/thermal/linux.git thermal/thermal/linux-next +Merge made by the 'ort' strategy. + .../bindings/thermal/qcom,pm8775-mbg-tm.yaml | 7 +- + .../bindings/thermal/renesas,r9a08g045-tsu.yaml | 4 +- + .../bindings/thermal/ti,omap-bandgap.yaml | 177 +++++++++++++++++++++ + .../devicetree/bindings/thermal/ti_soc_thermal.txt | 88 ---------- + drivers/thermal/airoha_thermal.c | 24 +-- + drivers/thermal/imx91_thermal.c | 2 +- + drivers/thermal/qcom/tsens-v2.c | 4 +- + drivers/thermal/renesas/rzg3e_thermal.c | 35 ++-- + drivers/thermal/renesas/rzg3s_thermal.c | 38 +++-- + tools/thermal/tmon/sysfs.c | 2 +- + tools/thermal/tmon/tmon.h | 2 +- + tools/thermal/tmon/tui.c | 2 +- + 12 files changed, 258 insertions(+), 127 deletions(-) + create mode 100644 Documentation/devicetree/bindings/thermal/ti,omap-bandgap.yaml + delete mode 100644 Documentation/devicetree/bindings/thermal/ti_soc_thermal.txt +Merging rdma/for-next (485effd117d03 RDMA/ionic: Fix double removal of CMB mmap entries in create QP error path) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git rdma/for-next +Auto-merging MAINTAINERS +Auto-merging drivers/infiniband/core/nldev.c +Auto-merging drivers/infiniband/core/verbs.c +Auto-merging drivers/infiniband/hw/bnxt_re/main.c +Auto-merging drivers/infiniband/hw/erdma/erdma_main.c +Auto-merging drivers/infiniband/hw/erdma/erdma_verbs.c +Auto-merging drivers/infiniband/hw/hns/hns_roce_debugfs.c +Auto-merging drivers/infiniband/hw/irdma/verbs.c +Auto-merging drivers/infiniband/hw/mlx5/main.c +Auto-merging drivers/infiniband/sw/rxe/rxe_mr.c +Auto-merging drivers/infiniband/sw/rxe/rxe_odp.c +Auto-merging drivers/infiniband/sw/rxe/rxe_verbs.c +CONFLICT (content): Merge conflict in drivers/infiniband/sw/rxe/rxe_verbs.c +Auto-merging drivers/infiniband/ulp/isert/ib_isert.c +Auto-merging drivers/infiniband/ulp/rtrs/rtrs-clt.c +Auto-merging drivers/net/ethernet/microsoft/mana/gdma_main.c +Recorded preimage for 'drivers/infiniband/sw/rxe/rxe_verbs.c' +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +Recorded resolution for 'drivers/infiniband/sw/rxe/rxe_verbs.c'. +[master e20a0522c63b9] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git +$ git diff -M --stat --summary HEAD^.. + Documentation/ABI/stable/sysfs-class-infiniband | 118 ----- + MAINTAINERS | 2 +- + drivers/infiniband/core/cma.c | 34 +- + drivers/infiniband/core/cma_trace.h | 1 + + drivers/infiniband/core/device.c | 4 +- + drivers/infiniband/core/iwcm.c | 2 +- + drivers/infiniband/core/lag.c | 2 +- + drivers/infiniband/core/mad_rmpp.c | 7 +- + drivers/infiniband/core/multicast.c | 7 +- + drivers/infiniband/core/nldev.c | 9 +- + drivers/infiniband/core/restrack.c | 3 +- + drivers/infiniband/core/umem.c | 77 ++- + drivers/infiniband/core/uverbs_std_types_device.c | 1 + + drivers/infiniband/core/uverbs_std_types_wq.c | 2 +- + drivers/infiniband/core/uverbs_uapi.c | 2 +- + drivers/infiniband/core/verbs.c | 2 +- + drivers/infiniband/hw/bnxt_re/bnxt_re.h | 4 + + drivers/infiniband/hw/bnxt_re/hw_counters.c | 6 +- + drivers/infiniband/hw/bnxt_re/ib_verbs.c | 69 ++- + drivers/infiniband/hw/bnxt_re/main.c | 26 +- + drivers/infiniband/hw/bnxt_re/qplib_res.c | 9 +- + drivers/infiniband/hw/bnxt_re/qplib_sp.c | 2 +- + drivers/infiniband/hw/cxgb4/cm.c | 5 +- + drivers/infiniband/hw/cxgb4/cq.c | 2 +- + drivers/infiniband/hw/efa/efa_admin_cmds_defs.h | 5 +- + drivers/infiniband/hw/efa/efa_com_cmd.c | 9 +- + drivers/infiniband/hw/efa/efa_com_cmd.h | 8 +- + drivers/infiniband/hw/efa/efa_verbs.c | 5 +- + drivers/infiniband/hw/erdma/erdma_cq.c | 18 +- + drivers/infiniband/hw/erdma/erdma_main.c | 2 +- + drivers/infiniband/hw/erdma/erdma_qp.c | 38 +- + drivers/infiniband/hw/erdma/erdma_verbs.c | 583 ++++++++++++---------- + drivers/infiniband/hw/erdma/erdma_verbs.h | 66 ++- + drivers/infiniband/hw/hfi1/affinity.c | 2 +- + drivers/infiniband/hw/hfi1/firmware.c | 2 +- + drivers/infiniband/hw/hfi1/pcie.c | 14 +- + drivers/infiniband/hw/hns/hns_roce_bond.c | 4 +- + drivers/infiniband/hw/hns/hns_roce_cq.c | 3 +- + drivers/infiniband/hw/hns/hns_roce_debugfs.c | 55 ++ + drivers/infiniband/hw/hns/hns_roce_debugfs.h | 1 + + drivers/infiniband/hw/hns/hns_roce_device.h | 6 +- + drivers/infiniband/hw/hns/hns_roce_hw_v2.c | 63 ++- + drivers/infiniband/hw/hns/hns_roce_mr.c | 22 +- + drivers/infiniband/hw/hns/hns_roce_qp.c | 2 +- + drivers/infiniband/hw/hns/hns_roce_srq.c | 4 +- + drivers/infiniband/hw/ionic/ionic_controlpath.c | 41 +- + drivers/infiniband/hw/ionic/ionic_fw.h | 2 + + drivers/infiniband/hw/ionic/ionic_ibdev.c | 4 +- + drivers/infiniband/hw/ionic/ionic_lif_cfg.c | 1 + + drivers/infiniband/hw/ionic/ionic_lif_cfg.h | 1 + + drivers/infiniband/hw/irdma/ctrl.c | 23 +- + drivers/infiniband/hw/irdma/hw.c | 15 +- + drivers/infiniband/hw/irdma/icrdma_if.c | 9 +- + drivers/infiniband/hw/irdma/main.c | 2 + + drivers/infiniband/hw/irdma/pble.c | 8 +- + drivers/infiniband/hw/irdma/protos.h | 3 +- + drivers/infiniband/hw/irdma/type.h | 7 +- + drivers/infiniband/hw/irdma/verbs.c | 7 +- + drivers/infiniband/hw/mana/cq.c | 396 +++++++++++---- + drivers/infiniband/hw/mana/device.c | 3 +- + drivers/infiniband/hw/mana/main.c | 43 +- + drivers/infiniband/hw/mana/mana_ib.h | 134 ++++- + drivers/infiniband/hw/mana/qp.c | 150 ++++-- + drivers/infiniband/hw/mana/shadow_queue.h | 61 +-- + drivers/infiniband/hw/mana/wq.c | 3 +- + drivers/infiniband/hw/mana/wr.c | 170 ++++--- + drivers/infiniband/hw/mlx4/cq.c | 17 +- + drivers/infiniband/hw/mlx4/mcg.c | 4 +- + drivers/infiniband/hw/mlx5/cq.c | 9 +- + drivers/infiniband/hw/mlx5/data_direct.c | 10 +- + drivers/infiniband/hw/mlx5/fs.c | 45 +- + drivers/infiniband/hw/mlx5/main.c | 54 +- + drivers/infiniband/hw/mlx5/odp.c | 2 +- + drivers/infiniband/hw/mlx5/qp.c | 32 +- + drivers/infiniband/hw/mlx5/qpc.c | 20 +- + drivers/infiniband/hw/mlx5/srq.c | 3 +- + drivers/infiniband/hw/mlx5/wr.c | 1 + + drivers/infiniband/hw/mthca/mthca_cq.c | 3 +- + drivers/infiniband/hw/ocrdma/ocrdma_verbs.c | 6 +- + drivers/infiniband/hw/qedr/verbs.c | 25 +- + drivers/infiniband/hw/usnic/usnic_abi.h | 2 +- + drivers/infiniband/hw/usnic/usnic_ib_verbs.c | 2 +- + drivers/infiniband/hw/usnic/usnic_transport.h | 2 +- + drivers/infiniband/hw/vmw_pvrdma/pvrdma_qp.c | 6 +- + drivers/infiniband/hw/vmw_pvrdma/pvrdma_srq.c | 3 +- + drivers/infiniband/sw/rdmavt/qp.c | 2 +- + drivers/infiniband/sw/rdmavt/srq.c | 2 +- + drivers/infiniband/sw/rxe/rxe_loc.h | 1 + + drivers/infiniband/sw/rxe/rxe_mr.c | 6 + + drivers/infiniband/sw/rxe/rxe_net.c | 71 ++- + drivers/infiniband/sw/rxe/rxe_odp.c | 4 +- + drivers/infiniband/sw/rxe/rxe_req.c | 2 +- + drivers/infiniband/sw/rxe/rxe_resp.c | 6 +- + drivers/infiniband/sw/rxe/rxe_verbs.c | 6 + + drivers/infiniband/sw/siw/siw_main.c | 5 +- + drivers/infiniband/sw/siw/siw_qp_tx.c | 1 - + drivers/infiniband/ulp/ipoib/ipoib_vlan.c | 2 +- + drivers/infiniband/ulp/iser/iscsi_iser.c | 6 +- + drivers/infiniband/ulp/iser/iser_verbs.c | 2 +- + drivers/infiniband/ulp/isert/ib_isert.c | 2 +- + drivers/infiniband/ulp/rtrs/rtrs-clt.c | 27 +- + drivers/infiniband/ulp/srpt/ib_srpt.c | 2 +- + drivers/infiniband/ulp/srpt/ib_srpt.h | 2 +- + drivers/net/ethernet/microsoft/mana/gdma_main.c | 51 +- + include/net/mana/gdma.h | 12 +- + include/rdma/ib_umem.h | 2 + + include/rdma/ib_verbs.h | 4 +- + include/rdma/uverbs_ioctl.h | 2 +- + include/uapi/rdma/ionic-abi.h | 9 +- + include/uapi/rdma/mana-abi.h | 12 + + 110 files changed, 1854 insertions(+), 1024 deletions(-) +Merging net-next/main (47a1446725732 Merge branch 'mptcp-misc-improvements-for-v7-4') +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net-next.git net-next/main +Auto-merging .mailmap +Auto-merging MAINTAINERS +Auto-merging drivers/net/ethernet/broadcom/bcmsysport.c +Auto-merging drivers/net/ethernet/broadcom/genet/bcmgenet.c +Auto-merging drivers/net/ethernet/marvell/octeontx2/nic/otx2_common.c +Auto-merging drivers/net/ethernet/marvell/octeontx2/nic/otx2_common.h +Auto-merging drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c +Auto-merging drivers/net/ethernet/stmicro/stmmac/stmmac_main.c +Auto-merging drivers/net/wireless/ath/ath11k/Kconfig +Auto-merging drivers/net/wireless/ath/ath12k/pci.c +Auto-merging include/linux/skbuff.h +Auto-merging net/core/dev.c +Auto-merging net/core/skbuff.c +Auto-merging net/core/sock.c +Auto-merging net/ipv4/af_inet.c +Auto-merging net/ipv4/tcp.c +Auto-merging net/ipv4/tcp_input.c +Auto-merging net/ipv6/addrconf.c +Auto-merging net/mac80211/ieee80211_i.h +CONFLICT (content): Merge conflict in net/mac80211/ieee80211_i.h +Auto-merging net/mac80211/iface.c +Auto-merging net/mac80211/mesh_pathtbl.c +Auto-merging net/mac80211/mlme.c +Auto-merging net/mac80211/rx.c +Auto-merging net/mac80211/tx.c +CONFLICT (content): Merge conflict in net/mac80211/tx.c +Auto-merging net/packet/af_packet.c +Auto-merging net/socket.c +Auto-merging net/unix/af_unix.c +Auto-merging net/wireless/nl80211.c +Auto-merging rust/kernel/net/netlink.rs +Auto-merging tools/testing/selftests/net/Makefile +Resolved 'net/mac80211/ieee80211_i.h' using previous resolution. +Resolved 'net/mac80211/tx.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 88d6478ff3a01] Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netdev/net-next.git +$ git diff -M --stat --summary HEAD^.. + .mailmap | 13 +- + .../ABI/testing/sysfs-devices-platform-soc-ipa | 14 - + Documentation/ABI/testing/sysfs-timecard | 37 +- + Documentation/admin-guide/sysctl/net.rst | 14 + + .../bindings/net/allwinner,sun8i-a83t-emac.yaml | 13 + + .../bindings/net/altr,socfpga-stmmac.yaml | 2 - + .../devicetree/bindings/net/brcm,asp-v2.0.yaml | 2 +- + .../bindings/net/cortina,gemini-ethernet.yaml | 52 +- + .../devicetree/bindings/net/dsa/microchip,ksz.yaml | 1 + + .../bindings/net/dsa/motorcomm,yt921x.yaml | 23 + + .../devicetree/bindings/net/dsa/realtek.yaml | 33 + + .../bindings/net/dsa/renesas,rzn1-a5psw.yaml | 4 +- + .../devicetree/bindings/net/ethernet-phy.yaml | 16 + + .../devicetree/bindings/net/fsl,fman-dtsec.yaml | 4 - + .../devicetree/bindings/net/intel,dwmac-plat.yaml | 6 +- + .../devicetree/bindings/net/mdio-gpio.yaml | 16 +- + .../devicetree/bindings/net/microchip,lan8650.yaml | 5 + + .../bindings/net/mscc,vsc7514-switch.yaml | 8 +- + .../bindings/net/nvidia,tegra234-mgbe.yaml | 8 +- + .../devicetree/bindings/net/qcom,ipa.yaml | 10 +- + .../devicetree/bindings/net/realtek,rtl82xx.yaml | 6 +- + .../bindings/net/realtek,rtl9301-mdio.yaml | 14 +- + .../devicetree/bindings/net/renesas,etheravb.yaml | 11 + + .../devicetree/bindings/net/snps,dwmac-common.yaml | 66 + + .../devicetree/bindings/net/snps,dwmac.yaml | 59 +- + .../bindings/net/socionext,uniphier-ave4.yaml | 32 +- + .../devicetree/bindings/net/ti,cpsw-switch.yaml | 70 +- + .../bindings/net/ultrarisc,dp1000-gmac.yaml | 74 ++ + .../bindings/net/wireless/marvell,sd8787.yaml | 6 + + .../bindings/net/wireless/mediatek,mt76.yaml | 32 +- + .../bindings/net/wireless/qcom,ath10k.yaml | 2 +- + .../bindings/net/wireless/qcom,ath12k-wsi.yaml | 7 - + .../bindings/net/wireless/silabs,wfx.yaml | 3 + + .../devicetree/bindings/net/wireless/ti,wl1251.txt | 64 - + .../bindings/net/wireless/ti,wl1251.yaml | 85 ++ + .../devicetree/bindings/net/wiznet,w5100.yaml | 88 ++ + .../devicetree/bindings/net/wiznet,w5x00.txt | 50 - + .../bindings/net/x-powers,acx00-ephy-package.yaml | 170 +++ + Documentation/netlink/netlink-raw.yaml | 31 +- + Documentation/netlink/specs/conntrack.yaml | 60 +- + Documentation/netlink/specs/devlink.yaml | 171 ++- + Documentation/netlink/specs/dpll.yaml | 22 +- + Documentation/netlink/specs/ethtool.yaml | 4 +- + Documentation/netlink/specs/fou.yaml | 4 + + Documentation/netlink/specs/handshake.yaml | 9 +- + Documentation/netlink/specs/netdev.yaml | 5 +- + Documentation/netlink/specs/nl80211.yaml | 4 +- + Documentation/netlink/specs/nlctrl.yaml | 20 +- + Documentation/netlink/specs/psp.yaml | 2 + + Documentation/netlink/specs/rt-link.yaml | 44 +- + Documentation/netlink/specs/tc.yaml | 2 +- + Documentation/netlink/specs/tcp_metrics.yaml | 18 +- + .../networking/device_drivers/ethernet/index.rst | 1 + + .../device_drivers/ethernet/intel/i40e.rst | 2 +- + .../device_drivers/ethernet/intel/iavf.rst | 2 +- + .../device_drivers/ethernet/microsoft/netvsc.rst | 2 - + .../device_drivers/ethernet/nebula-matrix/nbl.rst | 28 + + .../device_drivers/ethernet/stmicro/stmmac.rst | 1 - + Documentation/networking/devlink/index.rst | 1 + + Documentation/networking/devlink/ptp_ocp.rst | 75 ++ + Documentation/networking/dsa/dsa.rst | 2 +- + Documentation/networking/ethtool-netlink.rst | 2 +- + Documentation/networking/ip-sysctl.rst | 28 +- + Documentation/networking/mctp.rst | 4 +- + .../networking/net_cachelines/net_device.rst | 1 + + Documentation/networking/netconsole.rst | 5 + + Documentation/networking/netdevices.rst | 21 +- + Documentation/networking/nf_conntrack-sysctl.rst | 2 +- + Documentation/networking/page_pool.rst | 2 +- + Documentation/networking/phonet.rst | 2 +- + Documentation/networking/phy.rst | 17 +- + Documentation/networking/proc_net_tcp.rst | 20 +- + Documentation/networking/radiotap-headers.rst | 2 +- + Documentation/networking/scaling.rst | 9 + + MAINTAINERS | 68 +- + drivers/dibs/dibs_loopback.c | 5 +- + drivers/dpll/dpll_netlink.c | 2 +- + drivers/dpll/dpll_nl.c | 20 +- + drivers/infiniband/ulp/ipoib/ipoib_main.c | 27 +- + drivers/net/bonding/bond_alb.c | 63 +- + drivers/net/can/bxcan.c | 2 +- + drivers/net/can/c_can/c_can_platform.c | 4 +- + drivers/net/can/flexcan/flexcan-core.c | 2 +- + drivers/net/can/ifi_canfd/ifi_canfd.c | 2 +- + drivers/net/can/m_can/m_can_platform.c | 2 +- + drivers/net/can/sja1000/f81601.c | 2 +- + drivers/net/can/sja1000/sja1000_platform.c | 2 +- + drivers/net/can/slcan/slcan-core.c | 5 +- + drivers/net/dsa/Kconfig | 19 +- + drivers/net/dsa/Makefile | 3 +- + drivers/net/dsa/b53/b53_mdio.c | 2 +- + drivers/net/dsa/b53/b53_srab.c | 2 +- + drivers/net/dsa/bcm_sf2.c | 2 +- + drivers/net/dsa/hirschmann/hellcreek.c | 2 +- + drivers/net/dsa/ks8995.c | 857 ------------ + drivers/net/dsa/lan9303_i2c.c | 2 +- + drivers/net/dsa/lan9303_mdio.c | 2 +- + drivers/net/dsa/lantiq/lantiq_gswip_common.c | 9 +- + drivers/net/dsa/lantiq/mxl-gsw1xx.c | 31 +- + drivers/net/dsa/microchip/Kconfig | 1 + + drivers/net/dsa/microchip/ksz8.c | 252 +++- + drivers/net/dsa/microchip/ksz8.h | 2 + + drivers/net/dsa/microchip/ksz8_reg.h | 2 + + drivers/net/dsa/microchip/ksz_common.c | 114 +- + drivers/net/dsa/microchip/ksz_common.h | 16 +- + drivers/net/dsa/microchip/ksz_dcb.c | 68 +- + drivers/net/dsa/microchip/ksz_spi.c | 33 +- + drivers/net/dsa/motorcomm/Kconfig | 17 + + drivers/net/dsa/motorcomm/Makefile | 7 + + drivers/net/dsa/{yt921x.c => motorcomm/chip.c} | 752 ++--------- + drivers/net/dsa/{yt921x.h => motorcomm/chip.h} | 119 +- + drivers/net/dsa/motorcomm/leds.c | 641 +++++++++ + drivers/net/dsa/motorcomm/leds.h | 118 ++ + drivers/net/dsa/motorcomm/mdio_bus.c | 301 +++++ + drivers/net/dsa/motorcomm/mdio_bus.h | 54 + + drivers/net/dsa/motorcomm/pcs-921x.c | 247 ++++ + drivers/net/dsa/motorcomm/pcs.h | 13 + + drivers/net/dsa/motorcomm/smi.c | 180 +++ + drivers/net/dsa/motorcomm/smi.h | 61 + + drivers/net/dsa/mt7530-mdio.c | 14 +- + drivers/net/dsa/mt7530-mmio.c | 2 +- + drivers/net/dsa/mt7530.c | 834 ++++++------ + drivers/net/dsa/mt7530.h | 221 ++-- + drivers/net/dsa/mv88e6060.c | 3 +- + drivers/net/dsa/mv88e6xxx/chip.c | 9 +- + drivers/net/dsa/qca/qca8k-8xxx.c | 2 +- + drivers/net/dsa/qca/qca8k-common.c | 15 +- + drivers/net/dsa/realtek/realtek.h | 7 + + drivers/net/dsa/realtek/rtl8365mb_main.c | 145 +- + drivers/net/dsa/realtek/rtl8366rb.c | 2 +- + drivers/net/dsa/realtek/rtl83xx.c | 43 +- + drivers/net/dsa/rzn1_a5psw.c | 2 +- + drivers/net/dsa/sja1105/sja1105_main.c | 2 +- + drivers/net/dsa/vitesse-vsc73xx-spi.c | 9 +- + drivers/net/ethernet/8390/pcnet_cs.c | 9 +- + drivers/net/ethernet/Kconfig | 1 + + drivers/net/ethernet/Makefile | 1 + + drivers/net/ethernet/adaptec/starfire.c | 6 +- + drivers/net/ethernet/adi/adin1140.c | 4 +- + drivers/net/ethernet/airoha/airoha_eth.c | 363 ++++- + drivers/net/ethernet/airoha/airoha_eth.h | 29 +- + drivers/net/ethernet/airoha/airoha_regs.h | 23 +- + drivers/net/ethernet/alacritech/slicoss.c | 1 + + drivers/net/ethernet/amd/xgbe/xgbe-mdio.c | 5 +- + drivers/net/ethernet/aquantia/atlantic/aq_nic.c | 5 +- + .../net/ethernet/aquantia/atlantic/aq_pci_func.c | 2 - + drivers/net/ethernet/atheros/alx/main.c | 13 +- + drivers/net/ethernet/atheros/atl1c/atl1c_hw.h | 2 +- + drivers/net/ethernet/atheros/atl1e/atl1e.h | 2 +- + drivers/net/ethernet/atheros/atl1e/atl1e_hw.c | 2 +- + drivers/net/ethernet/atheros/atl1e/atl1e_main.c | 2 +- + drivers/net/ethernet/atheros/atlx/atl1.c | 2 +- + drivers/net/ethernet/atheros/atlx/atl2.c | 2 +- + drivers/net/ethernet/broadcom/asp2/bcmasp.c | 4 +- + drivers/net/ethernet/broadcom/asp2/bcmasp.h | 5 - + drivers/net/ethernet/broadcom/bcm63xx_enet.c | 2 +- + drivers/net/ethernet/broadcom/bcm63xx_enet.h | 2 +- + drivers/net/ethernet/broadcom/bcmsysport.c | 25 +- + drivers/net/ethernet/broadcom/bcmsysport.h | 2 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x.h | 4 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x_dcb.c | 4 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x_hsi.h | 2 +- + .../net/ethernet/broadcom/bnx2x/bnx2x_init_ops.h | 2 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x_link.c | 8 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x_main.c | 2 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x_reg.h | 4 +- + drivers/net/ethernet/broadcom/bnx2x/bnx2x_sriov.c | 18 +- + drivers/net/ethernet/broadcom/bnxt/bnxt.c | 4 +- + drivers/net/ethernet/broadcom/genet/bcmgenet.c | 11 +- + drivers/net/ethernet/cadence/macb.h | 7 +- + drivers/net/ethernet/cadence/macb_main.c | 172 ++- + drivers/net/ethernet/cadence/macb_ptp.c | 62 +- + .../net/ethernet/cavium/liquidio/cn66xx_device.c | 2 +- + drivers/net/ethernet/cavium/liquidio/lio_core.c | 2 +- + .../net/ethernet/cavium/liquidio/octeon_config.h | 4 +- + drivers/net/ethernet/cavium/liquidio/octeon_iq.h | 2 +- + drivers/net/ethernet/cavium/liquidio/octeon_main.h | 2 +- + drivers/net/ethernet/cavium/liquidio/octeon_nic.h | 2 +- + .../net/ethernet/cavium/liquidio/request_manager.c | 2 +- + drivers/net/ethernet/cavium/thunder/nic.h | 2 +- + drivers/net/ethernet/cavium/thunder/nic_main.c | 2 +- + drivers/net/ethernet/cisco/enic/enic_main.c | 19 +- + drivers/net/ethernet/cortina/gemini.c | 32 +- + drivers/net/ethernet/cortina/gemini.h | 2 +- + drivers/net/ethernet/davicom/dm9000.c | 6 +- + drivers/net/ethernet/dec/tulip/de2104x.c | 2 +- + drivers/net/ethernet/dec/tulip/interrupt.c | 4 +- + drivers/net/ethernet/dec/tulip/media.c | 2 +- + drivers/net/ethernet/dec/tulip/tulip.h | 4 +- + drivers/net/ethernet/dec/tulip/winbond-840.c | 2 +- + drivers/net/ethernet/freescale/dpaa2/dpaa2-ptp.c | 2 + + drivers/net/ethernet/freescale/enetc/Kconfig | 1 + + drivers/net/ethernet/freescale/enetc/enetc.c | 68 +- + drivers/net/ethernet/freescale/enetc/enetc.h | 11 +- + .../net/ethernet/freescale/enetc/enetc4_debugfs.c | 51 +- + drivers/net/ethernet/freescale/enetc/enetc4_hw.h | 1 + + drivers/net/ethernet/freescale/enetc/enetc4_pf.c | 85 +- + .../net/ethernet/freescale/enetc/enetc_ethtool.c | 6 + + drivers/net/ethernet/freescale/enetc/enetc_hw.h | 26 + + .../net/ethernet/freescale/enetc/enetc_mailbox.h | 109 +- + drivers/net/ethernet/freescale/enetc/enetc_mdio.c | 2 +- + drivers/net/ethernet/freescale/enetc/enetc_msg.c | 641 ++++++++- + drivers/net/ethernet/freescale/enetc/enetc_pf.c | 64 +- + drivers/net/ethernet/freescale/enetc/enetc_pf.h | 21 +- + .../net/ethernet/freescale/enetc/enetc_pf_common.c | 150 ++- + .../net/ethernet/freescale/enetc/enetc_pf_common.h | 19 + + drivers/net/ethernet/freescale/enetc/enetc_vf.c | 453 ++++++- + drivers/net/ethernet/freescale/fec.h | 2 +- + drivers/net/ethernet/freescale/fec_main.c | 18 +- + drivers/net/ethernet/freescale/fec_ptp.c | 24 +- + .../net/ethernet/freescale/fs_enet/fs_enet-main.c | 1 + + .../net/ethernet/hisilicon/hibmcge/hbg_common.h | 2 +- + drivers/net/ethernet/hisilicon/hibmcge/hbg_hw.c | 4 +- + drivers/net/ethernet/hisilicon/hibmcge/hbg_irq.c | 4 +- + drivers/net/ethernet/hisilicon/hibmcge/hbg_reg.h | 4 +- + drivers/net/ethernet/hisilicon/hibmcge/hbg_txrx.c | 12 +- + drivers/net/ethernet/hisilicon/hip04_eth.c | 5 +- + drivers/net/ethernet/hisilicon/hns/hns_dsaf_misc.c | 2 +- + drivers/net/ethernet/hisilicon/hns/hns_dsaf_ppe.c | 2 +- + drivers/net/ethernet/hisilicon/hns/hns_enet.c | 4 +- + drivers/net/ethernet/hisilicon/hns3/hns3_ethtool.c | 10 +- + .../ethernet/hisilicon/hns3/hns3pf/hclge_main.c | 11 +- + .../net/ethernet/hisilicon/hns3/hns3pf/hclge_mbx.c | 7 +- + .../net/ethernet/hisilicon/hns3/hns3pf/hclge_tm.c | 2 +- + .../ethernet/hisilicon/hns3/hns3vf/hclgevf_main.c | 2 +- + drivers/net/ethernet/huawei/hinic3/hinic3_hwdev.c | 7 +- + drivers/net/ethernet/intel/fm10k/fm10k_pci.c | 2 - + drivers/net/ethernet/intel/i40e/i40e_common.c | 2 +- + drivers/net/ethernet/intel/i40e/i40e_ethtool.c | 12 +- + drivers/net/ethernet/intel/i40e/i40e_main.c | 70 +- + drivers/net/ethernet/intel/i40e/i40e_txrx.c | 5 +- + drivers/net/ethernet/intel/i40e/i40e_txrx.h | 7 +- + drivers/net/ethernet/intel/i40e/i40e_type.h | 5 + + drivers/net/ethernet/intel/i40e/i40e_virtchnl_pf.c | 5 +- + drivers/net/ethernet/intel/i40e/i40e_xsk.c | 12 + + drivers/net/ethernet/intel/ice/Makefile | 2 +- + drivers/net/ethernet/intel/ice/ice.h | 9 +- + drivers/net/ethernet/intel/ice/ice_arfs.c | 8 +- + drivers/net/ethernet/intel/ice/ice_arfs.h | 2 +- + drivers/net/ethernet/intel/ice/ice_dpll.c | 495 ++++++- + drivers/net/ethernet/intel/ice/ice_dpll.h | 4 + + drivers/net/ethernet/intel/ice/ice_ethtool.c | 14 +- + .../{ice_ethtool_fdir.c => ice_ethtool_ntuple.c} | 88 +- + drivers/net/ethernet/intel/ice/ice_fdir.c | 38 +- + drivers/net/ethernet/intel/ice/ice_fdir.h | 14 +- + drivers/net/ethernet/intel/ice/ice_main.c | 11 +- + drivers/net/ethernet/intel/ice/ice_ptp.c | 82 ++ + drivers/net/ethernet/intel/ice/ice_ptp.h | 11 + + drivers/net/ethernet/intel/ice/ice_sriov.c | 2 +- + drivers/net/ethernet/intel/ice/ice_tspll.c | 120 +- + drivers/net/ethernet/intel/ice/ice_tspll.h | 6 + + drivers/net/ethernet/intel/ice/ice_txclk.c | 20 +- + drivers/net/ethernet/intel/ice/ice_type.h | 4 +- + drivers/net/ethernet/intel/ice/virt/fdir.c | 28 +- + drivers/net/ethernet/intel/idpf/idpf.h | 12 + + drivers/net/ethernet/intel/idpf/idpf_lib.c | 19 + + drivers/net/ethernet/intel/idpf/idpf_txrx.c | 71 +- + drivers/net/ethernet/intel/idpf/idpf_txrx.h | 8 +- + drivers/net/ethernet/intel/idpf/idpf_virtchnl.c | 60 +- + drivers/net/ethernet/intel/ixgbevf/ixgbevf_main.c | 2 +- + drivers/net/ethernet/marvell/mvmdio.c | 4 +- + drivers/net/ethernet/marvell/mvneta.c | 2 +- + drivers/net/ethernet/marvell/mvpp2/mvpp2.h | 2 +- + drivers/net/ethernet/marvell/mvpp2/mvpp2_main.c | 2 +- + drivers/net/ethernet/marvell/mvpp2/mvpp2_prs.c | 8 +- + .../ethernet/marvell/octeon_ep/octep_pfvf_mbox.c | 1 - + drivers/net/ethernet/marvell/octeontx2/af/cgx.c | 25 +- + drivers/net/ethernet/marvell/octeontx2/af/cgx.h | 2 + + .../net/ethernet/marvell/octeontx2/af/cn20k/npa.c | 1 - + .../net/ethernet/marvell/octeontx2/af/cn20k/npc.c | 5 +- + .../ethernet/marvell/octeontx2/af/lmac_common.h | 2 + + .../net/ethernet/marvell/octeontx2/af/mcs_reg.h | 2 - + drivers/net/ethernet/marvell/octeontx2/af/npc.h | 4 +- + drivers/net/ethernet/marvell/octeontx2/af/rpm.c | 40 +- + drivers/net/ethernet/marvell/octeontx2/af/rpm.h | 2 + + drivers/net/ethernet/marvell/octeontx2/af/rvu.c | 4 +- + drivers/net/ethernet/marvell/octeontx2/af/rvu.h | 5 +- + .../net/ethernet/marvell/octeontx2/af/rvu_cgx.c | 12 + + .../ethernet/marvell/octeontx2/af/rvu_debugfs.c | 8 + + .../net/ethernet/marvell/octeontx2/af/rvu_nix.c | 33 +- + .../net/ethernet/marvell/octeontx2/af/rvu_npa.c | 10 +- + .../net/ethernet/marvell/octeontx2/af/rvu_npc.c | 5 +- + .../net/ethernet/marvell/octeontx2/af/rvu_npc_fs.c | 1 - + drivers/net/ethernet/marvell/octeontx2/nic/cn10k.c | 2 +- + .../ethernet/marvell/octeontx2/nic/otx2_common.c | 8 +- + .../ethernet/marvell/octeontx2/nic/otx2_common.h | 1 - + .../ethernet/marvell/octeontx2/nic/otx2_dcbnl.c | 7 +- + .../ethernet/marvell/octeontx2/nic/otx2_ethtool.c | 18 + + .../ethernet/marvell/octeontx2/nic/otx2_flows.c | 3 - + .../net/ethernet/marvell/octeontx2/nic/otx2_pf.c | 101 +- + .../net/ethernet/marvell/octeontx2/nic/otx2_reg.h | 2 + + .../net/ethernet/marvell/octeontx2/nic/otx2_tc.c | 1 - + .../net/ethernet/marvell/octeontx2/nic/otx2_vf.c | 9 +- + drivers/net/ethernet/marvell/octeontx2/nic/qos.h | 2 - + .../ethernet/marvell/prestera/prestera_router.c | 10 +- + drivers/net/ethernet/marvell/pxa168_eth.c | 4 +- + drivers/net/ethernet/mellanox/mlx4/main.c | 8 +- + drivers/net/ethernet/mellanox/mlx5/core/en/ptp.c | 2 +- + .../net/ethernet/mellanox/mlx5/core/en/rep/neigh.c | 29 +- + .../net/ethernet/mellanox/mlx5/core/en/tc_priv.h | 14 +- + .../ethernet/mellanox/mlx5/core/en/tc_tun_encap.c | 23 +- + .../ethernet/mellanox/mlx5/core/en/tc_tun_vxlan.c | 11 +- + drivers/net/ethernet/mellanox/mlx5/core/en/xdp.c | 2 +- + .../ethernet/mellanox/mlx5/core/en_accel/ipsec.c | 6 +- + .../net/ethernet/mellanox/mlx5/core/en_ethtool.c | 6 +- + drivers/net/ethernet/mellanox/mlx5/core/en_main.c | 3 +- + drivers/net/ethernet/mellanox/mlx5/core/en_rx.c | 2 +- + drivers/net/ethernet/mellanox/mlx5/core/en_tc.c | 75 +- + drivers/net/ethernet/mellanox/mlx5/core/en_tx.c | 2 +- + drivers/net/ethernet/mellanox/mlx5/core/eswitch.c | 20 +- + drivers/net/ethernet/mellanox/mlx5/core/eswitch.h | 2 +- + .../ethernet/mellanox/mlx5/core/eswitch_offloads.c | 37 +- + .../net/ethernet/mellanox/mlx5/core/fpga/conn.c | 2 +- + drivers/net/ethernet/mellanox/mlx5/core/fw_reset.c | 6 +- + .../net/ethernet/mellanox/mlx5/core/lag/debugfs.c | 14 +- + drivers/net/ethernet/mellanox/mlx5/core/lag/lag.c | 154 ++- + drivers/net/ethernet/mellanox/mlx5/core/lag/lag.h | 2 +- + .../ethernet/mellanox/mlx5/core/lag/shared_fdb.c | 3 +- + drivers/net/ethernet/mellanox/mlx5/core/lib/aso.c | 2 +- + drivers/net/ethernet/mellanox/mlx5/core/main.c | 8 +- + .../ethernet/mellanox/mlx5/core/steering/hws/bwc.c | 4 +- + .../mellanox/mlx5/core/steering/hws/send.c | 34 +- + drivers/net/ethernet/mellanox/mlxsw/pci.c | 7 +- + .../ethernet/mellanox/mlxsw/spectrum_nve_vxlan.c | 19 +- + .../net/ethernet/mellanox/mlxsw/spectrum_router.c | 31 +- + .../net/ethernet/mellanox/mlxsw/spectrum_span.c | 10 +- + .../ethernet/mellanox/mlxsw/spectrum_switchdev.c | 57 +- + drivers/net/ethernet/meta/Kconfig | 13 + + drivers/net/ethernet/meta/Makefile | 1 + + drivers/net/ethernet/meta/fbnic/fbnic.h | 2 + + drivers/net/ethernet/meta/fbnic/fbnic_devlink.c | 2 + + drivers/net/ethernet/meta/fbnic/fbnic_ethtool.c | 15 +- + drivers/net/ethernet/meta/fbnic/fbnic_fw.c | 9 +- + drivers/net/ethernet/meta/fbnic/fbnic_mac.c | 9 +- + drivers/net/ethernet/meta/fbnic/fbnic_pci.c | 2 - + drivers/net/ethernet/meta/fbnic/fbnic_rpc.c | 8 +- + drivers/net/ethernet/meta/fbnic/fbnic_tlv.c | 13 +- + drivers/net/ethernet/meta/fbnic/fbnic_txrx.c | 18 + + drivers/net/ethernet/meta/fbnic/fbnic_txrx.h | 10 + + drivers/net/ethernet/meta/mpnic/Makefile | 16 + + drivers/net/ethernet/meta/mpnic/mpnic.h | 66 + + drivers/net/ethernet/meta/mpnic/mpnic_csr.h | 319 +++++ + drivers/net/ethernet/meta/mpnic/mpnic_init.c | 556 ++++++++ + drivers/net/ethernet/meta/mpnic/mpnic_irq.c | 64 + + drivers/net/ethernet/meta/mpnic/mpnic_netdev.c | 186 +++ + drivers/net/ethernet/meta/mpnic/mpnic_netdev.h | 35 + + drivers/net/ethernet/meta/mpnic/mpnic_pci.c | 187 +++ + drivers/net/ethernet/meta/mpnic/mpnic_txrx.c | 1395 ++++++++++++++++++++ + drivers/net/ethernet/meta/mpnic/mpnic_txrx.h | 151 +++ + drivers/net/ethernet/microchip/lan865x/lan865x.c | 4 +- + drivers/net/ethernet/microchip/vcap/Kconfig | 1 - + drivers/net/ethernet/microsoft/mana/hw_channel.c | 24 +- + drivers/net/ethernet/microsoft/mana/mana_bpf.c | 3 - + drivers/net/ethernet/nebula-matrix/Kconfig | 32 + + drivers/net/ethernet/nebula-matrix/Makefile | 6 + + drivers/net/ethernet/nebula-matrix/nbl/Makefile | 7 + + drivers/net/ethernet/nebula-matrix/nbl/nbl_core.h | 31 + + .../nbl/nbl_hw/nbl_hw_leonis/nbl_hw_leonis.c | 153 +++ + .../nbl/nbl_hw/nbl_hw_leonis/nbl_hw_leonis.h | 15 + + .../ethernet/nebula-matrix/nbl/nbl_hw/nbl_hw_reg.h | 31 + + .../nebula-matrix/nbl/nbl_include/nbl_def_common.h | 31 + + .../nebula-matrix/nbl/nbl_include/nbl_def_hw.h | 17 + + .../nebula-matrix/nbl/nbl_include/nbl_include.h | 23 + + drivers/net/ethernet/nebula-matrix/nbl/nbl_main.c | 192 +++ + .../ethernet/netronome/nfp/flower/tunnel_conf.c | 14 +- + drivers/net/ethernet/oa_tc6.c | 46 +- + .../net/ethernet/oki-semi/pch_gbe/pch_gbe_main.c | 10 +- + .../net/ethernet/pensando/ionic/ionic_bus_pci.c | 1 + + .../net/ethernet/pensando/ionic/ionic_ethtool.c | 10 +- + drivers/net/ethernet/pensando/ionic/ionic_if.h | 31 +- + .../net/ethernet/qlogic/netxen/netxen_nic_ctx.c | 2 +- + drivers/net/ethernet/qlogic/qed/qed_cxt.c | 2 +- + drivers/net/ethernet/qlogic/qed/qed_mcp.h | 2 +- + drivers/net/ethernet/qlogic/qede/qede_ethtool.c | 6 +- + drivers/net/ethernet/qlogic/qla3xxx.c | 19 +- + drivers/net/ethernet/qlogic/qlcnic/qlcnic.h | 2 +- + .../net/ethernet/qlogic/qlcnic/qlcnic_83xx_init.c | 4 +- + drivers/net/ethernet/qlogic/qlcnic/qlcnic_ctx.c | 2 +- + drivers/net/ethernet/qlogic/qlcnic/qlcnic_dcb.c | 5 +- + drivers/net/ethernet/qlogic/qlcnic/qlcnic_main.c | 4 +- + .../ethernet/qlogic/qlcnic/qlcnic_sriov_common.c | 6 +- + .../net/ethernet/qlogic/qlcnic/qlcnic_sriov_pf.c | 6 +- + drivers/net/ethernet/qualcomm/rmnet/rmnet_config.c | 75 +- + drivers/net/ethernet/qualcomm/rmnet/rmnet_config.h | 2 +- + .../net/ethernet/qualcomm/rmnet/rmnet_handlers.c | 34 +- + drivers/net/ethernet/qualcomm/rmnet/rmnet_map.h | 9 +- + .../ethernet/qualcomm/rmnet/rmnet_map_command.c | 9 +- + .../net/ethernet/qualcomm/rmnet/rmnet_map_data.c | 14 +- + drivers/net/ethernet/qualcomm/rmnet/rmnet_vnd.c | 13 +- + drivers/net/ethernet/realtek/Kconfig | 2 +- + drivers/net/ethernet/realtek/r8169_main.c | 706 ++++++++-- + drivers/net/ethernet/renesas/ravb.h | 35 +- + drivers/net/ethernet/renesas/ravb_main.c | 253 ++-- + drivers/net/ethernet/renesas/ravb_ptp.c | 38 +- + drivers/net/ethernet/renesas/rswitch_main.c | 7 +- + drivers/net/ethernet/rocker/rocker_main.c | 2 +- + drivers/net/ethernet/rocker/rocker_ofdpa.c | 2 +- + drivers/net/ethernet/sfc/Kconfig | 2 +- + drivers/net/ethernet/sfc/tc_counters.c | 8 +- + drivers/net/ethernet/sfc/tc_encap_actions.c | 4 +- + drivers/net/ethernet/smsc/Kconfig | 3 +- + drivers/net/ethernet/smsc/smc9194.h | 241 ---- + drivers/net/ethernet/spacemit/k1_emac.c | 4 +- + drivers/net/ethernet/stmicro/stmmac/Kconfig | 11 + + drivers/net/ethernet/stmicro/stmmac/Makefile | 1 + + drivers/net/ethernet/stmicro/stmmac/common.h | 5 + + .../net/ethernet/stmicro/stmmac/dwmac-eic7700.c | 13 +- + drivers/net/ethernet/stmicro/stmmac/dwmac-sophgo.c | 10 +- + drivers/net/ethernet/stmicro/stmmac/dwmac-sun8i.c | 77 +- + .../net/ethernet/stmicro/stmmac/dwmac-ultrarisc.c | 52 + + drivers/net/ethernet/stmicro/stmmac/dwmac4.h | 6 +- + drivers/net/ethernet/stmicro/stmmac/dwmac4_core.c | 79 +- + drivers/net/ethernet/stmicro/stmmac/dwmac4_dma.c | 2 + + .../net/ethernet/stmicro/stmmac/dwxgmac2_core.c | 18 - + drivers/net/ethernet/stmicro/stmmac/hwif.h | 3 - + drivers/net/ethernet/stmicro/stmmac/stmmac.h | 7 + + .../net/ethernet/stmicro/stmmac/stmmac_ethtool.c | 3 +- + drivers/net/ethernet/stmicro/stmmac/stmmac_main.c | 176 ++- + .../net/ethernet/stmicro/stmmac/stmmac_platform.c | 17 +- + .../net/ethernet/stmicro/stmmac/stmmac_selftests.c | 112 -- + drivers/net/ethernet/sun/niu.c | 10 +- + drivers/net/ethernet/ti/am65-cpsw-nuss.c | 2 +- + drivers/net/ethernet/ti/cpsw.c | 2 +- + drivers/net/ethernet/ti/cpsw_new.c | 2 +- + drivers/net/ethernet/ti/davinci_mdio.c | 4 +- + drivers/net/ethernet/wangxun/libwx/wx_hw.c | 2 +- + drivers/net/ethernet/wangxun/libwx/wx_ptp.c | 5 +- + drivers/net/ethernet/wangxun/libwx/wx_type.h | 6 +- + drivers/net/ethernet/wangxun/txgbe/txgbe_phy.c | 6 +- + drivers/net/ethernet/wiznet/w5100.c | 234 +++- + drivers/net/ethernet/xilinx/ll_temac_main.c | 7 +- + drivers/net/ethernet/xilinx/xilinx_axienet_main.c | 12 +- + drivers/net/ethernet/xscale/ptp_ixp46x.c | 1 + + drivers/net/fddi/skfp/cfm.c | 2 +- + drivers/net/fddi/skfp/ecm.c | 2 +- + drivers/net/fddi/skfp/fplustm.c | 4 +- + drivers/net/fddi/skfp/h/hwmtm.h | 2 +- + drivers/net/fddi/skfp/h/sba.h | 2 +- + drivers/net/fddi/skfp/pcmplc.c | 2 +- + drivers/net/fddi/skfp/rmt.c | 2 +- + drivers/net/gtp.c | 2 +- + drivers/net/hyperv/netvsc.c | 10 +- + drivers/net/ieee802154/at86rf230.c | 2 +- + drivers/net/macsec.c | 32 + + drivers/net/mdio/Kconfig | 4 +- + drivers/net/mdio/mdio-bcm-iproc.c | 2 +- + drivers/net/mdio/mdio-bcm-unimac.c | 2 +- + drivers/net/mdio/mdio-mux-meson-g12a.c | 2 +- + drivers/net/mdio/mdio-realtek-rtl9300.c | 426 +++++- + drivers/net/netconsole.c | 132 +- + drivers/net/netdevsim/bus.c | 7 +- + drivers/net/netdevsim/ethtool.c | 53 + + drivers/net/netdevsim/ipsec.c | 10 +- + drivers/net/netdevsim/netdev.c | 10 +- + drivers/net/netdevsim/netdevsim.h | 8 +- + drivers/net/netdevsim/psp.c | 33 - + drivers/net/netkit.c | 17 +- + drivers/net/pcs/pcs-lynx.c | 59 +- + drivers/net/pcs/pcs-xpcs-plat.c | 2 +- + drivers/net/phy/Kconfig | 11 + + drivers/net/phy/Makefile | 3 +- + drivers/net/phy/air_en8811h.c | 84 +- + drivers/net/phy/broadcom.c | 7 + + drivers/net/phy/dp83848.c | 3 + + drivers/net/phy/dp83867.c | 109 +- + drivers/net/phy/mediatek/mtk-phy-lib.c | 26 +- + drivers/net/phy/microchip_t1.c | 16 +- + drivers/net/phy/motorcomm.c | 84 +- + drivers/net/phy/mxl-gpy.c | 2 +- + drivers/net/phy/nxp-c45-tja11xx.c | 2 +- + drivers/net/phy/phy_device.c | 360 +++-- + drivers/net/phy/phy_fixup.c | 98 ++ + drivers/net/phy/phy_led_triggers.c | 11 +- + drivers/net/phy/phylib-internal.h | 1 + + drivers/net/phy/phylink.c | 176 ++- + drivers/net/phy/qcom/qca808x.c | 15 +- + drivers/net/phy/realtek/realtek_main.c | 103 +- + drivers/net/phy/sfp.c | 21 +- + drivers/net/phy/xpowers/Makefile | 3 + + drivers/net/phy/xpowers/ac200.c | 314 +++++ + drivers/net/phy/xpowers/ac300.c | 387 ++++++ + drivers/net/phy/xpowers/acx00.c | 633 +++++++++ + drivers/net/phy/xpowers/acx00.h | 28 + + drivers/net/ppp/ppp_synctty.c | 20 +- + drivers/net/usb/ax88179_178a.c | 9 +- + drivers/net/usb/ch9200.c | 14 +- + drivers/net/usb/qmi_wwan.c | 1 + + drivers/net/usb/r8152.c | 51 +- + drivers/net/virtio_net.c | 33 +- + drivers/net/vrf.c | 2 +- + drivers/net/vxlan/vxlan_core.c | 780 ++++++----- + drivers/net/vxlan/vxlan_mdb.c | 45 +- + drivers/net/vxlan/vxlan_multicast.c | 94 +- + drivers/net/vxlan/vxlan_private.h | 34 +- + drivers/net/vxlan/vxlan_vnifilter.c | 206 ++- + drivers/net/wan/slic_ds26522.c | 2 +- + drivers/net/wireless/ath/ath10k/debug.c | 10 +- + drivers/net/wireless/ath/ath10k/snoc.c | 6 +- + drivers/net/wireless/ath/ath11k/Kconfig | 3 + + drivers/net/wireless/ath/ath11k/ahb.c | 16 +- + drivers/net/wireless/ath/ath11k/ce.c | 2 +- + drivers/net/wireless/ath/ath11k/core.c | 10 + + drivers/net/wireless/ath/ath11k/dp.c | 2 + + drivers/net/wireless/ath/ath11k/dp_rx.c | 21 +- + drivers/net/wireless/ath/ath11k/htc.c | 5 +- + drivers/net/wireless/ath/ath12k/ahb.c | 166 ++- + drivers/net/wireless/ath/ath12k/ahb.h | 9 + + drivers/net/wireless/ath/ath12k/ce.c | 2 +- + drivers/net/wireless/ath/ath12k/core.c | 23 + + drivers/net/wireless/ath/ath12k/dp.c | 2 +- + drivers/net/wireless/ath/ath12k/dp_mon.c | 2 - + drivers/net/wireless/ath/ath12k/dp_rx.c | 20 +- + drivers/net/wireless/ath/ath12k/htc.c | 5 +- + drivers/net/wireless/ath/ath12k/hw.h | 20 +- + drivers/net/wireless/ath/ath12k/mac.c | 45 +- + drivers/net/wireless/ath/ath12k/pci.c | 55 +- + drivers/net/wireless/ath/ath12k/qmi.c | 4 + + drivers/net/wireless/ath/ath12k/wifi7/ahb.c | 2 + + drivers/net/wireless/ath/ath12k/wifi7/dp_mon.c | 23 +- + drivers/net/wireless/ath/ath12k/wifi7/dp_rx.c | 20 +- + drivers/net/wireless/ath/ath12k/wifi7/dp_rx.h | 12 +- + drivers/net/wireless/ath/ath12k/wifi7/dp_tx.c | 17 +- + drivers/net/wireless/ath/ath12k/wifi7/hal_tx.c | 2 +- + drivers/net/wireless/ath/ath12k/wifi7/hw.c | 9 +- + drivers/net/wireless/ath/ath12k/wmi.c | 22 +- + drivers/net/wireless/ath/ath9k/ar9003_calib.c | 5 +- + drivers/net/wireless/ath/ath9k/mci.c | 5 +- + drivers/net/wireless/ath/ath9k/xmit.c | 2 +- + drivers/net/wireless/broadcom/b43legacy/dma.c | 24 +- + drivers/net/wireless/broadcom/b43legacy/dma.h | 4 +- + .../broadcom/brcm80211/brcmfmac/cfg80211.c | 129 +- + .../broadcom/brcm80211/brcmfmac/cfg80211.h | 5 +- + .../wireless/broadcom/brcm80211/brcmfmac/feature.c | 2 + + .../wireless/broadcom/brcm80211/brcmfmac/feature.h | 4 +- + drivers/net/wireless/intel/iwlwifi/mld/agg.c | 17 +- + drivers/net/wireless/intel/iwlwifi/mld/agg.h | 4 +- + drivers/net/wireless/intel/iwlwifi/mld/rx.c | 50 +- + drivers/net/wireless/intel/iwlwifi/mld/rx.h | 2 +- + drivers/net/wireless/intel/iwlwifi/mld/tests/agg.c | 7 +- + drivers/net/wireless/intel/iwlwifi/mvm/rx.c | 2 +- + drivers/net/wireless/intel/iwlwifi/mvm/rxmq.c | 6 +- + drivers/net/wireless/marvell/libertas/if_usb.c | 10 +- + drivers/net/wireless/marvell/libertas/mesh.c | 506 ------- + drivers/net/wireless/marvell/libertas_tf/if_usb.c | 10 +- + .../net/wireless/marvell/mwifiex/11n_rxreorder.c | 20 +- + drivers/net/wireless/marvell/mwifiex/cfg80211.c | 7 + + drivers/net/wireless/marvell/mwifiex/main.h | 2 +- + drivers/net/wireless/mediatek/mt76/mac80211.c | 25 +- + .../net/wireless/mediatek/mt76/mt76_connac_mcu.c | 5 +- + drivers/net/wireless/mediatek/mt76/mt7925/mcu.c | 5 +- + drivers/net/wireless/ralink/rt2x00/rt2x00pci.c | 104 +- + drivers/net/wireless/ralink/rt2x00/rt2x00usb.c | 67 +- + drivers/net/wireless/realtek/rtw88/debug.c | 5 +- + drivers/net/wireless/realtek/rtw89/core.c | 12 +- + drivers/net/wireless/realtek/rtw89/mac.c | 10 +- + drivers/net/wireless/realtek/rtw89/regd.c | 5 +- + drivers/net/wireless/silabs/wfx/bh.c | 6 +- + drivers/net/wireless/silabs/wfx/bus_sdio.c | 1 + + drivers/net/wireless/silabs/wfx/main.c | 37 +- + drivers/net/wireless/ti/wlcore/rx.c | 5 +- + drivers/net/wireless/virtual/mac80211_hwsim.h | 5 +- + drivers/net/wireless/virtual/mac80211_hwsim_main.c | 98 +- + drivers/ptp/ptp_idt82p33.c | 25 +- + drivers/ptp/ptp_idt82p33.h | 1 + + drivers/ptp/ptp_ocp.c | 939 ++++++++++++- + drivers/usb/atm/cxacru.c | 18 +- + drivers/usb/atm/speedtch.c | 10 +- + include/linux/bnge/hsi.h | 136 +- + include/linux/ieee80211-eht.h | 18 +- + include/linux/ieee80211-uhr.h | 101 +- + include/linux/ieee80211.h | 3 +- + include/linux/igmp.h | 3 +- + include/linux/mdio-mux.h | 2 +- + include/linux/mlx5/eswitch.h | 3 + + include/linux/net.h | 34 +- + include/linux/netdevice.h | 23 +- + include/linux/netpoll.h | 5 - + include/linux/phy.h | 28 +- + include/linux/phylink.h | 2 + + include/linux/platform_data/microchip-ksz.h | 2 + + include/linux/skbuff.h | 7 +- + include/linux/stmmac.h | 2 + + include/net/arp.h | 10 +- + include/net/bond_alb.h | 14 +- + include/net/cfg80211.h | 6 +- + include/net/cfg802154.h | 9 +- + include/net/dropreason-core.h | 6 + + include/net/dsa.h | 2 + + include/net/ieee80211_radiotap.h | 68 +- + include/net/inet_connection_sock.h | 1 + + include/net/ip6_tunnel.h | 1 + + include/net/ip_tunnels.h | 7 +- + include/net/mac80211.h | 66 +- + include/net/mana/hw_channel.h | 8 +- + include/net/ndisc.h | 20 +- + include/net/neighbour.h | 22 +- + include/net/net_namespace.h | 4 + + include/net/netdev_lock.h | 20 +- + include/net/netfilter/nf_conntrack_seqadj.h | 1 + + include/net/netmem.h | 27 + + include/net/netns/netfilter.h | 2 + + include/net/page_pool/helpers.h | 12 +- + include/net/phy/realtek_phy.h | 7 - + include/net/psp/functions.h | 5 - + include/net/psp/types.h | 4 + + include/net/route.h | 7 +- + include/net/tcp.h | 15 +- + include/net/vxlan.h | 12 +- + include/uapi/linux/atm.h | 2 +- + include/uapi/linux/dpll.h | 3 +- + include/uapi/linux/handshake.h | 5 +- + include/uapi/linux/if_link.h | 1 + + include/uapi/linux/mii.h | 2 +- + include/uapi/linux/netdev.h | 2 +- + include/uapi/linux/netfilter/nfnetlink_conntrack.h | 1 + + include/uapi/linux/netlink.h | 14 + + include/uapi/linux/nl80211.h | 27 +- + net/6lowpan/debugfs.c | 10 +- + net/8021q/vlan.c | 8 +- + net/8021q/vlan.h | 22 +- + net/8021q/vlan_core.c | 2 +- + net/8021q/vlan_dev.c | 37 +- + net/8021q/vlan_netlink.c | 52 +- + net/8021q/vlanproc.c | 18 +- + net/Kconfig | 1 - + net/atm/proc.c | 7 +- + net/batman-adv/bat_iv_ogm.c | 29 +- + net/batman-adv/bat_v.c | 42 +- + net/batman-adv/bridge_loop_avoidance.c | 6 +- + net/batman-adv/distributed-arp-table.c | 10 +- + net/batman-adv/hash.h | 2 +- + net/batman-adv/mesh-interface.c | 12 +- + net/batman-adv/multicast.c | 6 +- + net/batman-adv/translation-table.c | 1118 ++++++++++------ + net/batman-adv/types.h | 49 +- + net/bridge/br.c | 5 +- + net/bridge/br_arp_nd_proxy.c | 57 +- + net/bridge/br_device.c | 8 +- + net/bridge/br_forward.c | 276 ++-- + net/bridge/br_input.c | 4 +- + net/bridge/br_mst.c | 28 +- + net/bridge/br_multicast.c | 18 +- + net/bridge/br_netlink.c | 10 +- + net/bridge/br_netlink_tunnel.c | 14 +- + net/bridge/br_private.h | 94 +- + net/bridge/br_sysfs_if.c | 10 +- + net/bridge/br_vlan.c | 188 +-- + net/bridge/br_vlan_options.c | 20 +- + net/bridge/br_vlan_tunnel.c | 2 +- + net/core/dev.c | 37 + + net/core/dev.h | 2 + + net/core/devmem.c | 229 ++-- + net/core/devmem.h | 44 +- + net/core/gro_cells.c | 5 +- + net/core/neighbour.c | 428 +++--- + net/core/net-sysfs.c | 112 +- + net/core/netdev-genl.c | 26 +- + net/core/netdev_work.c | 15 +- + net/core/page_pool_priv.h | 14 +- + net/core/page_pool_user.c | 8 +- + net/core/pktgen.c | 17 +- + net/core/rtnetlink.c | 22 + + net/core/skbuff.c | 32 +- + net/core/sock.c | 4 +- + net/core/sysctl_net_core.c | 17 +- + net/devlink/dev.c | 4 +- + net/devlink/netlink_gen.c | 9 +- + net/devlink/netlink_gen.h | 2 +- + net/devlink/port.c | 19 +- + net/dsa/Kconfig | 6 + + net/dsa/Makefile | 1 + + net/dsa/tag_ks8995.c | 180 +++ + net/dsa/tag_ksz.c | 5 +- + net/ethtool/common.c | 174 +++ + net/ethtool/ioctl.c | 9 - + net/ethtool/ts.h | 8 +- + net/handshake/genl.c | 4 +- + net/handshake/handshake-test.c | 2 +- + net/handshake/request.c | 2 +- + net/hsr/hsr_device.c | 2 +- + net/hsr/hsr_forward.c | 2 +- + net/hsr/hsr_netlink.c | 37 +- + net/ieee802154/6lowpan/tx.c | 3 +- + net/ieee802154/socket.c | 5 +- + net/ipv4/af_inet.c | 3 - + net/ipv4/arp.c | 137 +- + net/ipv4/bpf_tcp_ca.c | 5 +- + net/ipv4/devinet.c | 18 +- + net/ipv4/fib_semantics.c | 7 +- + net/ipv4/fou_nl.c | 6 +- + net/ipv4/igmp.c | 23 +- + net/ipv4/inet_connection_sock.c | 1 + + net/ipv4/inet_hashtables.c | 11 + + net/ipv4/inet_timewait_sock.c | 6 +- + net/ipv4/ip_forward.c | 2 +- + net/ipv4/ip_gre.c | 6 +- + net/ipv4/ip_sockglue.c | 10 +- + net/ipv4/ip_tunnel.c | 146 +- + net/ipv4/ip_vti.c | 2 +- + net/ipv4/ipip.c | 2 +- + net/ipv4/ipmr.c | 17 +- + net/ipv4/netfilter/arp_tables.c | 2 +- + net/ipv4/ping.c | 5 +- + net/ipv4/raw.c | 4 +- + net/ipv4/route.c | 2 +- + net/ipv4/tcp.c | 9 +- + net/ipv4/tcp_bbr.c | 13 +- + net/ipv4/tcp_input.c | 19 +- + net/ipv4/tcp_ipv4.c | 13 +- + net/ipv4/tcp_output.c | 28 +- + net/ipv4/udp.c | 26 +- + net/ipv6/addrconf.c | 17 +- + net/ipv6/af_inet6.c | 4 - + net/ipv6/ah6.c | 2 +- + net/ipv6/datagram.c | 5 +- + net/ipv6/exthdrs.c | 6 +- + net/ipv6/ip6_fib.c | 14 +- + net/ipv6/ip6_gre.c | 268 ++-- + net/ipv6/ip6_output.c | 6 +- + net/ipv6/ip6_vti.c | 2 +- + net/ipv6/ip6mr.c | 5 +- + net/ipv6/ipv6_sockglue.c | 123 +- + net/ipv6/ndisc.c | 193 +-- + net/ipv6/route.c | 63 +- + net/ipv6/sit.c | 480 ++++--- + net/ipv6/tcp_ipv6.c | 12 +- + net/ipv6/udp.c | 8 +- + net/iucv/af_iucv.c | 20 +- + net/key/af_key.c | 5 +- + net/llc/llc_conn.c | 2 +- + net/mac80211/agg-tx.c | 9 +- + net/mac80211/ap.c | 3 - + net/mac80211/debugfs.c | 2 +- + net/mac80211/debugfs_netdev.c | 8 +- + net/mac80211/ht.c | 2 +- + net/mac80211/ieee80211_i.h | 30 +- + net/mac80211/iface.c | 12 +- + net/mac80211/main.c | 13 +- + net/mac80211/mesh_pathtbl.c | 2 +- + net/mac80211/mlme.c | 48 +- + net/mac80211/offchannel.c | 13 +- + net/mac80211/rate.c | 1 - + net/mac80211/rx.c | 470 ++++--- + net/mac80211/s1g.c | 8 - + net/mac80211/scan.c | 10 +- + net/mac80211/tdls.c | 4 +- + net/mac80211/tests/chan-mode.c | 6 +- + net/mac80211/tests/util.c | 6 +- + net/mac80211/tx.c | 427 +++--- + net/mac80211/util.c | 2 +- + net/mac802154/iface.c | 10 +- + net/mptcp/options.c | 6 +- + net/mptcp/protocol.h | 45 +- + net/mptcp/subflow.c | 18 +- + net/ncsi/ncsi-rsp.c | 10 +- + net/netfilter/core.c | 18 +- + net/netfilter/ipset/ip_set_core.c | 2 +- + net/netfilter/ipvs/ip_vs_sync.c | 2 +- + net/netfilter/nf_conncount.c | 2 +- + net/netfilter/nf_conntrack_core.c | 5 +- + net/netfilter/nf_conntrack_netlink.c | 19 +- + net/netfilter/nf_conntrack_ovs.c | 2 +- + net/netfilter/nf_conntrack_seqadj.c | 28 +- + net/netfilter/nf_conntrack_standalone.c | 2 +- + net/netfilter/nf_flow_table_core.c | 17 +- + net/netfilter/nf_hooks_lwtunnel.c | 2 +- + net/netfilter/nf_log.c | 2 +- + net/netfilter/nf_nat_core.c | 16 +- + net/netfilter/nf_synproxy_core.c | 6 +- + net/netfilter/nfnetlink_acct.c | 2 +- + net/netfilter/nfnetlink_cthelper.c | 3 +- + net/netfilter/nfnetlink_cttimeout.c | 8 +- + net/netfilter/nfnetlink_hook.c | 74 +- + net/netfilter/nfnetlink_osf.c | 3 +- + net/netfilter/nft_ct.c | 5 +- + net/netfilter/nft_set_pipapo.c | 6 +- + net/netfilter/xt_CT.c | 6 +- + net/netfilter/xt_IDLETIMER.c | 8 +- + net/netfilter/xt_LED.c | 5 +- + net/netfilter/xt_RATEEST.c | 2 +- + net/netfilter/xt_TEE.c | 2 +- + net/netfilter/xt_hashlimit.c | 4 +- + net/netfilter/xt_limit.c | 2 +- + net/netfilter/xt_quota.c | 2 +- + net/netfilter/xt_recent.c | 3 +- + net/netfilter/xt_statistic.c | 2 +- + net/netfilter/xt_string.c | 2 +- + net/netlink/af_netlink.c | 8 +- + net/netlink/policy.c | 16 +- + net/openvswitch/datapath.c | 14 +- + net/openvswitch/datapath.h | 2 +- + net/openvswitch/vport-internal_dev.c | 2 + + net/packet/af_packet.c | 7 +- + net/phonet/socket.c | 5 +- + net/psp/psp.h | 13 + + net/psp/psp_main.c | 29 +- + net/psp/psp_nl.c | 51 +- + net/psp/psp_sock.c | 115 +- + net/rds/ib.h | 4 +- + net/rds/ib_cm.c | 29 +- + net/sched/act_api.c | 15 +- + net/sched/act_connmark.c | 6 + + net/sched/act_ct.c | 2 +- + net/sched/act_mirred.c | 3 +- + net/sched/act_mpls.c | 11 + + net/sched/act_nat.c | 6 + + net/sched/act_simple.c | 7 + + net/sched/act_skbmod.c | 9 + + net/sched/cls_flower.c | 39 +- + net/sched/ematch.c | 9 +- + net/sched/sch_api.c | 92 +- + net/sched/sch_fq.c | 71 +- + net/sctp/proc.c | 6 +- + net/socket.c | 32 +- + net/tipc/bearer.c | 6 +- + net/tipc/bearer.h | 3 +- + net/tipc/monitor.c | 3 +- + net/tls/tls_main.c | 2 - + net/unix/af_unix.c | 3 +- + net/vmw_vsock/af_vsock.c | 35 + + net/vmw_vsock/af_vsock_tap.c | 4 - + net/wireless/chan.c | 11 +- + net/wireless/nl80211.c | 38 +- + net/x25/af_x25.c | 5 +- + rust/kernel/net/netlink.rs | 8 +- + tools/include/uapi/linux/netdev.h | 2 +- + tools/net/ynl/Makefile | 1 + + tools/net/ynl/Makefile.deps | 3 +- + tools/net/ynl/generated/Makefile | 3 +- + tools/net/ynl/lib/Makefile | 3 +- + tools/net/ynl/pyynl/lib/ynl.py | 8 +- + tools/net/ynl/pyynl/ynl_gen_c.py | 15 +- + tools/net/ynl/tests/Makefile | 2 +- + tools/net/ynl/ynltool/Makefile | 10 +- + tools/testing/selftests/bpf/progs/tcp_ca_kfunc.c | 8 +- + tools/testing/selftests/drivers/net/Makefile | 11 +- + tools/testing/selftests/drivers/net/README.rst | 9 + + .../testing/selftests/drivers/net/bonding/Makefile | 3 +- + tools/testing/selftests/drivers/net/dsa/Makefile | 1 + + tools/testing/selftests/drivers/net/gro_hw.py | 13 + + .../selftests/drivers/net/{gro.py => gro_lib.py} | 190 +-- + tools/testing/selftests/drivers/net/gro_lro.py | 14 + + tools/testing/selftests/drivers/net/gro_sw.py | 13 + + tools/testing/selftests/drivers/net/hds.py | 29 - + tools/testing/selftests/drivers/net/hw/Makefile | 18 +- + tools/testing/selftests/drivers/net/hw/config | 1 + + .../selftests/drivers/net/hw/devlink_rate_tc_bw.py | 5 +- + .../testing/selftests/drivers/net/hw/devmem_lib.py | 4 +- + .../drivers/net/hw/{gro_hw.py => gro_stats.py} | 0 + tools/testing/selftests/drivers/net/hw/iou-zcrx.c | 422 ++++-- + tools/testing/selftests/drivers/net/hw/iou-zcrx.py | 50 +- + .../selftests/drivers/net/hw/lib/py/__init__.py | 7 +- + tools/testing/selftests/drivers/net/hw/toeplitz.py | 10 +- + tools/testing/selftests/drivers/net/hw/tso.py | 161 +++ + tools/testing/selftests/drivers/net/hw/vlan.py | 127 ++ + .../selftests/drivers/net/lib/py/__init__.py | 7 +- + tools/testing/selftests/drivers/net/lib/py/env.py | 19 + + tools/testing/selftests/drivers/net/lib/py/feat.py | 43 + + .../selftests/drivers/net/netconsole/Makefile | 3 +- + .../selftests/drivers/net/netdevsim/Makefile | 8 + + tools/testing/selftests/drivers/net/pppoe_gro.py | 45 + + tools/testing/selftests/drivers/net/psp.py | 180 ++- + .../testing/selftests/drivers/net/ring_reconfig.py | 26 +- + tools/testing/selftests/drivers/net/rss_key.py | 343 +++++ + tools/testing/selftests/drivers/net/ruff.toml | 4 + + tools/testing/selftests/drivers/net/settings | 2 +- + tools/testing/selftests/drivers/net/so_txtime.c | 161 ++- + tools/testing/selftests/drivers/net/so_txtime.py | 80 +- + tools/testing/selftests/drivers/net/team/Makefile | 4 +- + .../selftests/drivers/net/virtio_net/Makefile | 1 + + tools/testing/selftests/net/Makefile | 1 + + tools/testing/selftests/net/bind_wildcard.c | 5 + + tools/testing/selftests/net/forwarding/Makefile | 2 + + tools/testing/selftests/net/fou_mcast_encap.sh | 11 +- + tools/testing/selftests/net/icmp_redirect.sh | 10 +- + tools/testing/selftests/net/ip_local_port_range.c | 52 +- + tools/testing/selftests/net/lib/csum.c | 55 +- + tools/testing/selftests/net/lib/gro.c | 136 +- + .../selftests/net/lib/ksft_setup_loopback.sh | 2 +- + tools/testing/selftests/net/lib/py/__init__.py | 4 +- + tools/testing/selftests/net/lib/py/utils.py | 22 + + tools/testing/selftests/net/mptcp/config | 2 +- + .../selftests/net/ndisc_unsolicited_na_test.sh | 197 ++- + tools/testing/selftests/net/netdev_lock.py | 42 + + tools/testing/selftests/net/netfilter/Makefile | 2 + + tools/testing/selftests/net/netfilter/config | 2 +- + .../selftests/net/netfilter/conntrack_dump_flush.c | 145 +- + .../net/netfilter/conntrack_icmp_related.sh | 2 +- + tools/testing/selftests/net/openvswitch/config | 3 + + .../selftests/net/openvswitch/openvswitch.sh | 202 +++ + .../packetdrill/tcp_rcv_ssthresh_scaling_ratio.pkt | 46 + + tools/testing/selftests/net/rtnetlink.py | 113 +- + tools/testing/selftests/net/rtnetlink.sh | 8 +- + tools/testing/selftests/net/ruff.toml | 4 + + tools/testing/selftests/net/so_incoming_cpu.c | 7 +- + .../selftests/net/srv6_end_dt4_l3vpn_test.sh | 21 + + .../selftests/net/srv6_end_dt6_l3vpn_test.sh | 21 + + tools/testing/selftests/net/test_neigh.sh | 66 +- + tools/testing/selftests/net/tun.c | 9 + + tools/testing/selftests/net/xfrm_policy.sh | 2 +- + tools/testing/vsock/util.c | 9 + + tools/testing/vsock/util.h | 1 + + tools/testing/vsock/vsock_uring_test.c | 391 ++++++ + 903 files changed, 27189 insertions(+), 10462 deletions(-) + create mode 100644 Documentation/devicetree/bindings/net/snps,dwmac-common.yaml + create mode 100644 Documentation/devicetree/bindings/net/ultrarisc,dp1000-gmac.yaml + delete mode 100644 Documentation/devicetree/bindings/net/wireless/ti,wl1251.txt + create mode 100644 Documentation/devicetree/bindings/net/wireless/ti,wl1251.yaml + create mode 100644 Documentation/devicetree/bindings/net/wiznet,w5100.yaml + delete mode 100644 Documentation/devicetree/bindings/net/wiznet,w5x00.txt + create mode 100644 Documentation/devicetree/bindings/net/x-powers,acx00-ephy-package.yaml + create mode 100644 Documentation/networking/device_drivers/ethernet/nebula-matrix/nbl.rst + create mode 100644 Documentation/networking/devlink/ptp_ocp.rst + delete mode 100644 drivers/net/dsa/ks8995.c + create mode 100644 drivers/net/dsa/motorcomm/Kconfig + create mode 100644 drivers/net/dsa/motorcomm/Makefile + rename drivers/net/dsa/{yt921x.c => motorcomm/chip.c} (88%) + rename drivers/net/dsa/{yt921x.h => motorcomm/chip.h} (94%) + create mode 100644 drivers/net/dsa/motorcomm/leds.c + create mode 100644 drivers/net/dsa/motorcomm/leds.h + create mode 100644 drivers/net/dsa/motorcomm/mdio_bus.c + create mode 100644 drivers/net/dsa/motorcomm/mdio_bus.h + create mode 100644 drivers/net/dsa/motorcomm/pcs-921x.c + create mode 100644 drivers/net/dsa/motorcomm/pcs.h + create mode 100644 drivers/net/dsa/motorcomm/smi.c + create mode 100644 drivers/net/dsa/motorcomm/smi.h + rename drivers/net/ethernet/intel/ice/{ice_ethtool_fdir.c => ice_ethtool_ntuple.c} (96%) + create mode 100644 drivers/net/ethernet/meta/mpnic/Makefile + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic.h + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_csr.h + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_init.c + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_irq.c + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_netdev.c + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_netdev.h + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_pci.c + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_txrx.c + create mode 100644 drivers/net/ethernet/meta/mpnic/mpnic_txrx.h + create mode 100644 drivers/net/ethernet/nebula-matrix/Kconfig + create mode 100644 drivers/net/ethernet/nebula-matrix/Makefile + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/Makefile + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_core.h + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_hw/nbl_hw_leonis/nbl_hw_leonis.c + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_hw/nbl_hw_leonis/nbl_hw_leonis.h + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_hw/nbl_hw_reg.h + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_include/nbl_def_common.h + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_include/nbl_def_hw.h + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_include/nbl_include.h + create mode 100644 drivers/net/ethernet/nebula-matrix/nbl/nbl_main.c + delete mode 100644 drivers/net/ethernet/smsc/smc9194.h + create mode 100644 drivers/net/ethernet/stmicro/stmmac/dwmac-ultrarisc.c + create mode 100644 drivers/net/phy/phy_fixup.c + create mode 100644 drivers/net/phy/xpowers/Makefile + create mode 100644 drivers/net/phy/xpowers/ac200.c + create mode 100644 drivers/net/phy/xpowers/ac300.c + create mode 100644 drivers/net/phy/xpowers/acx00.c + create mode 100644 drivers/net/phy/xpowers/acx00.h + delete mode 100644 include/net/phy/realtek_phy.h + create mode 100644 net/dsa/tag_ks8995.c + create mode 100755 tools/testing/selftests/drivers/net/gro_hw.py + rename tools/testing/selftests/drivers/net/{gro.py => gro_lib.py} (74%) + mode change 100755 => 100644 + create mode 100755 tools/testing/selftests/drivers/net/gro_lro.py + create mode 100755 tools/testing/selftests/drivers/net/gro_sw.py + rename tools/testing/selftests/drivers/net/hw/{gro_hw.py => gro_stats.py} (100%) + create mode 100755 tools/testing/selftests/drivers/net/hw/vlan.py + create mode 100644 tools/testing/selftests/drivers/net/lib/py/feat.py + create mode 100755 tools/testing/selftests/drivers/net/pppoe_gro.py + create mode 100755 tools/testing/selftests/drivers/net/rss_key.py + create mode 100644 tools/testing/selftests/drivers/net/ruff.toml + create mode 100755 tools/testing/selftests/net/netdev_lock.py + create mode 100644 tools/testing/selftests/net/packetdrill/tcp_rcv_ssthresh_scaling_ratio.pkt + create mode 100644 tools/testing/selftests/net/ruff.toml +Merging bpf-next/for-next (acff58e305175 libbpf: Fix cleanup on invalid CO-RE relocation offsets) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf-next.git bpf-next/for-next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging MAINTAINERS +Auto-merging arch/arm64/net/bpf_jit_comp.c +CONFLICT (content): Merge conflict in arch/arm64/net/bpf_jit_comp.c +Auto-merging arch/riscv/net/bpf_jit_comp64.c +Auto-merging arch/x86/Kconfig +Auto-merging arch/x86/net/bpf_jit_comp.c +CONFLICT (content): Merge conflict in arch/x86/net/bpf_jit_comp.c +Auto-merging fs/bpf_fs_kfuncs.c +Auto-merging fs/exec.c +Auto-merging include/linux/bpf.h +Auto-merging include/linux/mm.h +Auto-merging include/net/tcp.h +Auto-merging kernel/bpf/arena.c +Auto-merging kernel/bpf/core.c +Auto-merging kernel/bpf/hashtab.c +Auto-merging kernel/bpf/syscall.c +Auto-merging kernel/bpf/trampoline.c +Auto-merging mm/internal.h +CONFLICT (content): Merge conflict in mm/internal.h +Auto-merging mm/memory.c +Auto-merging mm/nommu.c +Auto-merging mm/util.c +Auto-merging net/ipv4/af_inet.c +Auto-merging net/ipv4/bpf_tcp_ca.c +Auto-merging net/ipv4/tcp.c +Auto-merging net/ipv4/tcp_input.c +Auto-merging net/ipv4/tcp_output.c +Auto-merging tools/testing/selftests/bpf/testing_helpers.c +Auto-merging tools/testing/selftests/bpf/testing_helpers.h +Resolved 'arch/arm64/net/bpf_jit_comp.c' using previous resolution. +Resolved 'arch/x86/net/bpf_jit_comp.c' using previous resolution. +Resolved 'mm/internal.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 1d2642c1a986c] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/bpf/bpf-next.git +$ git diff -M --stat --summary HEAD^.. + Documentation/admin-guide/kernel-parameters.txt | 16 + + Documentation/bpf/bpf_design_QA.rst | 11 +- + Documentation/bpf/btf.rst | 85 +- + Documentation/bpf/clang-notes.rst | 7 +- + Documentation/bpf/kfuncs.rst | 126 +- + Documentation/bpf/linux-notes.rst | 31 +- + Documentation/bpf/signing.rst | 283 +- + MAINTAINERS | 2 - + arch/arm64/net/bpf_jit_comp.c | 172 +- + arch/mips/include/asm/uasm.h | 8 + + arch/mips/mm/uasm-mips.c | 12 + + arch/mips/mm/uasm.c | 24 +- + arch/mips/net/bpf_jit_comp.c | 57 +- + arch/mips/net/bpf_jit_comp.h | 6 +- + arch/mips/net/bpf_jit_comp32.c | 109 +- + arch/mips/net/bpf_jit_comp64.c | 139 +- + arch/parisc/net/bpf_jit.h | 2 + + arch/parisc/net/bpf_jit_comp32.c | 29 +- + arch/parisc/net/bpf_jit_comp64.c | 35 +- + arch/parisc/net/bpf_jit_core.c | 14 + + arch/riscv/net/bpf_jit.h | 4 + + arch/riscv/net/bpf_jit_comp64.c | 109 +- + arch/riscv/net/bpf_jit_core.c | 8 + + arch/riscv/net/bpf_timed_may_goto.S | 8 +- + arch/x86/Kconfig | 1 + + arch/x86/net/bpf_jit_comp.c | 501 ++- + arch/x86/net/bpf_timed_may_goto.S | 10 +- + crypto/Makefile | 3 - + crypto/bpf_crypto_skcipher.c | 83 - + drivers/net/ethernet/netronome/nfp/bpf/verifier.c | 2 +- + fs/bpf_fs_kfuncs.c | 99 + + fs/exec.c | 7 +- + include/linux/bpf-cgroup-defs.h | 1 + + include/linux/bpf-cgroup.h | 28 + + include/linux/bpf.h | 169 +- + include/linux/bpf_crypto.h | 24 - + include/linux/bpf_lsm.h | 11 +- + include/linux/bpf_verifier.h | 297 +- + include/linux/btf.h | 23 +- + include/linux/filter.h | 45 + + include/linux/key.h | 2 + + include/linux/mm.h | 8 +- + include/net/tcp.h | 156 +- + include/uapi/linux/bpf.h | 82 +- + include/uapi/linux/btf.h | 83 +- + include/uapi/linux/keyctl.h | 1 + + kernel/bpf/Kconfig | 26 + + kernel/bpf/Makefile | 7 +- + kernel/bpf/arena.c | 83 +- + kernel/bpf/arraymap.c | 2 +- + kernel/bpf/backtrack.c | 153 +- + kernel/bpf/bpf_insn_array.c | 19 +- + kernel/bpf/bpf_local_storage.c | 15 +- + kernel/bpf/bpf_lsm.c | 26 +- + kernel/bpf/bpf_lsm_proto.c | 19 +- + kernel/bpf/bpf_struct_ops.c | 142 +- + kernel/bpf/btf.c | 645 ++- + kernel/bpf/cfg.c | 309 +- + kernel/bpf/cgroup.c | 502 ++- + kernel/bpf/cnum_defs.h | 37 +- + kernel/bpf/const_fold.c | 6 +- + kernel/bpf/core.c | 137 +- + kernel/bpf/crypto.c | 288 +- + kernel/bpf/diagnostics.c | 71 +- + kernel/bpf/diagnostics.h | 1 + + kernel/bpf/disasm.c | 5 +- + kernel/bpf/fixups.c | 341 +- + kernel/bpf/hashtab.c | 63 +- + kernel/bpf/helpers.c | 259 +- + kernel/bpf/keys.c | 73 + + kernel/bpf/liveness.c | 633 ++- + kernel/bpf/log.c | 13 +- + kernel/bpf/map_in_map.c | 4 + + kernel/bpf/map_iter.c | 6 + + kernel/bpf/queue_stack_maps.c | 3 + + kernel/bpf/range_tree.c | 52 +- + kernel/bpf/states.c | 108 +- + kernel/bpf/stream.c | 271 +- + kernel/bpf/syscall.c | 46 +- + kernel/bpf/trampoline.c | 8 +- + kernel/bpf/verifier.c | 4455 +++++++++++++------- + kernel/trace/bpf_trace.c | 6 +- + mm/bpf_memcontrol.c | 65 +- + mm/internal.h | 13 +- + mm/memory.c | 43 +- + mm/nommu.c | 43 +- + mm/util.c | 60 + + net/core/filter.c | 33 +- + net/ipv4/Makefile | 1 + + net/ipv4/af_inet.c | 1 + + net/ipv4/bpf_tcp_ca.c | 16 + + net/ipv4/bpf_tcp_ops.c | 326 ++ + net/ipv4/tcp.c | 1 + + net/ipv4/tcp_bpf.c | 1 - + net/ipv4/tcp_input.c | 17 + + net/ipv4/tcp_output.c | 104 +- + net/ipv4/tcp_timer.c | 1 + + net/sched/bpf_qdisc.c | 2 - + samples/bpf/Makefile | 1 + + samples/bpf/hash_func01.h | 55 - + security/bpf/hooks.c | 1 + + security/keys/process_keys.c | 25 + + tools/bpf/bpftool/Documentation/bpftool-btf.rst | 7 +- + tools/bpf/bpftool/Documentation/bpftool-cgroup.rst | 25 +- + tools/bpf/bpftool/Documentation/bpftool-map.rst | 14 +- + tools/bpf/bpftool/Documentation/bpftool-prog.rst | 23 +- + tools/bpf/bpftool/bash-completion/bpftool | 65 +- + tools/bpf/bpftool/btf.c | 337 +- + tools/bpf/bpftool/cgroup.c | 99 +- + tools/bpf/bpftool/common.c | 40 + + tools/bpf/bpftool/gen.c | 26 +- + tools/bpf/bpftool/main.c | 61 +- + tools/bpf/bpftool/main.h | 4 +- + tools/bpf/bpftool/map.c | 95 +- + tools/bpf/bpftool/map_perf_ring.c | 87 +- + tools/bpf/bpftool/prog.c | 302 +- + tools/bpf/bpftool/sign.c | 24 +- + tools/bpf/bpftool/skeleton/profiler.bpf.c | 29 +- + tools/include/uapi/linux/bpf.h | 82 +- + tools/include/uapi/linux/btf.h | 83 +- + tools/lib/bpf/bpf.c | 21 + + tools/lib/bpf/bpf.h | 27 +- + tools/lib/bpf/bpf_gen_internal.h | 13 +- + tools/lib/bpf/btf.c | 424 +- + tools/lib/bpf/btf.h | 50 + + tools/lib/bpf/btf_dump.c | 217 +- + tools/lib/bpf/btf_iter.c | 18 + + tools/lib/bpf/elf.c | 2 +- + tools/lib/bpf/gen_loader.c | 188 +- + tools/lib/bpf/libbpf.c | 670 ++- + tools/lib/bpf/libbpf.h | 41 +- + tools/lib/bpf/libbpf.map | 10 + + tools/lib/bpf/libbpf_internal.h | 6 +- + tools/lib/bpf/libbpf_probes.c | 11 +- + tools/lib/bpf/linker.c | 31 +- + tools/lib/bpf/skel_internal.h | 48 +- + tools/lib/bpf/usdt.c | 4 + + tools/testing/selftests/bpf/.gitignore | 1 - + tools/testing/selftests/bpf/Makefile | 758 +--- + tools/testing/selftests/bpf/Makefile.buildvars | 162 + + tools/testing/selftests/bpf/Makefile.runner | 125 + + tools/testing/selftests/bpf/Makefile.skel | 140 + + tools/testing/selftests/bpf/README.rst | 2 +- + tools/testing/selftests/bpf/bench.c | 6 + + .../testing/selftests/bpf/benchs/bench_libarena.c | 210 + + .../selftests/bpf/benchs/run_bench_libarena.sh | 31 + + tools/testing/selftests/bpf/bpf_kfuncs.h | 19 +- + .../selftests/bpf/bpftool_btf_dump_sorted.expected | 48 + + .../bpf/bpftool_btf_dump_unsorted.expected | 48 + + tools/testing/selftests/bpf/bpftool_helpers.c | 9 +- + tools/testing/selftests/bpf/bpftool_helpers.h | 1 + + tools/testing/selftests/bpf/btf_helpers.c | 45 +- + tools/testing/selftests/bpf/cgroup_helpers.c | 67 + + tools/testing/selftests/bpf/cgroup_helpers.h | 4 + + tools/testing/selftests/bpf/config | 6 +- + tools/testing/selftests/bpf/gen_bpf_skel.sh | 99 + + tools/testing/selftests/bpf/libarena/Makefile | 31 +- + .../bpf/libarena/benchs/bench_malloc.bpf.c | 51 + + .../bpf/libarena/include/bpf_arena_spin_lock.h | 2 +- + .../selftests/bpf/libarena/include/bpf_atomic.h | 2 +- + .../selftests/bpf/libarena/include/bpf_may_goto.h | 1 + + .../bpf/libarena/include/libarena/bitmap.h | 30 +- + .../bpf/libarena/include/libarena/common.h | 71 + + .../bpf/libarena/selftests/test_bitmap.bpf.c | 3 + + .../testing/selftests/bpf/libarena/src/asan.bpf.c | 21 +- + .../selftests/bpf/libarena/src/bitmap.bpf.c | 18 - + .../testing/selftests/bpf/libarena/src/buddy.bpf.c | 44 +- + .../selftests/bpf/libarena/src/common.bpf.c | 27 +- + tools/testing/selftests/bpf/network_helpers.c | 46 +- + tools/testing/selftests/bpf/network_helpers.h | 15 +- + .../selftests/bpf/prog_tests/aggregate_arg.c | 11 + + .../selftests/bpf/prog_tests/aggregate_ret.c | 53 + + .../testing/selftests/bpf/prog_tests/arena_memcg.c | 196 + + .../selftests/bpf/prog_tests/attach_probe.c | 4 +- + .../selftests/bpf/prog_tests/bpf_insn_array.c | 650 ++- + tools/testing/selftests/bpf/prog_tests/bpf_nf.c | 14 +- + .../testing/selftests/bpf/prog_tests/bpf_tcp_ops.c | 560 +++ + .../selftests/bpf/prog_tests/bpf_tcp_ops_hdr.c | 77 + + .../selftests/bpf/prog_tests/bpf_verif_scale.c | 2 +- + .../selftests/bpf/prog_tests/bpftool_btf_dump.c | 297 ++ + .../selftests/bpf/prog_tests/bpftool_ringbuf.c | 417 ++ + tools/testing/selftests/bpf/prog_tests/btf.c | 126 + + .../selftests/bpf/prog_tests/btf_dedup_split.c | 116 + + .../testing/selftests/bpf/prog_tests/btf_distill.c | 106 + + tools/testing/selftests/bpf/prog_tests/btf_dump.c | 400 +- + .../selftests/bpf/prog_tests/btf_field_iter.c | 31 +- + .../bpf/prog_tests/btf_module_allowlist.c | 162 + + tools/testing/selftests/bpf/prog_tests/btf_write.c | 34 + + tools/testing/selftests/bpf/prog_tests/call_rcu.c | 275 ++ + .../selftests/bpf/prog_tests/callx_func_ptr_map.c | 207 + + .../selftests/bpf/prog_tests/callx_rodata_lskel.c | 74 + + tools/testing/selftests/bpf/prog_tests/cb_refs.c | 48 - + .../bpf/prog_tests/cgroup_getset_retval.c | 2 +- + .../selftests/bpf/prog_tests/cgroup_mprog_opts.c | 71 +- + .../selftests/bpf/prog_tests/copy_from_user_bprm.c | 65 + + .../testing/selftests/bpf/prog_tests/core_reloc.c | 10 +- + .../testing/selftests/bpf/prog_tests/exceptions.c | 7 + + .../selftests/bpf/prog_tests/fexit_bpf2bpf.c | 22 +- + .../bpf/prog_tests/flow_dissector_classification.c | 2 +- + .../selftests/bpf/prog_tests/global_data_init.c | 96 +- + tools/testing/selftests/bpf/prog_tests/kasan.c | 479 +++ + .../testing/selftests/bpf/prog_tests/kernel_flag.c | 22 +- + .../testing/selftests/bpf/prog_tests/kfunc_call.c | 2 +- + .../selftests/bpf/prog_tests/kprobe_multi_test.c | 3 +- + .../selftests/bpf/prog_tests/linked_externs.c | 120 + + .../testing/selftests/bpf/prog_tests/lirc_mode2.c | 334 ++ + .../bpf/prog_tests/lsm_inode_init_xattr.c | 399 ++ + .../selftests/bpf/prog_tests/lwt_ip_encap.c | 53 + + .../selftests/bpf/prog_tests/may_goto_far.c | 92 + + .../selftests/bpf/prog_tests/may_goto_priv_stack.c | 41 + + .../bpf/prog_tests/prog_tests_framework.c | 23 + + .../selftests/bpf/prog_tests/queue_stack_map.c | 29 + + .../selftests/bpf/prog_tests/signed_loader.c | 600 ++- + tools/testing/selftests/bpf/prog_tests/snprintf.c | 4 +- + tools/testing/selftests/bpf/prog_tests/stream.c | 553 +++ + .../bpf/prog_tests/struct_ops_private_stack.c | 31 + + tools/testing/selftests/bpf/prog_tests/tailcalls.c | 116 + + .../bpf/prog_tests/test_struct_ops_multi_args.c | 57 +- + .../selftests/bpf/prog_tests/test_task_work.c | 2 +- + tools/testing/selftests/bpf/prog_tests/timer.c | 33 + + tools/testing/selftests/bpf/prog_tests/usdt.c | 54 + + tools/testing/selftests/bpf/prog_tests/verifier.c | 29 + + .../selftests/bpf/prog_tests/verify_pkcs7_sig.c | 4 +- + .../selftests/bpf/prog_tests/xdp_adjust_frags.c | 3 + + .../selftests/bpf/prog_tests/xdp_devmap_attach.c | 5 + + .../selftests/bpf/prog_tests/xdp_metadata.c | 4 +- + tools/testing/selftests/bpf/prog_tests/xsk.c | 2 +- + .../selftests/bpf/progs/aggregate_arg_func.c | 188 + + .../selftests/bpf/progs/aggregate_arg_kfunc.c | 168 + + .../selftests/bpf/progs/aggregate_ret_func.c | 383 ++ + .../selftests/bpf/progs/aggregate_ret_kfunc.c | 203 + + .../bpf/progs/aggregate_ret_kfunc_arena.c | 121 + + .../selftests/bpf/progs/aggregate_ret_target.c | 29 + + tools/testing/selftests/bpf/progs/arena_kfunc.c | 20 +- + tools/testing/selftests/bpf/progs/arena_memcg.c | 23 + + .../selftests/bpf/progs/async_stack_depth.c | 75 + + .../testing/selftests/bpf/progs/bpf_iter_netlink.c | 4 +- + tools/testing/selftests/bpf/progs/bpf_iter_tcp4.c | 11 +- + tools/testing/selftests/bpf/progs/bpf_iter_tcp6.c | 11 +- + tools/testing/selftests/bpf/progs/bpf_iter_udp4.c | 4 +- + tools/testing/selftests/bpf/progs/bpf_iter_udp6.c | 4 +- + tools/testing/selftests/bpf/progs/bpf_iter_unix.c | 5 +- + tools/testing/selftests/bpf/progs/bpf_misc.h | 22 +- + tools/testing/selftests/bpf/progs/bpf_tcp_ops.c | 141 + + .../testing/selftests/bpf/progs/bpf_tcp_ops_hdr.c | 86 + + .../testing/selftests/bpf/progs/bpftool_ringbuf.c | 47 + + .../selftests/bpf/progs/btf__stack_arg_precision.c | 3 +- + .../bpf/progs/btf__verifier_stack_arg_order.c | 3 +- + .../selftests/bpf/progs/btf_module_allowlist.c | 13 + + tools/testing/selftests/bpf/progs/call_rcu.c | 110 + + tools/testing/selftests/bpf/progs/call_rcu_fail.c | 114 + + tools/testing/selftests/bpf/progs/callx_rodata.c | 75 + + tools/testing/selftests/bpf/progs/cb_refs.c | 5 + + .../bpf/{ => progs}/cgroup_getset_retval_hooks.h | 0 + .../selftests/bpf/progs/cgrp_kfunc_failure.c | 8 +- + .../selftests/bpf/progs/compute_live_registers.c | 36 + + .../selftests/bpf/progs/copy_from_user_bprm.c | 69 + + .../testing/selftests/bpf/progs/cpumask_failure.c | 4 +- + tools/testing/selftests/bpf/progs/dynptr_fail.c | 14 +- + .../selftests/bpf/progs/exceptions_dead_subprog.c | 40 + + .../testing/selftests/bpf/progs/exceptions_fail.c | 2 +- + .../selftests/bpf/progs/freplace_ret_pair.c | 12 + + tools/testing/selftests/bpf/progs/irq.c | 17 +- + tools/testing/selftests/bpf/progs/iters.c | 388 +- + .../selftests/bpf/progs/iters_state_safety.c | 3 + + tools/testing/selftests/bpf/progs/iters_testmod.c | 7 +- + tools/testing/selftests/bpf/progs/kasan.c | 502 +++ + tools/testing/selftests/bpf/progs/kasan_harden.c | 52 + + tools/testing/selftests/bpf/progs/linked_arena1.c | 42 + + tools/testing/selftests/bpf/progs/linked_arena2.c | 22 + + tools/testing/selftests/bpf/progs/lirc_mode2.c | 32 + + tools/testing/selftests/bpf/progs/lsm.c | 5 +- + .../selftests/bpf/progs/lsm_inode_init_xattr.c | 98 + + .../bpf/progs/lsm_inode_init_xattr_budget.c | 37 + + .../bpf/progs/lsm_inode_init_xattr_value.c | 47 + + tools/testing/selftests/bpf/progs/map_kptr_fail.c | 10 +- + .../selftests/bpf/progs/may_goto_priv_stack.c | 52 + + .../selftests/bpf/progs/mem_rdonly_untrusted.c | 3 +- + .../selftests/bpf/progs/percpu_alloc_fail.c | 35 + + tools/testing/selftests/bpf/progs/rbtree_fail.c | 4 +- + .../selftests/bpf/progs/refcounted_kptr_fail.c | 9 +- + .../selftests/bpf/progs/res_spin_lock_fail.c | 2 +- + tools/testing/selftests/bpf/progs/stack_arg.c | 3 +- + tools/testing/selftests/bpf/progs/stack_arg_fail.c | 13 +- + .../testing/selftests/bpf/progs/stack_arg_kfunc.c | 3 +- + .../selftests/bpf/progs/stack_arg_precision.c | 4 +- + tools/testing/selftests/bpf/progs/stream.c | 48 + + tools/testing/selftests/bpf/progs/stream_fail.c | 2 +- + .../selftests/bpf/progs/struct_ops_multi_args.c | 14 +- + .../bpf/progs/struct_ops_private_stack_fail.c | 47 +- + .../bpf/progs/struct_ops_private_stack_large.c | 51 + + .../selftests/bpf/progs/tailcall_freplace_multi.c | 27 + + .../selftests/bpf/progs/tailcall_large_stack.c | 62 + + .../selftests/bpf/progs/task_kfunc_failure.c | 10 +- + tools/testing/selftests/bpf/progs/task_work_fail.c | 2 +- + .../selftests/bpf/progs/test_global_func1.c | 65 + + .../selftests/bpf/progs/test_global_func5.c | 2 +- + .../bpf/progs/test_global_func_deep_stack.c | 33 +- + .../selftests/bpf/progs/test_global_percpu_data.c | 10 +- + .../selftests/bpf/progs/test_kfunc_dynptr_param.c | 2 +- + .../selftests/bpf/progs/test_lirc_mode2_kern.c | 26 - + tools/testing/selftests/bpf/progs/test_snprintf.c | 4 +- + tools/testing/selftests/bpf/progs/timer.c | 60 +- + tools/testing/selftests/bpf/progs/timer_failure.c | 29 + + .../selftests/bpf/progs/verifier_aggregate_arg.c | 200 + + .../selftests/bpf/progs/verifier_aggregate_ret.c | 179 + + tools/testing/selftests/bpf/progs/verifier_arena.c | 48 + + .../selftests/bpf/progs/verifier_bpf_fastcall.c | 169 + + tools/testing/selftests/bpf/progs/verifier_bswap.c | 3 +- + tools/testing/selftests/bpf/progs/verifier_callx.c | 1085 +++++ + .../selftests/bpf/progs/verifier_callx_rodata.c | 902 ++++ + tools/testing/selftests/bpf/progs/verifier_cfg.c | 39 + + tools/testing/selftests/bpf/progs/verifier_ctx.c | 2 +- + .../selftests/bpf/progs/verifier_global_ptr_args.c | 9 +- + .../selftests/bpf/progs/verifier_global_subprogs.c | 33 +- + tools/testing/selftests/bpf/progs/verifier_gotol.c | 1 + + tools/testing/selftests/bpf/progs/verifier_gotox.c | 125 +- + .../bpf/progs/verifier_helper_access_var_len.c | 336 +- + .../bpf/progs/verifier_helper_packet_access.c | 4 +- + .../selftests/bpf/progs/verifier_jit_inline.c | 2 +- + .../bpf/progs/verifier_kfunc_packet_access.c | 47 + + .../selftests/bpf/progs/verifier_kfunc_uninit.c | 236 ++ + .../bpf/progs/verifier_kfunc_uninit_multi.c | 113 + + .../selftests/bpf/progs/verifier_large_stack.c | 425 ++ + tools/testing/selftests/bpf/progs/verifier_ldsx.c | 36 +- + .../selftests/bpf/progs/verifier_live_stack.c | 137 +- + .../selftests/bpf/progs/verifier_load_acquire.c | 1 + + tools/testing/selftests/bpf/progs/verifier_lsm.c | 13 + + .../selftests/bpf/progs/verifier_lsm_init_xattr.c | 245 ++ + .../selftests/bpf/progs/verifier_map_in_map.c | 3 +- + .../bpf/progs/verifier_map_lookup_refine.c | 2 +- + .../selftests/bpf/progs/verifier_may_goto_1.c | 121 + + tools/testing/selftests/bpf/progs/verifier_movsx.c | 3 +- + tools/testing/selftests/bpf/progs/verifier_mtu.c | 88 + + .../selftests/bpf/progs/verifier_percpu_addr.c | 1 + + .../selftests/bpf/progs/verifier_private_stack.c | 1 + + .../selftests/bpf/progs/verifier_raw_stack.c | 25 + + .../selftests/bpf/progs/verifier_ref_tracking.c | 6 +- + tools/testing/selftests/bpf/progs/verifier_sdiv.c | 3 +- + tools/testing/selftests/bpf/progs/verifier_sock.c | 10 +- + .../selftests/bpf/progs/verifier_spill_fill.c | 43 + + .../selftests/bpf/progs/verifier_stack_arg.c | 4 +- + .../selftests/bpf/progs/verifier_stack_arg_order.c | 8 +- + .../selftests/bpf/progs/verifier_stack_ptr.c | 53 + + .../selftests/bpf/progs/verifier_store_release.c | 1 + + .../selftests/bpf/progs/verifier_tailcall.c | 57 + + .../testing/selftests/bpf/progs/verifier_unpriv.c | 37 + + .../testing/selftests/bpf/progs/verifier_var_off.c | 32 + + .../selftests/bpf/progs/verifier_vfs_reject.c | 6 +- + tools/testing/selftests/bpf/progs/verifier_xadd.c | 27 + + .../selftests/bpf/progs/wakeup_source_fail.c | 2 +- + tools/testing/selftests/bpf/progs/wq_failures.c | 4 +- + tools/testing/selftests/bpf/test_btf.h | 2 +- + .../testing/selftests/bpf/test_kmods/bpf_testmod.c | 303 +- + .../testing/selftests/bpf/test_kmods/bpf_testmod.h | 5 + + .../selftests/bpf/test_kmods/bpf_testmod_kfunc.h | 136 + + tools/testing/selftests/bpf/test_lirc_mode2.sh | 41 - + tools/testing/selftests/bpf/test_lirc_mode2_user.c | 177 - + tools/testing/selftests/bpf/test_loader.c | 86 +- + tools/testing/selftests/bpf/test_progs.h | 3 + + tools/testing/selftests/bpf/testing_helpers.c | 88 +- + tools/testing/selftests/bpf/testing_helpers.h | 4 + + tools/testing/selftests/bpf/unpriv_helpers.c | 35 +- + tools/testing/selftests/bpf/unpriv_helpers.h | 4 + + tools/testing/selftests/bpf/usdt_2.c | 16 + + tools/testing/selftests/bpf/verifier/basic_call.c | 2 +- + tools/testing/selftests/bpf/verifier/calls.c | 132 +- + tools/testing/selftests/bpf/verifier/map_kptr.c | 4 +- + tools/testing/selftests/bpf/verify_sig_setup.sh | 63 +- + tools/testing/selftests/bpf/veristat.c | 4 +- + tools/testing/selftests/bpf/vmtest.sh | 16 +- + tools/testing/selftests/bpf/xdp_hw_metadata.c | 2 +- + 372 files changed, 29911 insertions(+), 5208 deletions(-) + delete mode 100644 crypto/bpf_crypto_skcipher.c + delete mode 100644 include/linux/bpf_crypto.h + create mode 100644 kernel/bpf/keys.c + create mode 100644 net/ipv4/bpf_tcp_ops.c + delete mode 100644 samples/bpf/hash_func01.h + create mode 100644 tools/testing/selftests/bpf/Makefile.buildvars + create mode 100644 tools/testing/selftests/bpf/Makefile.runner + create mode 100644 tools/testing/selftests/bpf/Makefile.skel + create mode 100644 tools/testing/selftests/bpf/benchs/bench_libarena.c + create mode 100755 tools/testing/selftests/bpf/benchs/run_bench_libarena.sh + create mode 100644 tools/testing/selftests/bpf/bpftool_btf_dump_sorted.expected + create mode 100644 tools/testing/selftests/bpf/bpftool_btf_dump_unsorted.expected + create mode 100755 tools/testing/selftests/bpf/gen_bpf_skel.sh + create mode 100644 tools/testing/selftests/bpf/libarena/benchs/bench_malloc.bpf.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/aggregate_arg.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/aggregate_ret.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/arena_memcg.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/bpf_tcp_ops.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/bpf_tcp_ops_hdr.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/bpftool_btf_dump.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/bpftool_ringbuf.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/btf_module_allowlist.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/call_rcu.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/callx_func_ptr_map.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/callx_rodata_lskel.c + delete mode 100644 tools/testing/selftests/bpf/prog_tests/cb_refs.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/copy_from_user_bprm.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/kasan.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/linked_externs.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/lirc_mode2.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/lsm_inode_init_xattr.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/may_goto_far.c + create mode 100644 tools/testing/selftests/bpf/prog_tests/may_goto_priv_stack.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_arg_func.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_arg_kfunc.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_ret_func.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_ret_kfunc.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_ret_kfunc_arena.c + create mode 100644 tools/testing/selftests/bpf/progs/aggregate_ret_target.c + create mode 100644 tools/testing/selftests/bpf/progs/arena_memcg.c + create mode 100644 tools/testing/selftests/bpf/progs/bpf_tcp_ops.c + create mode 100644 tools/testing/selftests/bpf/progs/bpf_tcp_ops_hdr.c + create mode 100644 tools/testing/selftests/bpf/progs/bpftool_ringbuf.c + create mode 100644 tools/testing/selftests/bpf/progs/btf_module_allowlist.c + create mode 100644 tools/testing/selftests/bpf/progs/call_rcu.c + create mode 100644 tools/testing/selftests/bpf/progs/call_rcu_fail.c + create mode 100644 tools/testing/selftests/bpf/progs/callx_rodata.c + rename tools/testing/selftests/bpf/{ => progs}/cgroup_getset_retval_hooks.h (100%) + create mode 100644 tools/testing/selftests/bpf/progs/copy_from_user_bprm.c + create mode 100644 tools/testing/selftests/bpf/progs/exceptions_dead_subprog.c + create mode 100644 tools/testing/selftests/bpf/progs/freplace_ret_pair.c + create mode 100644 tools/testing/selftests/bpf/progs/kasan.c + create mode 100644 tools/testing/selftests/bpf/progs/kasan_harden.c + create mode 100644 tools/testing/selftests/bpf/progs/linked_arena1.c + create mode 100644 tools/testing/selftests/bpf/progs/linked_arena2.c + create mode 100644 tools/testing/selftests/bpf/progs/lirc_mode2.c + create mode 100644 tools/testing/selftests/bpf/progs/lsm_inode_init_xattr.c + create mode 100644 tools/testing/selftests/bpf/progs/lsm_inode_init_xattr_budget.c + create mode 100644 tools/testing/selftests/bpf/progs/lsm_inode_init_xattr_value.c + create mode 100644 tools/testing/selftests/bpf/progs/may_goto_priv_stack.c + create mode 100644 tools/testing/selftests/bpf/progs/struct_ops_private_stack_large.c + create mode 100644 tools/testing/selftests/bpf/progs/tailcall_freplace_multi.c + create mode 100644 tools/testing/selftests/bpf/progs/tailcall_large_stack.c + delete mode 100644 tools/testing/selftests/bpf/progs/test_lirc_mode2_kern.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_aggregate_arg.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_aggregate_ret.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_callx.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_callx_rodata.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_kfunc_packet_access.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_kfunc_uninit.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_kfunc_uninit_multi.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_large_stack.c + create mode 100644 tools/testing/selftests/bpf/progs/verifier_lsm_init_xattr.c + delete mode 100755 tools/testing/selftests/bpf/test_lirc_mode2.sh + delete mode 100644 tools/testing/selftests/bpf/test_lirc_mode2_user.c +$ git am -3 ../patches/0001-perf-Fixup-for-btf_vlan-API-change.patch +Applying: perf: Fixup for btf_vlan() API change +Using index info to reconstruct a base tree... +M tools/perf/builtin-trace.c +M tools/perf/util/btf.c +Falling back to patching base and 3-way merge... +Auto-merging tools/perf/builtin-trace.c +No changes -- Patch already applied. +Merging ipsec-next/master (014d795c73837 idpf: fix kernel-doc parameter descriptions) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/klassert/ipsec-next.git ipsec-next/master +Already up to date. +Merging mlx5-next/mlx5-next (36b1d3299d0b6 net/mlx5: Add qp_latency_sensitive_disable cap bit) +$ git merge -m Merge branch 'mlx5-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mellanox/linux.git mlx5-next/mlx5-next +Already up to date. +Merging netfilter-next/main (87b80c2f6b05c net: txgbe: free the fixed-rate clock on cleanup) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/netfilter/nf-next.git netfilter-next/main +Already up to date. +Merging ipvs-next/main (6ebcf5074cff0 net: openvswitch: don't schedule rebalancing if there are no datapaths) +$ git merge -m Merge branch 'main' of https://git.kernel.org/pub/scm/linux/kernel/git/horms/ipvs-next.git ipvs-next/main +Already up to date. +Merging bluetooth/master (fdd5964bd3899 Bluetooth: btusb: drop BROKEN_EXT_SCAN quirk for 0bda:a728) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/bluetooth/bluetooth-next.git bluetooth/master +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../bindings/net/bluetooth/brcm,bluetooth.yaml | 1 + + MAINTAINERS | 2 + + drivers/bluetooth/Kconfig | 15 + + drivers/bluetooth/Makefile | 3 + + drivers/bluetooth/btbcm.c | 6 +- + drivers/bluetooth/btintel.c | 57 +- + drivers/bluetooth/btintel.h | 14 + + drivers/bluetooth/btintel_pcie.c | 1535 ++++++- + drivers/bluetooth/btintel_pcie.h | 257 +- + drivers/bluetooth/btmrvl_main.c | 4 +- + drivers/bluetooth/btmtk.c | 104 +- + drivers/bluetooth/btmtk.h | 9 + + drivers/bluetooth/btmtksdio.c | 57 +- + drivers/bluetooth/btnxpuart.c | 4 +- + drivers/bluetooth/{btusb.c => btusb_main.c} | 369 +- + drivers/bluetooth/btusb_qcom.c | 4557 ++++++++++++++++++++ + drivers/bluetooth/btusb_qcom.h | 99 + + drivers/bluetooth/hci_bcm.c | 1 + + drivers/bluetooth/hci_bcsp.c | 26 +- + drivers/bluetooth/hci_h4.c | 123 +- + drivers/bluetooth/hci_h5.c | 11 +- + drivers/bluetooth/hci_ll.c | 2 +- + drivers/bluetooth/hci_serdev.c | 46 +- + drivers/bluetooth/hci_uart.h | 39 +- + drivers/bluetooth/virtio_bt.c | 21 +- + include/net/bluetooth/hci_core.h | 49 + + include/net/bluetooth/hci_h4.h | 60 + + include/net/bluetooth/hci_mon.h | 2 + + include/net/bluetooth/l2cap.h | 102 +- + net/bluetooth/6lowpan.c | 44 +- + net/bluetooth/Makefile | 2 +- + net/bluetooth/bnep/core.c | 16 +- + net/bluetooth/hci_core.c | 47 +- + net/bluetooth/hci_event.c | 68 +- + net/bluetooth/hci_h4.c | 160 + + net/bluetooth/hci_sock.c | 8 + + net/bluetooth/hci_sync.c | 64 +- + net/bluetooth/iso.c | 15 +- + net/bluetooth/l2cap_core.c | 284 +- + net/bluetooth/l2cap_sock.c | 118 +- + net/bluetooth/mgmt.c | 14 +- + net/bluetooth/rfcomm/core.c | 6 +- + net/bluetooth/rfcomm/sock.c | 5 +- + net/bluetooth/sco.c | 10 +- + net/bluetooth/smp.c | 4 +- + scripts/context-analysis-suppression.txt | 1 + + 46 files changed, 7675 insertions(+), 766 deletions(-) + rename drivers/bluetooth/{btusb.c => btusb_main.c} (94%) + create mode 100644 drivers/bluetooth/btusb_qcom.c + create mode 100644 drivers/bluetooth/btusb_qcom.h + create mode 100644 include/net/bluetooth/hci_h4.h + create mode 100644 net/bluetooth/hci_h4.c +$ git am -3 ../patches/0001-bluetooth-Fix-up-mismerge-due-to-drivers-bluetooth-b.patch +Applying: bluetooth: Fix up mismerge due to drivers/bluetooth/btusb_main.c +Using index info to reconstruct a base tree... +M drivers/bluetooth/btusb_main.c +Falling back to patching base and 3-way merge... +Auto-merging drivers/bluetooth/btusb_main.c +CONFLICT (content): Merge conflict in drivers/bluetooth/btusb_main.c +Recorded preimage for 'drivers/bluetooth/btusb_main.c' +error: Failed to merge in the changes. +hint: Use 'git am --show-current-patch=diff' to see the failed patch +hint: When you have resolved this problem, run "git am --continue". +hint: If you prefer to skip this patch, run "git am --skip" instead. +hint: To restore the original branch and stop patching, run "git am --abort". +hint: Disable this message with "git config advice.mergeConflict false" +Patch failed at 0001 bluetooth: Fix up mismerge due to drivers/bluetooth/btusb_main.c +Merging wireless-next/for-next (21b4248bfa0f4 Merge tag 'ath-next-20260927' of git://git.kernel.org/pub/scm/linux/kernel/git/ath/ath) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/wireless/wireless-next.git wireless-next/for-next +Auto-merging drivers/net/wireless/st/cw1200/txrx.c +Auto-merging net/mac80211/mesh_pathtbl.c +Auto-merging net/mac80211/mlme.c +Auto-merging net/mac80211/tx.c +Auto-merging net/wireless/nl80211.c +Merge made by the 'ort' strategy. + .../bindings/net/wireless/qcom,ath10k.yaml | 16 ++ + drivers/net/wireless/ath/ath11k/wmi.c | 9 + + drivers/net/wireless/ath/ath12k/core.h | 9 - + drivers/net/wireless/ath/ath12k/debugfs.c | 78 ++++++- + drivers/net/wireless/ath/ath12k/dp.c | 90 +++++--- + drivers/net/wireless/ath/ath12k/dp.h | 114 ++++++--- + drivers/net/wireless/ath/ath12k/dp_mon.c | 15 +- + drivers/net/wireless/ath/ath12k/dp_mon.h | 2 +- + drivers/net/wireless/ath/ath12k/dp_rx.c | 33 +-- + drivers/net/wireless/ath/ath12k/dp_rx.h | 3 - + drivers/net/wireless/ath/ath12k/dp_stats.h | 25 ++ + drivers/net/wireless/ath/ath12k/hw.h | 1 + + drivers/net/wireless/ath/ath12k/mac.c | 5 +- + drivers/net/wireless/ath/ath12k/qmi.c | 102 +++++--- + drivers/net/wireless/ath/ath12k/qmi.h | 4 + + drivers/net/wireless/ath/ath12k/wifi7/dp_mon.c | 22 +- + drivers/net/wireless/ath/ath12k/wifi7/dp_rx.c | 52 ++++- + drivers/net/wireless/ath/ath12k/wifi7/dp_tx.c | 39 ++-- + drivers/net/wireless/ath/ath12k/wifi7/hal_rx.c | 5 +- + drivers/net/wireless/ath/ath12k/wifi7/hw.c | 7 + + drivers/net/wireless/ath/ath6kl/cfg80211.c | 10 +- + drivers/net/wireless/ath/wil6210/cfg80211.c | 13 +- + drivers/net/wireless/broadcom/b43legacy/main.c | 3 +- + .../broadcom/brcm80211/brcmfmac/cfg80211.c | 11 +- + drivers/net/wireless/intel/iwlwifi/mld/d3.c | 6 +- + drivers/net/wireless/intel/iwlwifi/mvm/d3.c | 6 +- + drivers/net/wireless/marvell/libertas/cfg.c | 6 +- + drivers/net/wireless/marvell/libertas/rx.c | 9 + + drivers/net/wireless/marvell/mwifiex/cfg80211.c | 20 +- + drivers/net/wireless/marvell/mwifiex/sdio.c | 3 + + drivers/net/wireless/microchip/wilc1000/cfg80211.c | 13 +- + drivers/net/wireless/morsemicro/mm81x/core.h | 7 +- + drivers/net/wireless/morsemicro/mm81x/fw.c | 9 +- + drivers/net/wireless/morsemicro/mm81x/mac.c | 181 +++++++++------ + drivers/net/wireless/morsemicro/mm81x/sdio.c | 9 +- + drivers/net/wireless/morsemicro/mm81x/skbq.c | 7 +- + drivers/net/wireless/nxp/nxpwifi/cfg80211.c | 11 +- + drivers/net/wireless/quantenna/qtnfmac/cfg80211.c | 8 +- + drivers/net/wireless/realtek/rtw89/wow.c | 6 +- + drivers/net/wireless/st/cw1200/txrx.c | 2 +- + drivers/net/wireless/virtual/mac80211_hwsim_main.c | 35 +++ + drivers/net/wireless/virtual/mac80211_hwsim_nan.c | 22 ++ + drivers/net/wireless/virtual/mac80211_hwsim_nan.h | 3 + + drivers/staging/rtl8723bs/os_dep/ioctl_cfg80211.c | 9 +- + include/linux/ieee80211.h | 22 ++ + include/net/cfg80211.h | 45 +++- + include/net/ieee80211_radiotap.h | 45 +++- + include/net/mac80211.h | 13 +- + include/uapi/linux/nl80211.h | 55 ++++- + net/mac80211/cfg.c | 54 ++++- + net/mac80211/debugfs_key.c | 18 +- + net/mac80211/ieee80211_i.h | 7 + + net/mac80211/key.c | 96 ++++++-- + net/mac80211/key.h | 3 +- + net/mac80211/mesh_pathtbl.c | 13 +- + net/mac80211/mesh_plink.c | 4 +- + net/mac80211/mlme.c | 84 ++++++- + net/mac80211/nan.c | 33 +++ + net/mac80211/offchannel.c | 10 + + net/mac80211/parse.c | 4 + + net/mac80211/sta_info.h | 2 + + net/mac80211/tx.c | 2 +- + net/mac80211/util.c | 41 ++++ + net/wireless/ap.c | 7 +- + net/wireless/chan.c | 33 ++- + net/wireless/core.c | 87 +++++-- + net/wireless/core.h | 11 +- + net/wireless/ibss.c | 3 +- + net/wireless/nl80211.c | 257 +++++++++++++++++---- + net/wireless/rdev-ops.h | 40 +++- + net/wireless/reg.c | 13 +- + net/wireless/sme.c | 10 +- + net/wireless/sysfs.c | 12 +- + net/wireless/trace.h | 62 +++-- + net/wireless/util.c | 40 +++- + net/wireless/wext-compat.c | 8 +- + 76 files changed, 1688 insertions(+), 486 deletions(-) + create mode 100644 drivers/net/wireless/ath/ath12k/dp_stats.h +Merging ath-next/for-next (21b4248bfa0f4 Merge tag 'ath-next-20260927' of git://git.kernel.org/pub/scm/linux/kernel/git/ath/ath) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ath/ath.git ath-next/for-next +Already up to date. +Merging iwlwifi-next/next (b75dd71d260ee wifi: iwlwifi: disable HE ER mode) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/iwlwifi/iwlwifi-next.git iwlwifi-next/next +Merge made by the 'ort' strategy. + drivers/net/wireless/intel/iwlwifi/Makefile | 6 +- + drivers/net/wireless/intel/iwlwifi/cfg/bz.c | 8 +- + drivers/net/wireless/intel/iwlwifi/cfg/dr.c | 2 +- + drivers/net/wireless/intel/iwlwifi/cfg/rf-fm.c | 4 + + drivers/net/wireless/intel/iwlwifi/cfg/sc.c | 2 +- + drivers/net/wireless/intel/iwlwifi/fw/acpi.c | 306 ++++++++++----- + drivers/net/wireless/intel/iwlwifi/fw/acpi.h | 20 +- + .../net/wireless/intel/iwlwifi/fw/api/datapath.h | 31 +- + .../net/wireless/intel/iwlwifi/fw/api/mac-cfg.h | 38 +- + drivers/net/wireless/intel/iwlwifi/fw/api/power.h | 14 +- + drivers/net/wireless/intel/iwlwifi/fw/api/rx.h | 23 +- + drivers/net/wireless/intel/iwlwifi/fw/api/scan.h | 68 +++- + drivers/net/wireless/intel/iwlwifi/fw/api/stats.h | 60 ++- + drivers/net/wireless/intel/iwlwifi/fw/dbg.c | 51 ++- + drivers/net/wireless/intel/iwlwifi/fw/file.h | 2 + + drivers/net/wireless/intel/iwlwifi/fw/init.c | 6 +- + drivers/net/wireless/intel/iwlwifi/fw/regulatory.c | 76 +++- + drivers/net/wireless/intel/iwlwifi/fw/regulatory.h | 26 +- + drivers/net/wireless/intel/iwlwifi/fw/runtime.h | 13 +- + drivers/net/wireless/intel/iwlwifi/fw/uefi.c | 412 +++++++++++++++++---- + drivers/net/wireless/intel/iwlwifi/fw/uefi.h | 26 ++ + drivers/net/wireless/intel/iwlwifi/iwl-dbg-tlv.c | 50 ++- + drivers/net/wireless/intel/iwlwifi/iwl-drv.c | 9 +- + drivers/net/wireless/intel/iwlwifi/iwl-io.c | 231 ------------ + drivers/net/wireless/intel/iwlwifi/iwl-io.h | 4 - + drivers/net/wireless/intel/iwlwifi/iwl-nvm-parse.c | 8 +- + drivers/net/wireless/intel/iwlwifi/iwl-trans.c | 45 ++- + drivers/net/wireless/intel/iwlwifi/iwl-trans.h | 12 +- + drivers/net/wireless/intel/iwlwifi/mld/agg.c | 7 +- + drivers/net/wireless/intel/iwlwifi/mld/debugfs.c | 2 +- + drivers/net/wireless/intel/iwlwifi/mld/fw.c | 4 +- + drivers/net/wireless/intel/iwlwifi/mld/hcmd.h | 2 + + drivers/net/wireless/intel/iwlwifi/mld/key.c | 6 + + drivers/net/wireless/intel/iwlwifi/mld/link.c | 14 +- + drivers/net/wireless/intel/iwlwifi/mld/mac80211.c | 4 + + drivers/net/wireless/intel/iwlwifi/mld/mld.c | 7 + + drivers/net/wireless/intel/iwlwifi/mld/nan.c | 35 +- + drivers/net/wireless/intel/iwlwifi/mld/notif.c | 6 +- + drivers/net/wireless/intel/iwlwifi/mld/power.c | 2 +- + .../net/wireless/intel/iwlwifi/mld/regulatory.c | 83 ++++- + .../net/wireless/intel/iwlwifi/mld/regulatory.h | 4 +- + drivers/net/wireless/intel/iwlwifi/mld/rx.c | 17 +- + drivers/net/wireless/intel/iwlwifi/mld/scan.c | 15 +- + drivers/net/wireless/intel/iwlwifi/mld/stats.c | 35 +- + drivers/net/wireless/intel/iwlwifi/mld/tlc.c | 184 ++++++--- + drivers/net/wireless/intel/iwlwifi/mvm/d3.c | 89 +++-- + drivers/net/wireless/intel/iwlwifi/mvm/debugfs.c | 135 +------ + drivers/net/wireless/intel/iwlwifi/mvm/fw.c | 7 +- + drivers/net/wireless/intel/iwlwifi/mvm/mac80211.c | 4 +- + drivers/net/wireless/intel/iwlwifi/mvm/mvm.h | 3 +- + drivers/net/wireless/intel/iwlwifi/mvm/nvm.c | 15 + + drivers/net/wireless/intel/iwlwifi/mvm/ops.c | 2 +- + drivers/net/wireless/intel/iwlwifi/mvm/rxmq.c | 26 +- + drivers/net/wireless/intel/iwlwifi/mvm/scan.c | 4 +- + drivers/net/wireless/intel/iwlwifi/mvm/sta.c | 8 +- + .../net/wireless/intel/iwlwifi/pcie/ctxt-info-v2.c | 19 +- + .../net/wireless/intel/iwlwifi/pcie/ctxt-info.c | 2 +- + drivers/net/wireless/intel/iwlwifi/pcie/drv.c | 12 +- + .../intel/iwlwifi/pcie/{gen1_2 => }/internal.h | 68 +--- + .../wireless/intel/iwlwifi/pcie/{gen1_2 => }/rx.c | 9 +- + .../intel/iwlwifi/pcie/{gen1_2 => }/trans-gen2.c | 4 +- + .../intel/iwlwifi/pcie/{gen1_2 => }/trans.c | 248 +++---------- + .../intel/iwlwifi/pcie/{gen1_2 => }/tx-gen2.c | 0 + .../wireless/intel/iwlwifi/pcie/{gen1_2 => }/tx.c | 60 ++- + 64 files changed, 1623 insertions(+), 1072 deletions(-) + rename drivers/net/wireless/intel/iwlwifi/pcie/{gen1_2 => }/internal.h (95%) + rename drivers/net/wireless/intel/iwlwifi/pcie/{gen1_2 => }/rx.c (99%) + rename drivers/net/wireless/intel/iwlwifi/pcie/{gen1_2 => }/trans-gen2.c (99%) + rename drivers/net/wireless/intel/iwlwifi/pcie/{gen1_2 => }/trans.c (95%) + rename drivers/net/wireless/intel/iwlwifi/pcie/{gen1_2 => }/tx-gen2.c (100%) + rename drivers/net/wireless/intel/iwlwifi/pcie/{gen1_2 => }/tx.c (98%) +Merging wpan-next/master (a6bfdfcc6711d ieee802154: allow legacy LLSEC ADD/DEL ops to pass strict validation) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan-next.git wpan-next/master +Already up to date. +Merging wpan-staging/staging (a6bfdfcc6711d ieee802154: allow legacy LLSEC ADD/DEL ops to pass strict validation) +$ git merge -m Merge branch 'staging' of https://git.kernel.org/pub/scm/linux/kernel/git/wpan/wpan-next.git wpan-staging/staging +Already up to date. +Merging mtd/mtd/next (112a666bd82e9 mtd: virt-concat: unlink discarded items before freeing) +$ git merge -m Merge branch 'mtd/next' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git mtd/mtd/next +Auto-merging drivers/mtd/chips/cfi_cmdset_0001.c +Auto-merging drivers/mtd/mtd_virt_concat.c +Auto-merging drivers/mtd/mtdconcat.c +Auto-merging drivers/mtd/mtdcore.c +Auto-merging drivers/mtd/nand/raw/cadence-nand-controller.c +Auto-merging drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c +Merge made by the 'ort' strategy. + drivers/mtd/chips/cfi_cmdset_0001.c | 5 +- + drivers/mtd/chips/cfi_cmdset_0002.c | 15 ++++ + drivers/mtd/devices/docg3.c | 20 ++--- + drivers/mtd/devices/powernv_flash.c | 2 +- + drivers/mtd/hyperbus/hbmc-am654.c | 7 +- + drivers/mtd/maps/Kconfig | 17 ++++ + drivers/mtd/maps/Makefile | 1 + + drivers/mtd/maps/amd76xrom.c | 2 + + drivers/mtd/maps/int0800.c | 115 ++++++++++++++++++++++++ + drivers/mtd/maps/l440gx.c | 2 - + drivers/mtd/maps/pci.c | 2 +- + drivers/mtd/mtd_blkdevs.c | 5 ++ + drivers/mtd/mtd_virt_concat.c | 2 + + drivers/mtd/mtdconcat.c | 2 +- + drivers/mtd/mtdcore.c | 2 +- + drivers/mtd/mtdsuper.c | 4 +- + drivers/mtd/nand/onenand/onenand_base.c | 2 +- + drivers/mtd/nand/raw/cadence-nand-controller.c | 2 +- + drivers/mtd/nand/raw/fsmc_nand.c | 2 +- + drivers/mtd/nand/raw/gpmi-nand/gpmi-nand.c | 2 +- + drivers/mtd/nand/raw/intel-nand-controller.c | 6 +- + drivers/mtd/nand/raw/loongson-nand-controller.c | 4 +- + drivers/mtd/nand/raw/lpc32xx_mlc.c | 10 +-- + drivers/mtd/nand/raw/lpc32xx_slc.c | 10 +-- + drivers/mtd/nand/raw/marvell_nand.c | 7 +- + drivers/mtd/nand/raw/mtk_nand.c | 2 +- + drivers/mtd/nand/raw/nand_base.c | 2 +- + drivers/mtd/nand/raw/nand_legacy.c | 4 +- + drivers/mtd/nand/raw/omap2.c | 10 +-- + drivers/mtd/nand/raw/sh_flctl.c | 8 +- + drivers/mtd/spi-nor/otp.c | 2 +- + include/linux/mtd/bbm.h | 2 +- + include/linux/mtd/rawnand.h | 2 +- + include/linux/mtd/sh_flctl.h | 4 +- + 34 files changed, 215 insertions(+), 69 deletions(-) + create mode 100644 drivers/mtd/maps/int0800.c +Merging nand/nand/next (55c5b6d5f59f5 mtd: nand: qpic_common: drop stray empty line from struct 'bam_transaction') +$ git merge -m Merge branch 'nand/next' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git nand/nand/next +Auto-merging drivers/mtd/nand/raw/lpc32xx_mlc.c +Auto-merging drivers/mtd/nand/raw/lpc32xx_slc.c +Auto-merging drivers/mtd/nand/raw/omap2.c +Auto-merging drivers/mtd/nand/spi/core.c +Merge made by the 'ort' strategy. + drivers/mtd/nand/raw/atmel/nand-controller.c | 6 +- + drivers/mtd/nand/raw/brcmnand/brcmnand.c | 2 + + drivers/mtd/nand/raw/fsl_ifc_nand.c | 34 +--- + drivers/mtd/nand/raw/internals.h | 1 + + drivers/mtd/nand/raw/lpc32xx_mlc.c | 21 +-- + drivers/mtd/nand/raw/lpc32xx_slc.c | 21 +-- + drivers/mtd/nand/raw/nand_esmt.c | 34 ++++ + drivers/mtd/nand/raw/nand_ids.c | 2 +- + drivers/mtd/nand/raw/ndfc.c | 28 ++- + .../mtd/nand/raw/nuvoton-ma35d1-nand-controller.c | 1 - + drivers/mtd/nand/raw/omap2.c | 4 +- + drivers/mtd/nand/raw/stm32_fmc2_nand.c | 8 +- + drivers/mtd/nand/raw/sunxi_nand.c | 37 ++-- + drivers/mtd/nand/spi/Makefile | 2 +- + drivers/mtd/nand/spi/core.c | 1 + + drivers/mtd/nand/spi/issi.c | 187 +++++++++++++++++++++ + drivers/mtd/nand/spi/winbond.c | 87 +++++++++- + include/linux/mtd/lpc32xx_mlc.h | 17 -- + include/linux/mtd/lpc32xx_slc.h | 17 -- + include/linux/mtd/nand-qpic-common.h | 1 - + include/linux/mtd/spinand.h | 3 +- + 21 files changed, 370 insertions(+), 144 deletions(-) + create mode 100644 drivers/mtd/nand/spi/issi.c + delete mode 100644 include/linux/mtd/lpc32xx_mlc.h + delete mode 100644 include/linux/mtd/lpc32xx_slc.h +Merging spi-nor/spi-nor/next (68d7115d77a87 mtd: spi-nor: sfdp: get the 1-1-8 and 1-8-8 page programs from 4BAIT) +$ git merge -m Merge branch 'spi-nor/next' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git spi-nor/spi-nor/next +Auto-merging drivers/mtd/spi-nor/core.c +Auto-merging drivers/mtd/spi-nor/otp.c +Merge made by the 'ort' strategy. + Documentation/driver-api/mtd/spi-nor.rst | 2 +- + drivers/mtd/spi-nor/atmel.c | 75 ++- + drivers/mtd/spi-nor/core.c | 1035 ++++++++++++------------------ + drivers/mtd/spi-nor/core.h | 144 +++-- + drivers/mtd/spi-nor/debugfs.c | 31 +- + drivers/mtd/spi-nor/everspin.c | 7 +- + drivers/mtd/spi-nor/gigadevice.c | 15 +- + drivers/mtd/spi-nor/issi.c | 29 +- + drivers/mtd/spi-nor/macronix.c | 73 ++- + drivers/mtd/spi-nor/micron-st.c | 70 +- + drivers/mtd/spi-nor/otp.c | 16 +- + drivers/mtd/spi-nor/sfdp.c | 255 ++++++-- + drivers/mtd/spi-nor/sfdp.h | 23 +- + drivers/mtd/spi-nor/spansion.c | 109 ++-- + drivers/mtd/spi-nor/sst.c | 16 +- + drivers/mtd/spi-nor/swp.c | 133 ++-- + drivers/mtd/spi-nor/sysfs.c | 4 +- + drivers/mtd/spi-nor/winbond.c | 320 +++++++-- + drivers/mtd/spi-nor/xmc.c | 1 + + include/linux/mtd/spi-nor.h | 10 +- + 20 files changed, 1310 insertions(+), 1058 deletions(-) +Merging crypto/master (67aefeccc4101 crypto: ccp - kill v5 IRQ tasklet during teardown) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/herbert/cryptodev-2.6.git crypto/master +Auto-merging Documentation/devicetree/bindings/trivial-devices.yaml +Auto-merging MAINTAINERS +Auto-merging arch/arm64/crypto/aes-neonbs-glue.c +Auto-merging lib/Kconfig.debug +Auto-merging lib/tests/Makefile +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-driver-qat_kpt | 3 +- + .../bindings/crypto/hisilicon,hip06-sec.yaml | 134 -- + .../devicetree/bindings/trivial-devices.yaml | 2 + + MAINTAINERS | 12 +- + arch/arm/crypto/aes-neonbs-glue.c | 2 +- + arch/arm64/boot/dts/hisilicon/hip07.dtsi | 280 ----- + arch/arm64/crypto/aes-neonbs-glue.c | 2 +- + arch/powerpc/crypto/vmx.c | 14 +- + crypto/algapi.c | 23 +- + crypto/asymmetric_keys/Kconfig | 12 + + crypto/asymmetric_keys/Makefile | 2 + + crypto/asymmetric_keys/pkcs7_parser.c | 6 + + crypto/asymmetric_keys/restrict.c | 8 +- + crypto/asymmetric_keys/verify_pefile.c | 5 + + crypto/asymmetric_keys/verify_pefile_test.c | 100 ++ + crypto/asymmetric_keys/x509_public_key.c | 6 +- + crypto/lskcipher.c | 6 +- + crypto/rsassa-pkcs1.c | 4 +- + crypto/zstd.c | 40 +- + drivers/char/hw_random/cctrng.c | 7 +- + drivers/char/hw_random/imx-rngc.c | 8 +- + drivers/char/hw_random/jh7110-trng.c | 39 +- + drivers/char/hw_random/omap-rng.c | 8 +- + drivers/crypto/amcc/crypto4xx_core.c | 35 +- + drivers/crypto/amcc/crypto4xx_trng.c | 15 +- + drivers/crypto/amcc/crypto4xx_trng.h | 6 +- + drivers/crypto/amlogic/amlogic-gxl-cipher.c | 25 +- + drivers/crypto/amlogic/amlogic-gxl-core.c | 48 +- + drivers/crypto/aspeed/aspeed-hace-crypto.c | 3 +- + drivers/crypto/atmel-aes.c | 2 +- + drivers/crypto/atmel-ecc.c | 12 +- + drivers/crypto/atmel-sha.c | 7 +- + drivers/crypto/atmel-tdes.c | 74 +- + drivers/crypto/cavium/cpt/cptpf_main.c | 12 +- + drivers/crypto/cavium/cpt/cptvf_main.c | 4 +- + drivers/crypto/cavium/cpt/cptvf_mbox.c | 4 +- + drivers/crypto/cavium/nitrox/nitrox_main.c | 2 +- + drivers/crypto/ccp/ccp-dev-v3.c | 4 + + drivers/crypto/ccp/ccp-dev-v5.c | 4 + + drivers/crypto/ccp/ccp-dev.c | 2 +- + drivers/crypto/ccp/dbc.c | 4 +- + drivers/crypto/ccp/sfs.c | 9 +- + drivers/crypto/ccp/sp-platform.c | 4 +- + drivers/crypto/ccree/cc_aead.c | 4 +- + drivers/crypto/ccree/cc_buffer_mgr.c | 4 +- + drivers/crypto/chelsio/chcr_algo.c | 2 +- + drivers/crypto/hisilicon/Kconfig | 14 - + drivers/crypto/hisilicon/Makefile | 1 - + drivers/crypto/hisilicon/hpre/hpre.h | 1 - + drivers/crypto/hisilicon/hpre/hpre_crypto.c | 28 +- + drivers/crypto/hisilicon/hpre/hpre_main.c | 47 +- + drivers/crypto/hisilicon/sec/Makefile | 3 - + drivers/crypto/hisilicon/sec/sec_algs.c | 1122 ----------------- + drivers/crypto/hisilicon/sec/sec_drv.c | 1307 -------------------- + drivers/crypto/hisilicon/sec/sec_drv.h | 428 ------- + drivers/crypto/hisilicon/zip/dae_main.c | 23 +- + drivers/crypto/inside-secure/eip93/eip93-aead.c | 3 +- + drivers/crypto/inside-secure/eip93/eip93-regs.h | 2 +- + drivers/crypto/inside-secure/safexcel_cipher.c | 16 +- + drivers/crypto/inside-secure/safexcel_hash.c | 6 +- + drivers/crypto/intel/qat/qat_common/adf_cfg.c | 84 +- + .../crypto/intel/qat/qat_common/adf_gen4_vf_mig.c | 5 + + drivers/crypto/intel/qat/qat_common/adf_gen6_ras.c | 20 +- + drivers/crypto/intel/qat/qat_common/adf_gen6_ras.h | 24 +- + drivers/crypto/intel/qat/qat_common/adf_init.c | 24 +- + .../crypto/intel/qat/qat_common/adf_sysfs_kpt.c | 24 +- + drivers/crypto/intel/qat/qat_common/qat_algs.c | 18 +- + .../crypto/intel/qat/qat_common/qat_asym_algs.c | 38 +- + drivers/crypto/marvell/octeontx/otx_cptvf_algs.c | 7 +- + drivers/crypto/marvell/octeontx/otx_cptvf_mbox.c | 4 +- + drivers/crypto/marvell/octeontx2/otx2_cptvf_algs.c | 7 +- + drivers/crypto/mxs-dcp.c | 3 + + drivers/crypto/omap-des.c | 2 + + drivers/crypto/omap-sham.c | 5 +- + drivers/crypto/padlock-aes.c | 22 +- + drivers/crypto/qce/aead.c | 11 +- + drivers/crypto/qce/sha.c | 9 +- + drivers/crypto/qce/skcipher.c | 9 +- + drivers/crypto/rockchip/rk3288_crypto.c | 4 +- + drivers/crypto/s5p-sss.c | 7 +- + drivers/crypto/sa2ul.c | 2 +- + drivers/crypto/starfive/jh7110-cryp.c | 20 +- + drivers/crypto/talitos.c | 21 +- + drivers/crypto/xilinx/zynqmp-aes-gcm.c | 14 +- + drivers/crypto/xilinx/zynqmp-sha.c | 2 +- + include/crypto/aes.h | 13 + + include/linux/ccp.h | 58 +- + include/linux/psp-sev.h | 86 +- + include/linux/rhashtable-types.h | 74 +- + include/linux/rhashtable.h | 19 +- + include/uapi/linux/psp-sev.h | 26 +- + kernel/padata.c | 2 +- + lib/842/842_decompress.c | 7 +- + lib/Kconfig.debug | 15 + + lib/rhashtable.c | 70 +- + lib/tests/842_decompress_kunit.c | 144 +++ + lib/tests/Makefile | 1 + + 97 files changed, 1057 insertions(+), 3854 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/crypto/hisilicon,hip06-sec.yaml + create mode 100644 crypto/asymmetric_keys/verify_pefile_test.c + delete mode 100644 drivers/crypto/hisilicon/sec/Makefile + delete mode 100644 drivers/crypto/hisilicon/sec/sec_algs.c + delete mode 100644 drivers/crypto/hisilicon/sec/sec_drv.c + delete mode 100644 drivers/crypto/hisilicon/sec/sec_drv.h + create mode 100644 lib/tests/842_decompress_kunit.c +Merging libcrypto/libcrypto-next (b63b9b3d5ddaf MAINTAINERS: Place libcrypto docs under CRYPTO LIBRARY) +$ git merge -m Merge branch 'libcrypto-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git libcrypto/libcrypto-next +Auto-merging MAINTAINERS +Auto-merging include/crypto/aes.h +Merge made by the 'ort' strategy. + Documentation/crypto/libcrypto-zeroization.rst | 150 +++++++++++++++++++++++++ + Documentation/crypto/libcrypto.rst | 14 +++ + MAINTAINERS | 1 + + include/crypto/aes-ccm.h | 22 +++- + include/crypto/aes-gcm.h | 22 +++- + include/crypto/aes-xts.h | 13 ++- + include/crypto/aes.h | 18 +++ + include/crypto/blake2b.h | 9 ++ + include/crypto/blake2s.h | 9 ++ + include/crypto/md5.h | 19 ++++ + include/crypto/sha1.h | 19 ++++ + include/crypto/sm3.h | 10 ++ + lib/crypto/aes.c | 28 ++--- + lib/crypto/blake2b.c | 2 +- + lib/crypto/blake2s.c | 2 +- + lib/crypto/md5.c | 2 +- + lib/crypto/sha1.c | 2 +- + lib/crypto/sm3.c | 2 +- + security/keys/trusted-keys/trusted_tpm1.c | 2 +- + 19 files changed, 318 insertions(+), 28 deletions(-) + create mode 100644 Documentation/crypto/libcrypto-zeroization.rst +Merging drm/drm-next (2abe8e7e33973 Merge tag 'amd-drm-next-7.4-2026-09-24' of https://gitlab.freedesktop.org/drm/amdgpu/kernel into drm-next) +$ git merge -m Merge branch 'drm-next' of https://gitlab.freedesktop.org/drm/kernel.git drm/drm-next +Auto-merging MAINTAINERS +Auto-merging drivers/accel/ivpu/ivpu_drv.c +Auto-merging drivers/accel/ivpu/ivpu_drv.h +Auto-merging drivers/accel/ivpu/ivpu_job.c +Auto-merging drivers/accel/ivpu/ivpu_mmu.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_debugfs.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_ring.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_userq.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c +Auto-merging drivers/gpu/drm/amd/amdgpu/vcn_v4_0_3.c +Auto-merging drivers/gpu/drm/amd/amdgpu/vcn_v5_0_1.c +Auto-merging drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c +Auto-merging drivers/gpu/drm/amd/display/dc/dml2_0/Makefile +Auto-merging drivers/gpu/drm/bridge/samsung-dsim.c +Auto-merging drivers/gpu/drm/clients/drm_fbdev_client.c +Auto-merging drivers/gpu/drm/drm_gpusvm.c +Auto-merging drivers/gpu/drm/i915/display/intel_cursor.c +Auto-merging drivers/gpu/drm/i915/display/intel_display_types.h +Auto-merging drivers/gpu/drm/i915/display/intel_dp_mst.c +Auto-merging drivers/gpu/drm/i915/display/intel_psr.c +Auto-merging drivers/gpu/drm/i915/display/intel_vrr.c +Auto-merging drivers/gpu/drm/i915/display/skl_universal_plane.c +Auto-merging drivers/gpu/drm/nouveau/nouveau_connector.c +CONFLICT (content): Merge conflict in drivers/gpu/drm/nouveau/nouveau_connector.c +Auto-merging drivers/gpu/drm/nouveau/nouveau_drm.c +Auto-merging drivers/gpu/drm/sti/sti_cursor.c +Auto-merging drivers/gpu/drm/sti/sti_hqvdp.c +Auto-merging drivers/gpu/drm/vc4/vc4_drv.c +Auto-merging drivers/gpu/drm/vc4/vc4_v3d.c +CONFLICT (content): Merge conflict in drivers/gpu/drm/vc4/vc4_v3d.c +Auto-merging drivers/gpu/drm/virtio/virtgpu_plane.c +Auto-merging drivers/gpu/drm/vmwgfx/vmwgfx_kms.c +Auto-merging drivers/gpu/drm/xe/regs/xe_gt_regs.h +Auto-merging drivers/gpu/drm/xe/xe_bo.c +Auto-merging drivers/gpu/drm/xe/xe_bo.h +Auto-merging drivers/gpu/drm/xe/xe_configfs.c +Auto-merging drivers/gpu/drm/xe/xe_tlb_inval.c +Auto-merging drivers/gpu/drm/xe/xe_vm.c +Resolved 'drivers/gpu/drm/nouveau/nouveau_connector.c' using previous resolution. +Resolved 'drivers/gpu/drm/vc4/vc4_v3d.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master de7ab1c71354d] Merge branch 'drm-next' of https://gitlab.freedesktop.org/drm/kernel.git +$ git diff -M --stat --summary HEAD^.. + .../bindings/display/brcm,bcm2835-v3d.yaml | 3 + + .../bindings/display/bridge/analogix,dp.yaml | 19 +- + .../display/bridge/renesas,r8a779g0-dsc.yaml | 96 + + .../bindings/display/panel/ilitek,ili7836a.yaml | 58 + + .../bindings/display/panel/novatek,nt36532.yaml | 83 + + .../panel/samsung,s6e8aa5x01-ams561ra01.yaml | 2 +- + .../display/rockchip/rockchip,analogix-dp.yaml | 1 + + .../bindings/display/solomon,ssd1351.yaml | 42 + + Documentation/gpu/amdgpu/index.rst | 1 + + Documentation/gpu/amdgpu/ualink.rst | 75 + + Documentation/gpu/drm-kms-helpers.rst | 11 +- + Documentation/gpu/drm-kms.rst | 3 + + Documentation/gpu/drm-ras.rst | 39 + + Documentation/gpu/drm-uapi.rst | 93 +- + Documentation/gpu/todo.rst | 15 - + Documentation/gpu/xe/index.rst | 1 + + Documentation/gpu/xe/xe_sigid.rst | 14 + + Documentation/netlink/specs/drm_ras.yaml | 80 + + MAINTAINERS | 55 +- + arch/arm/boot/dts/broadcom/bcm2835-common.dtsi | 1 + + drivers/accel/amdxdna/aie2_ctx.c | 9 +- + drivers/accel/amdxdna/aie2_message.c | 1 + + drivers/accel/amdxdna/amdxdna_gem.c | 460 +- + drivers/accel/amdxdna/amdxdna_gem.h | 6 +- + drivers/accel/ethosu/ethosu_job.c | 4 +- + drivers/accel/ivpu/ivpu_drv.c | 13 + + drivers/accel/ivpu/ivpu_drv.h | 5 +- + drivers/accel/ivpu/ivpu_hw.c | 22 + + drivers/accel/ivpu/ivpu_hw_ip.c | 6 +- + drivers/accel/ivpu/ivpu_job.c | 92 +- + drivers/accel/ivpu/ivpu_job.h | 5 +- + drivers/accel/ivpu/ivpu_mmu.c | 12 +- + drivers/accel/qaic/qaic_data.c | 8 +- + drivers/accel/qaic/qaic_debugfs.c | 1 + + drivers/accel/qaic/qaic_drv.c | 1 + + drivers/accel/qaic/qaic_ras.c | 1 + + drivers/accel/qaic/qaic_ssr.c | 1 + + drivers/accel/qaic/qaic_timesync.c | 1 + + drivers/accel/qaic/sahara.c | 1 + + drivers/dma-buf/dma-buf.c | 2 +- + drivers/dma-buf/dma-resv.c | 2 +- + drivers/firmware/efi/sysfb_efi.c | 9 - + drivers/gpu/buddy.c | 1342 +- + drivers/gpu/drm/Kconfig | 9 +- + drivers/gpu/drm/Kconfig.debug | 1 + + drivers/gpu/drm/Makefile | 3 +- + drivers/gpu/drm/adp/adp_drv.c | 2 +- + drivers/gpu/drm/amd/amdgpu/Makefile | 5 +- + drivers/gpu/drm/amd/amdgpu/amdgpu.h | 12 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_acp.c | 6 - + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd.c | 31 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd.h | 8 +- + .../gpu/drm/amd/amdgpu/amdgpu_amdkfd_aldebaran.c | 3 +- + .../gpu/drm/amd/amdgpu/amdgpu_amdkfd_gc_9_4_3.c | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v10.c | 151 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v10.h | 6 +- + .../gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v10_3.c | 1 + + .../gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c | 36 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v9.c | 14 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v9.h | 4 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gpuvm.c | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_cs.c | 4 - + drivers/gpu/drm/amd/amdgpu/amdgpu_ctx.c | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_debugfs.c | 15 + + drivers/gpu/drm/amd/amdgpu/amdgpu_dev_coredump.c | 8 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_device.c | 121 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_discovery.c | 340 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_discovery.h | 13 + + drivers/gpu/drm/amd/amdgpu/amdgpu_dma_buf.c | 5 + + drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c | 31 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gem.c | 9 + + drivers/gpu/drm/amd/amdgpu/amdgpu_gfx.c | 11 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gfx.h | 4 + + drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.c | 17 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.h | 3 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ib.c | 124 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ids.c | 37 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ids.h | 27 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ih.h | 2 + + drivers/gpu/drm/amd/amdgpu/amdgpu_imu.h | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ip.c | 68 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ip.h | 20 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_irq.c | 35 + + drivers/gpu/drm/amd/amdgpu/amdgpu_irq.h | 11 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_isp.c | 6 - + drivers/gpu/drm/amd/amdgpu/amdgpu_job.c | 5 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_job.h | 2 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_jpeg.c | 39 + + drivers/gpu/drm/amd/amdgpu/amdgpu_jpeg.h | 2 + + drivers/gpu/drm/amd/amdgpu/amdgpu_kms.c | 11 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_lsdma.c | 20 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_lsdma.h | 4 + + drivers/gpu/drm/amd/amdgpu/amdgpu_mca.c | 6 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c | 99 + + drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h | 11 + + drivers/gpu/drm/amd/amdgpu/amdgpu_mmhub.h | 3 + + drivers/gpu/drm/amd/amdgpu/amdgpu_object.c | 11 + + drivers/gpu/drm/amd/amdgpu/amdgpu_object.h | 5 + + drivers/gpu/drm/amd/amdgpu/amdgpu_psp.c | 588 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_psp.h | 86 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ras.c | 110 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ras.h | 3 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ras_eeprom.c | 8 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_res_cursor.h | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_reset.c | 7 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ring.c | 19 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ring.h | 50 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_sdma.c | 169 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_sdma.h | 165 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_sdma_types.h | 163 + + drivers/gpu/drm/amd/amdgpu/amdgpu_trace.h | 45 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_trace_points.c | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.c | 58 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.h | 4 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.c | 6539 ++ + drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.h | 427 + + drivers/gpu/drm/amd/amdgpu/amdgpu_umc.c | 46 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_umc.h | 8 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_userq.c | 46 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_userq_fence.c | 68 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_virt.c | 42 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_virt.h | 2 + + drivers/gpu/drm/amd/amdgpu/amdgpu_vkms.c | 10 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c | 105 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vm.h | 122 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_cpu.c | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h | 146 + + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_pt.c | 35 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_sdma.c | 6 + + drivers/gpu/drm/amd/amdgpu/amdgpu_vm_tlb_fence.c | 4 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_vpe.c | 34 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_xcp.c | 18 +- + drivers/gpu/drm/amd/amdgpu/amdgpu_xcp.h | 7 + + drivers/gpu/drm/amd/amdgpu/amdgv_sriovmsg.h | 49 +- + drivers/gpu/drm/amd/amdgpu/aqua_vanjaram.c | 42 +- + drivers/gpu/drm/amd/amdgpu/atom.c | 29 +- + drivers/gpu/drm/amd/amdgpu/cik.c | 7 - + drivers/gpu/drm/amd/amdgpu/cik_ih.c | 12 - + drivers/gpu/drm/amd/amdgpu/cik_sdma.c | 68 +- + drivers/gpu/drm/amd/amdgpu/cz_ih.c | 12 - + drivers/gpu/drm/amd/amdgpu/dce_v10_0.c | 6 - + drivers/gpu/drm/amd/amdgpu/dce_v6_0.c | 6 - + drivers/gpu/drm/amd/amdgpu/dce_v8_0.c | 6 - + drivers/gpu/drm/amd/amdgpu/gfx_v10_0.c | 44 +- + drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c | 44 +- + drivers/gpu/drm/amd/amdgpu/gfx_v12_0.c | 44 +- + drivers/gpu/drm/amd/amdgpu/gfx_v12_1.c | 797 +- + drivers/gpu/drm/amd/amdgpu/gfx_v12_1_pkt.h | 39 - + drivers/gpu/drm/amd/amdgpu/gfx_v6_0.c | 26 +- + drivers/gpu/drm/amd/amdgpu/gfx_v7_0.c | 39 +- + drivers/gpu/drm/amd/amdgpu/gfx_v8_0.c | 153 +- + drivers/gpu/drm/amd/amdgpu/gfx_v9_0.c | 146 +- + drivers/gpu/drm/amd/amdgpu/gfx_v9_4_2.c | 111 +- + drivers/gpu/drm/amd/amdgpu/gfx_v9_4_2.h | 1 + + drivers/gpu/drm/amd/amdgpu/gfx_v9_4_3.c | 41 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v11_5_0.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v12_0.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v12_1.c | 41 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v1_0.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v1_2.c | 3 + + drivers/gpu/drm/amd/amdgpu/gfxhub_v2_0.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v2_1.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v3_0.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gfxhub_v3_0_3.c | 9 +- + drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c | 12 +- + drivers/gpu/drm/amd/amdgpu/gmc_v11_0.c | 18 +- + drivers/gpu/drm/amd/amdgpu/gmc_v12_0.c | 59 +- + drivers/gpu/drm/amd/amdgpu/gmc_v12_1.c | 184 +- + drivers/gpu/drm/amd/amdgpu/gmc_v12_1.h | 2 + + drivers/gpu/drm/amd/amdgpu/gmc_v6_0.c | 7 +- + drivers/gpu/drm/amd/amdgpu/gmc_v7_0.c | 7 +- + drivers/gpu/drm/amd/amdgpu/gmc_v8_0.c | 19 +- + drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c | 14 +- + drivers/gpu/drm/amd/amdgpu/iceland_ih.c | 12 - + drivers/gpu/drm/amd/amdgpu/ih_v6_0.c | 39 +- + drivers/gpu/drm/amd/amdgpu/ih_v6_1.c | 16 +- + drivers/gpu/drm/amd/amdgpu/ih_v7_0.c | 43 +- + drivers/gpu/drm/amd/amdgpu/imu_v12_1.c | 91 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v2_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/jpeg_v2_5.c | 2 - + drivers/gpu/drm/amd/amdgpu/jpeg_v3_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/jpeg_v4_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/jpeg_v4_0_3.c | 2 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v4_0_5.c | 2 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v5_0_0.c | 2 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v5_0_1.c | 4 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v5_0_1.h | 8 - + drivers/gpu/drm/amd/amdgpu/jpeg_v5_0_2.c | 265 +- + drivers/gpu/drm/amd/amdgpu/jpeg_v5_3_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/lsdma_v7_1.c | 27 +- + drivers/gpu/drm/amd/amdgpu/mes_userqueue.c | 62 +- + drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 35 +- + drivers/gpu/drm/amd/amdgpu/mes_v12_0.c | 29 +- + drivers/gpu/drm/amd/amdgpu/mes_v12_1.c | 282 +- + drivers/gpu/drm/amd/amdgpu/mmhub_v4_2_0.c | 372 +- + drivers/gpu/drm/amd/amdgpu/navi10_ih.c | 11 +- + drivers/gpu/drm/amd/amdgpu/nbif_v6_3_1.c | 155 +- + drivers/gpu/drm/amd/amdgpu/nbio_v4_3.c | 2 +- + drivers/gpu/drm/amd/amdgpu/nbio_v6_3_2.c | 18 +- + drivers/gpu/drm/amd/amdgpu/nbio_v7_11_5.c | 351 + + drivers/gpu/drm/amd/amdgpu/nbio_v7_11_5.h | 32 + + drivers/gpu/drm/amd/amdgpu/nbio_v7_9.c | 1 + + drivers/gpu/drm/amd/amdgpu/nv.c | 6 - + drivers/gpu/drm/amd/amdgpu/psp_gfx_if.h | 162 + + drivers/gpu/drm/amd/amdgpu/psp_v15_0_8.c | 146 + + drivers/gpu/drm/amd/amdgpu/sdma_v2_4.c | 105 +- + drivers/gpu/drm/amd/amdgpu/sdma_v3_0.c | 113 +- + drivers/gpu/drm/amd/amdgpu/sdma_v4_0.c | 124 +- + drivers/gpu/drm/amd/amdgpu/sdma_v4_4_2.c | 250 +- + drivers/gpu/drm/amd/amdgpu/sdma_v5_0.c | 141 +- + drivers/gpu/drm/amd/amdgpu/sdma_v5_2.c | 133 +- + drivers/gpu/drm/amd/amdgpu/sdma_v6_0.c | 143 +- + drivers/gpu/drm/amd/amdgpu/sdma_v7_0.c | 142 +- + drivers/gpu/drm/amd/amdgpu/sdma_v7_1.c | 114 +- + drivers/gpu/drm/amd/amdgpu/si.c | 6 - + drivers/gpu/drm/amd/amdgpu/si_dma.c | 38 +- + drivers/gpu/drm/amd/amdgpu/si_ih.c | 1 - + drivers/gpu/drm/amd/amdgpu/soc15.c | 6 - + drivers/gpu/drm/amd/amdgpu/soc15_common.h | 6 +- + drivers/gpu/drm/amd/amdgpu/soc21.c | 6 - + drivers/gpu/drm/amd/amdgpu/soc24.c | 6 - + drivers/gpu/drm/amd/amdgpu/soc_v1_0.c | 493 +- + drivers/gpu/drm/amd/amdgpu/soc_v1_0.h | 3 + + drivers/gpu/drm/amd/amdgpu/ta_ras_if.h | 1 + + drivers/gpu/drm/amd/amdgpu/tonga_ih.c | 12 - + drivers/gpu/drm/amd/amdgpu/ualink_v1_0.c | 143 + + drivers/gpu/drm/amd/amdgpu/ualink_v1_0.h | 30 + + drivers/gpu/drm/amd/amdgpu/uvd_v3_1.c | 8 - + drivers/gpu/drm/amd/amdgpu/uvd_v4_2.c | 8 - + drivers/gpu/drm/amd/amdgpu/uvd_v5_0.c | 8 - + drivers/gpu/drm/amd/amdgpu/uvd_v6_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vce_v1_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vce_v2_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vce_v3_0.c | 1 - + drivers/gpu/drm/amd/amdgpu/vcn_v1_0.c | 3 +- + drivers/gpu/drm/amd/amdgpu/vcn_v2_0.c | 2 +- + drivers/gpu/drm/amd/amdgpu/vcn_v2_5.c | 3 +- + drivers/gpu/drm/amd/amdgpu/vcn_v3_0.c | 17 +- + drivers/gpu/drm/amd/amdgpu/vcn_v4_0.c | 24 +- + drivers/gpu/drm/amd/amdgpu/vcn_v4_0_3.c | 22 +- + drivers/gpu/drm/amd/amdgpu/vcn_v4_0_5.c | 24 +- + drivers/gpu/drm/amd/amdgpu/vcn_v5_0_0.c | 24 +- + drivers/gpu/drm/amd/amdgpu/vcn_v5_0_1.c | 20 +- + drivers/gpu/drm/amd/amdgpu/vcn_v5_0_2.c | 351 +- + drivers/gpu/drm/amd/amdgpu/vega10_ih.c | 7 - + drivers/gpu/drm/amd/amdgpu/vega20_ih.c | 7 - + drivers/gpu/drm/amd/amdgpu/vi.c | 6 - + drivers/gpu/drm/amd/amdkfd/cwsr_trap_handler.h | 90 +- + .../gpu/drm/amd/amdkfd/cwsr_trap_handler_gfx12.asm | 122 +- + drivers/gpu/drm/amd/amdkfd/kfd_device.c | 18 +- + .../gpu/drm/amd/amdkfd/kfd_device_queue_manager.c | 151 +- + .../gpu/drm/amd/amdkfd/kfd_device_queue_manager.h | 3 + + drivers/gpu/drm/amd/amdkfd/kfd_flat_memory.c | 6 + + drivers/gpu/drm/amd/amdkfd/kfd_int_process_v9.c | 23 + + drivers/gpu/drm/amd/amdkfd/kfd_migrate.c | 53 +- + drivers/gpu/drm/amd/amdkfd/kfd_mqd_manager_v12_1.c | 31 +- + drivers/gpu/drm/amd/amdkfd/kfd_packet_manager_v9.c | 3 +- + drivers/gpu/drm/amd/amdkfd/kfd_process.c | 46 +- + .../gpu/drm/amd/amdkfd/kfd_process_queue_manager.c | 10 +- + drivers/gpu/drm/amd/amdkfd/kfd_queue.c | 10 +- + drivers/gpu/drm/amd/amdkfd/kfd_svm.c | 7 +- + drivers/gpu/drm/amd/display/Kconfig | 12 +- + drivers/gpu/drm/amd/display/Makefile | 1 + + drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.c | 454 +- + drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm.h | 182 +- + .../amd/display/amdgpu_dm/amdgpu_dm_backlight.c | 33 +- + .../amd/display/amdgpu_dm/amdgpu_dm_backlight.h | 1 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_color.c | 85 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_colorop.c | 137 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_colorop.h | 29 + + .../amd/display/amdgpu_dm/amdgpu_dm_connector.c | 379 +- + .../amd/display/amdgpu_dm/amdgpu_dm_connector.h | 11 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_crtc.c | 87 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_crtc.h | 9 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_cursor.c | 63 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_cursor.h | 6 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_debugfs.c | 88 + + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_dmub.c | 47 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_dmub.h | 15 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_freesync.c | 47 + + .../drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c | 32 + + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.c | 8 + + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_irq.h | 9 + + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_ism.c | 21 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_ism.h | 2 +- + .../amd/display/amdgpu_dm/amdgpu_dm_mst_types.c | 73 +- + .../amd/display/amdgpu_dm/amdgpu_dm_mst_types.h | 25 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_plane.c | 109 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_plane.h | 14 +- + .../drm/amd/display/amdgpu_dm/amdgpu_dm_services.c | 35 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_wb.c | 72 +- + .../gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_wb.h | 16 + + .../drm/amd/display/amdgpu_dm/tests/.kunitconfig | 1 - + .../gpu/drm/amd/display/amdgpu_dm/tests/Makefile | 4 +- + .../amdgpu_dm/tests/amdgpu_dm_backlight_test.c | 53 +- + .../amdgpu_dm/tests/amdgpu_dm_colorop_test.c | 232 +- + .../amdgpu_dm/tests/amdgpu_dm_connector_test.c | 4632 +- + .../display/amdgpu_dm/tests/amdgpu_dm_crtc_test.c | 813 +- + .../amdgpu_dm/tests/amdgpu_dm_cursor_test.c | 736 + + .../display/amdgpu_dm/tests/amdgpu_dm_dmub_test.c | 578 +- + .../amdgpu_dm/tests/amdgpu_dm_freesync_test.c | 429 + + .../amdgpu_dm/tests/amdgpu_dm_helpers_test.c | 1129 +- + .../display/amdgpu_dm/tests/amdgpu_dm_irq_test.c | 2308 +- + .../display/amdgpu_dm/tests/amdgpu_dm_ism_test.c | 436 +- + .../amdgpu_dm/tests/amdgpu_dm_mst_types_test.c | 2791 +- + .../display/amdgpu_dm/tests/amdgpu_dm_plane_test.c | 2660 +- + .../amdgpu_dm/tests/amdgpu_dm_services_test.c | 228 +- + .../amd/display/amdgpu_dm/tests/amdgpu_dm_test.c | 4282 +- + .../display/amdgpu_dm/tests/amdgpu_dm_wb_test.c | 334 + + drivers/gpu/drm/amd/display/dc/Makefile | 5 +- + .../drm/amd/display/dc/bios/command_table_helper.c | 7 +- + .../amd/display/dc/bios/command_table_helper2.c | 2 + + drivers/gpu/drm/amd/display/dc/clk_mgr/Makefile | 8 + + drivers/gpu/drm/amd/display/dc/clk_mgr/clk_mgr.c | 4 + + .../amd/display/dc/clk_mgr/dce112/dce112_clk_mgr.c | 12 +- + .../amd/display/dc/clk_mgr/dcn10/dcn10_clk_mgr.c | 181 + + .../amd/display/dc/clk_mgr/dcn10/dcn10_clk_mgr.h | 29 + + .../drm/amd/display/dc/clk_mgr/dcn10/rv1_clk_mgr.c | 9 +- + .../drm/amd/display/dc/clk_mgr/dcn10/rv2_clk_mgr.c | 6 +- + .../amd/display/dc/clk_mgr/dcn20/dcn20_clk_mgr.c | 12 +- + .../amd/display/dc/clk_mgr/dcn201/dcn201_clk_mgr.c | 6 +- + .../drm/amd/display/dc/clk_mgr/dcn21/rn_clk_mgr.c | 5 +- + .../amd/display/dc/clk_mgr/dcn30/dcn30_clk_mgr.c | 8 +- + .../drm/amd/display/dc/clk_mgr/dcn301/vg_clk_mgr.c | 8 +- + .../amd/display/dc/clk_mgr/dcn31/dcn31_clk_mgr.c | 10 +- + .../amd/display/dc/clk_mgr/dcn314/dcn314_clk_mgr.c | 10 +- + .../amd/display/dc/clk_mgr/dcn315/dcn315_clk_mgr.c | 14 +- + .../amd/display/dc/clk_mgr/dcn316/dcn316_clk_mgr.c | 10 +- + .../amd/display/dc/clk_mgr/dcn32/dcn32_clk_mgr.c | 18 +- + .../amd/display/dc/clk_mgr/dcn35/dcn35_clk_mgr.c | 12 +- + .../amd/display/dc/clk_mgr/dcn401/dcn401_clk_mgr.c | 19 +- + .../amd/display/dc/clk_mgr/dcn42/dcn42_clk_mgr.c | 33 +- + .../amd/display/dc/clk_mgr/dcn42/dcn42_clk_mgr.h | 5 + + .../amd/display/dc/clk_mgr/dcn42b/dcn42b_clk_mgr.c | 54 +- + .../gpu/drm/amd/display/dc/clk_mgr/dcn60/dalsmc.h | 345 +- + .../amd/display/dc/clk_mgr/dcn60/dcn60_clk_mgr.c | 341 +- + .../amd/display/dc/clk_mgr/dcn60/dcn60_clk_mgr.h | 17 +- + .../dc/clk_mgr/dcn60/dcn60_clk_mgr_smu_msg.c | 184 +- + .../dc/clk_mgr/dcn60/dcn60_clk_mgr_smu_msg.h | 167 +- + .../display/dc/clk_mgr/dcn60/dcn60_smu_driver_if.h | 78 + + drivers/gpu/drm/amd/display/dc/core/dc.c | 291 +- + drivers/gpu/drm/amd/display/dc/core/dc_debug.c | 2 + + .../gpu/drm/amd/display/dc/core/dc_hw_sequencer.c | 1035 +- + .../gpu/drm/amd/display/dc/core/dc_link_exports.c | 25 +- + drivers/gpu/drm/amd/display/dc/core/dc_resource.c | 243 +- + drivers/gpu/drm/amd/display/dc/core/dc_state.c | 6 +- + drivers/gpu/drm/amd/display/dc/core/dc_stream.c | 168 +- + drivers/gpu/drm/amd/display/dc/dc.h | 64 +- + drivers/gpu/drm/amd/display/dc/dc_dmub_srv.c | 281 +- + drivers/gpu/drm/amd/display/dc/dc_dmub_srv.h | 58 +- + drivers/gpu/drm/amd/display/dc/dc_edid_parser.c | 80 - + drivers/gpu/drm/amd/display/dc/dc_hw_types.h | 21 +- + drivers/gpu/drm/amd/display/dc/dc_memory_pool.c | 275 + + drivers/gpu/drm/amd/display/dc/dc_memory_pool.h | 107 + + drivers/gpu/drm/amd/display/dc/dc_probe.h | 1 + + drivers/gpu/drm/amd/display/dc/dc_stream.h | 14 +- + drivers/gpu/drm/amd/display/dc/dc_types.h | 6 +- + .../gpu/drm/amd/display/dc/dccg/dcn20/dcn20_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn20/dcn20_dccg.h | 4 + + .../drm/amd/display/dc/dccg/dcn201/dcn201_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn21/dcn21_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn30/dcn30_dccg.c | 2 + + .../drm/amd/display/dc/dccg/dcn301/dcn301_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn31/dcn31_dccg.c | 1 + + .../drm/amd/display/dc/dccg/dcn314/dcn314_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn32/dcn32_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn35/dcn35_dccg.c | 2 + + .../drm/amd/display/dc/dccg/dcn401/dcn401_dccg.c | 1 + + .../gpu/drm/amd/display/dc/dccg/dcn42/dcn42_dccg.c | 15 +- + .../gpu/drm/amd/display/dc/dccg/dcn42/dcn42_dccg.h | 2 +- + .../gpu/drm/amd/display/dc/dccg/dcn60/dcn60_dccg.c | 5 + + .../gpu/drm/amd/display/dc/dccg/dcn60/dcn60_dccg.h | 4 + + drivers/gpu/drm/amd/display/dc/dce/dce_abm.c | 2 + + drivers/gpu/drm/amd/display/dc/dce/dce_aux.c | 12 +- + .../gpu/drm/amd/display/dc/dce/dce_clock_source.c | 11 +- + drivers/gpu/drm/amd/display/dc/dce/dce_dmcu.c | 121 - + drivers/gpu/drm/amd/display/dc/dce/dmub_abm.c | 1 + + .../gpu/drm/amd/display/dc/dce/dmub_hw_lock_mgr.c | 10 +- + drivers/gpu/drm/amd/display/dc/dce/dmub_psr.c | 1 + + drivers/gpu/drm/amd/display/dc/dce/dmub_psr.h | 1 + + drivers/gpu/drm/amd/display/dc/dce/dmub_replay.c | 28 +- + drivers/gpu/drm/amd/display/dc/dce/dmub_replay.h | 3 + + drivers/gpu/drm/amd/display/dc/dcn20/dcn20_vmid.c | 2 +- + .../gpu/drm/amd/display/dc/dcn30/dcn30_cm_common.c | 8 +- + .../drm/amd/display/dc/dcn301/dcn301_panel_cntl.c | 1 + + .../drm/amd/display/dc/dcn31/dcn31_panel_cntl.c | 1 + + .../gpu/drm/amd/display/dc/dio/dcn10/dcn10_dio.c | 1 + + .../amd/display/dc/dio/dcn10/dcn10_link_encoder.c | 12 +- + .../display/dc/dio/dcn10/dcn10_stream_encoder.c | 2 + + .../amd/display/dc/dio/dcn20/dcn20_link_encoder.c | 2 + + .../display/dc/dio/dcn20/dcn20_stream_encoder.c | 2 + + .../display/dc/dio/dcn30/dcn30_dio_link_encoder.c | 2 + + .../dc/dio/dcn30/dcn30_dio_stream_encoder.c | 2 + + .../dc/dio/dcn301/dcn301_dio_link_encoder.c | 2 + + .../display/dc/dio/dcn31/dcn31_dio_link_encoder.c | 2 + + .../dc/dio/dcn314/dcn314_dio_stream_encoder.c | 2 + + .../display/dc/dio/dcn32/dcn32_dio_link_encoder.c | 2 + + .../dc/dio/dcn32/dcn32_dio_stream_encoder.c | 2 + + .../dc/dio/dcn321/dcn321_dio_link_encoder.c | 2 + + .../display/dc/dio/dcn35/dcn35_dio_link_encoder.c | 2 + + .../dc/dio/dcn35/dcn35_dio_stream_encoder.c | 2 + + .../dc/dio/dcn401/dcn401_dio_link_encoder.c | 2 + + .../dc/dio/dcn401/dcn401_dio_stream_encoder.c | 1 + + .../display/dc/dio/dcn42/dcn42_dio_link_encoder.c | 6 +- + .../dc/dio/dcn42/dcn42_dio_stream_encoder.c | 2 + + .../display/dc/dio/dcn60/dcn60_dio_link_encoder.c | 2 + + .../dc/dio/dcn60/dcn60_dio_stream_encoder.c | 2 + + .../display/dc/dio/virtual/virtual_link_encoder.c | 3 + + .../dc/dio/virtual/virtual_stream_encoder.c | 1 + + .../gpu/drm/amd/display/dc/dml/dcn10/dcn10_fpu.c | 1 - + .../gpu/drm/amd/display/dc/dml/dcn30/dcn30_fpu.c | 18 +- + .../amd/display/dc/dml/dcn30/display_mode_vba_30.c | 2 +- + .../amd/display/dc/dml/dcn31/display_mode_vba_31.c | 2 +- + .../gpu/drm/amd/display/dc/dml/dml1_frl_cap_chk.c | 8 +- + .../gpu/drm/amd/display/dc/dml/dsc/rc_calc_fpu.c | 1 - + drivers/gpu/drm/amd/display/dc/dml2_0/Makefile | 17 +- + .../amd/display/dc/dml2_0/dml21/dml21_wrapper.h | 106 - + .../dml2_0/dml21/inc/bounding_boxes/dcn42_soc_bb.h | 39 +- + .../dml21/inc/bounding_boxes/dcn42b_soc_bb.h | 37 +- + .../dml2_0/dml21/inc/bounding_boxes/dcn4_soc_bb.h | 27 +- + .../dml2_0/dml21/inc/bounding_boxes/dcn6_soc_bb.h | 12 + + .../dc/dml2_0/dml21/inc/dml_top_dchub_registers.h | 1 + + .../dml2_0/dml21/inc/dml_top_display_cfg_types.h | 1 + + .../dml2_0/dml21/inc/dml_top_soc_parameter_types.h | 28 +- + .../display/dc/dml2_0/dml21/inc/dml_top_types.h | 14 + + .../dc/dml2_0/dml21/src/dml2_cga/dml2_cga_dcn6.c | 4 +- + .../dc/dml2_0/dml21/src/dml2_core/dml2_core_dcn4.c | 4 + + .../dml21/src/dml2_core/dml2_core_dcn4_calcs.c | 222 +- + .../dml21/src/dml2_core/dml2_core_dcn4_calcs.h | 1 + + .../src/dml2_core/dml2_core_dcn5_calcs_dchub.c | 18 +- + .../src/dml2_core/dml2_core_dcn5_calcs_dchub.h | 23 + + .../dml2_core/dml2_core_dcn5_calcs_display_pipe.c | 2 +- + .../dml2_core/dml2_core_dcn5_funcs_initialize.c | 2 +- + .../dml2_core_dcn5_funcs_mode_programming.c | 4 + + .../dml2_core/dml2_core_dcn5_funcs_mode_support.c | 32 +- + .../dml21/src/dml2_core/dml2_core_dcn6_calcs.c | 63 +- + .../dml21/src/dml2_core/dml2_core_dcn6_calcs.h | 468 + + .../src/dml2_core/dml2_core_dcn6_calcs_dchub.c | 86 +- + .../dml2_core/dml2_core_dcn6_funcs_initialize.c | 4 + + .../dml2_core_dcn6_funcs_mode_programming.c | 52 +- + .../dml2_core_dcn6_funcs_mode_programming.h | 31 + + .../dml2_core/dml2_core_dcn6_funcs_mode_support.c | 350 +- + .../dml2_core/dml2_core_dcn6_funcs_mode_support.h | 324 + + .../dml21/src/dml2_core/dml2_core_shared_types.h | 11 +- + .../dc/dml2_0/dml21/src/dml2_dpmm/dml2_dpmm_dcn4.c | 106 +- + .../dml21/src/dml2_pmo/dml2_pmo_dcn4_fams2.c | 11 +- + .../alternate_pstate_shared_lib.c | 22 +- + .../dc/dml2_0/dml21/src/dml2_top/dml2_top_utm.c | 4 +- + .../src/dml2_utm_soc_bb/dml2_utm_soc_bb_dcn6.c | 15 +- + .../dml21/src/inc/dml2_internal_shared_types.h | 19 + + .../gpu/drm/amd/display/dc/dml2_wrapper/Makefile | 61 + + .../dml21_wrapper}/dml21_translation_helper.c | 46 +- + .../dml21_wrapper}/dml21_translation_helper.h | 7 +- + .../dml21_wrapper}/dml21_utils.c | 24 +- + .../dml21_wrapper}/dml21_utils.h | 7 +- + .../dml21_wrapper}/dml21_wrapper.c | 2 +- + .../dc/dml2_wrapper/dml21_wrapper/dml21_wrapper.h | 107 + + .../dml21_wrapper}/dml21_wrapper_fpu.c | 19 +- + .../dml21_wrapper}/dml21_wrapper_fpu.h | 7 +- + .../dml2_dc_resource_mgmt.c | 60 +- + .../dml2_dc_resource_mgmt.h | 0 + .../dc/{dml2_0 => dml2_wrapper}/dml2_dc_types.h | 0 + .../{dml2_0 => dml2_wrapper}/dml2_internal_types.h | 2 +- + .../{dml2_0 => dml2_wrapper}/dml2_mall_phantom.c | 70 +- + .../{dml2_0 => dml2_wrapper}/dml2_mall_phantom.h | 0 + .../dc/{dml2_0 => dml2_wrapper}/dml2_policy.c | 2 +- + .../dc/{dml2_0 => dml2_wrapper}/dml2_policy.h | 0 + .../dml2_translation_helper.c | 20 +- + .../dml2_translation_helper.h | 0 + .../dc/{dml2_0 => dml2_wrapper}/dml2_utils.c | 28 +- + .../dc/{dml2_0 => dml2_wrapper}/dml2_utils.h | 0 + .../dc/{dml2_0 => dml2_wrapper}/dml2_wrapper.c | 6 +- + .../dc/{dml2_0 => dml2_wrapper}/dml2_wrapper.h | 0 + .../dc/{dml2_0 => dml2_wrapper}/dml2_wrapper_fpu.c | 45 +- + .../dc/{dml2_0 => dml2_wrapper}/dml2_wrapper_fpu.h | 0 + .../drm/amd/display/dc/dpp/dcn30/dcn30_dpp_cm.c | 2 + + .../gpu/drm/amd/display/dc/dpp/dcn401/dcn401_dpp.h | 2 + + .../drm/amd/display/dc/dpp/dcn401/dcn401_dpp_cm.c | 19 + + .../gpu/drm/amd/display/dc/dpp/dcn50/dcn50_dpp.c | 20 +- + .../gpu/drm/amd/display/dc/dpp/dcn50/dcn50_dpp.h | 14 +- + .../gpu/drm/amd/display/dc/dpp/dcn60/dcn60_dpp.c | 1 + + .../gpu/drm/amd/display/dc/dpp/dcn60/dcn60_dpp.h | 7 +- + drivers/gpu/drm/amd/display/dc/dsc/dc_dsc.c | 8 + + .../gpu/drm/amd/display/dc/dsc/dcn60/dcn60_dsc.c | 4 +- + drivers/gpu/drm/amd/display/dc/gpio/hw_ddc.c | 7 +- + drivers/gpu/drm/amd/display/dc/gpio/hw_factory.c | 4 + + drivers/gpu/drm/amd/display/dc/gpio/hw_translate.c | 4 + + .../dc/hpo/dcn30/dcn30_hpo_frl_stream_encoder.c | 8 +- + .../dc/hpo/dcn401/dcn401_hpo_frl_stream_encoder.c | 2 + + .../dc/hpo/dcn42/dcn42_hpo_frl_stream_encoder.c | 11 +- + .../dc/hpo/dcn60/dcn60_hpo_frl_stream_encoder.c | 142 +- + .../drm/amd/display/dc/hubbub/dcn10/dcn10_hubbub.c | 2 + + .../drm/amd/display/dc/hubbub/dcn10/dcn10_hubbub.h | 11 +- + .../drm/amd/display/dc/hubbub/dcn20/dcn20_hubbub.c | 7 +- + .../amd/display/dc/hubbub/dcn201/dcn201_hubbub.c | 2 + + .../drm/amd/display/dc/hubbub/dcn21/dcn21_hubbub.c | 2 + + .../drm/amd/display/dc/hubbub/dcn30/dcn30_hubbub.c | 2 + + .../amd/display/dc/hubbub/dcn301/dcn301_hubbub.c | 2 + + .../drm/amd/display/dc/hubbub/dcn31/dcn31_hubbub.c | 6 +- + .../drm/amd/display/dc/hubbub/dcn32/dcn32_hubbub.c | 2 + + .../drm/amd/display/dc/hubbub/dcn35/dcn35_hubbub.c | 16 +- + .../amd/display/dc/hubbub/dcn401/dcn401_hubbub.c | 2 + + .../drm/amd/display/dc/hubbub/dcn42/dcn42_hubbub.c | 15 +- + .../drm/amd/display/dc/hubbub/dcn60/dcn60_hubbub.c | 241 +- + .../drm/amd/display/dc/hubbub/dcn60/dcn60_hubbub.h | 19 +- + .../gpu/drm/amd/display/dc/hubp/dcn10/dcn10_hubp.c | 4 +- + .../gpu/drm/amd/display/dc/hubp/dcn10/dcn10_hubp.h | 3 +- + .../gpu/drm/amd/display/dc/hubp/dcn20/dcn20_hubp.c | 5 +- + .../gpu/drm/amd/display/dc/hubp/dcn20/dcn20_hubp.h | 3 +- + .../gpu/drm/amd/display/dc/hubp/dcn21/dcn21_hubp.c | 11 +- + .../gpu/drm/amd/display/dc/hubp/dcn21/dcn21_hubp.h | 2 +- + .../gpu/drm/amd/display/dc/hubp/dcn30/dcn30_hubp.c | 7 +- + .../gpu/drm/amd/display/dc/hubp/dcn30/dcn30_hubp.h | 3 +- + .../gpu/drm/amd/display/dc/hubp/dcn31/dcn31_hubp.c | 1 + + .../drm/amd/display/dc/hubp/dcn401/dcn401_hubp.c | 114 +- + .../drm/amd/display/dc/hubp/dcn401/dcn401_hubp.h | 5 +- + .../gpu/drm/amd/display/dc/hubp/dcn42/dcn42_hubp.c | 6 +- + .../gpu/drm/amd/display/dc/hubp/dcn50/dcn50_hubp.c | 28 +- + .../gpu/drm/amd/display/dc/hubp/dcn50/dcn50_hubp.h | 3 +- + .../gpu/drm/amd/display/dc/hubp/dcn60/dcn60_hubp.c | 4 + + drivers/gpu/drm/amd/display/dc/hwss/Makefile | 6 + + .../gpu/drm/amd/display/dc/hwss/dce/dce_hwseq.c | 46 +- + .../gpu/drm/amd/display/dc/hwss/dce/dce_hwseq.h | 12 +- + .../drm/amd/display/dc/hwss/dce110/dce110_hwseq.c | 35 +- + .../drm/amd/display/dc/hwss/dce60/dce60_hwseq.c | 8 +- + .../drm/amd/display/dc/hwss/dce80/dce80_hwseq.c | 3 +- + .../drm/amd/display/dc/hwss/dcn10/dcn10_hwseq.c | 280 +- + .../drm/amd/display/dc/hwss/dcn10/dcn10_hwseq.h | 22 +- + .../gpu/drm/amd/display/dc/hwss/dcn10/dcn10_init.c | 6 +- + .../drm/amd/display/dc/hwss/dcn20/dcn20_hwseq.c | 291 +- + .../drm/amd/display/dc/hwss/dcn20/dcn20_hwseq.h | 26 +- + .../gpu/drm/amd/display/dc/hwss/dcn20/dcn20_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn201/dcn201_hwseq.c | 61 +- + .../drm/amd/display/dc/hwss/dcn201/dcn201_hwseq.h | 7 +- + .../drm/amd/display/dc/hwss/dcn201/dcn201_init.c | 6 +- + .../gpu/drm/amd/display/dc/hwss/dcn21/dcn21_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.c | 57 +- + .../drm/amd/display/dc/hwss/dcn30/dcn30_hwseq.h | 8 +- + .../gpu/drm/amd/display/dc/hwss/dcn30/dcn30_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn301/dcn301_init.c | 5 +- + .../gpu/drm/amd/display/dc/hwss/dcn31/dcn31_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn314/dcn314_hwseq.c | 8 +- + .../drm/amd/display/dc/hwss/dcn314/dcn314_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn32/dcn32_hwseq.c | 104 +- + .../drm/amd/display/dc/hwss/dcn32/dcn32_hwseq.h | 10 +- + .../gpu/drm/amd/display/dc/hwss/dcn32/dcn32_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn35/dcn35_hwseq.c | 128 +- + .../drm/amd/display/dc/hwss/dcn35/dcn35_hwseq.h | 14 +- + .../gpu/drm/amd/display/dc/hwss/dcn35/dcn35_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn351/dcn351_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn401/dcn401_hwseq.c | 314 +- + .../drm/amd/display/dc/hwss/dcn401/dcn401_hwseq.h | 24 +- + .../drm/amd/display/dc/hwss/dcn401/dcn401_init.c | 5 +- + .../drm/amd/display/dc/hwss/dcn42/dcn42_hwseq.c | 330 +- + .../drm/amd/display/dc/hwss/dcn42/dcn42_hwseq.h | 21 +- + .../gpu/drm/amd/display/dc/hwss/dcn42/dcn42_init.c | 10 +- + .../drm/amd/display/dc/hwss/dcn50/dcn50_hwseq.c | 19 +- + .../drm/amd/display/dc/hwss/dcn60/dcn60_hwseq.c | 244 +- + .../drm/amd/display/dc/hwss/dcn60/dcn60_hwseq.h | 5 +- + .../gpu/drm/amd/display/dc/hwss/dcn60/dcn60_init.c | 12 +- + drivers/gpu/drm/amd/display/dc/hwss/hw_sequencer.h | 259 +- + .../drm/amd/display/dc/hwss/hw_sequencer_private.h | 28 +- + drivers/gpu/drm/amd/display/dc/inc/clock_source.h | 4 +- + drivers/gpu/drm/amd/display/dc/inc/core_status.h | 2 + + drivers/gpu/drm/amd/display/dc/inc/core_types.h | 13 +- + drivers/gpu/drm/amd/display/dc/inc/custom_float.h | 21 +- + drivers/gpu/drm/amd/display/dc/inc/hw/abm.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/clk_mgr.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/dccg.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/dchubbub.h | 2 + + drivers/gpu/drm/amd/display/dc/inc/hw/dio.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/dmcu.h | 10 - + drivers/gpu/drm/amd/display/dc/inc/hw/dpp.h | 5 +- + drivers/gpu/drm/amd/display/dc/inc/hw/hubp.h | 6 +- + drivers/gpu/drm/amd/display/dc/inc/hw/hw_shared.h | 10 - + .../gpu/drm/amd/display/dc/inc/hw/link_encoder.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/mpc.h | 65 +- + drivers/gpu/drm/amd/display/dc/inc/hw/opp.h | 14 +- + drivers/gpu/drm/amd/display/dc/inc/hw/pg_cntl.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/hw/rmcm.h | 140 + + .../gpu/drm/amd/display/dc/inc/hw/stream_encoder.h | 2 + + .../drm/amd/display/dc/inc/hw/timing_generator.h | 1 + + drivers/gpu/drm/amd/display/dc/inc/link_service.h | 9 +- + drivers/gpu/drm/amd/display/dc/inc/resource.h | 20 +- + drivers/gpu/drm/amd/display/dc/irq/Makefile | 2 + + .../amd/display/dc/irq/dce110/irq_service_dce110.c | 21 - + .../amd/display/dc/irq/dce110/irq_service_dce110.h | 9 - + drivers/gpu/drm/amd/display/dc/irq/irq_service.c | 28 + + drivers/gpu/drm/amd/display/dc/irq/irq_service.h | 9 + + .../amd/display/dc/link/hwss/link_hwss_hpo_dp.c | 23 +- + .../gpu/drm/amd/display/dc/link/link_detection.c | 4 +- + drivers/gpu/drm/amd/display/dc/link/link_factory.c | 7 +- + .../display/dc/link/protocols/link_dp_capability.c | 3 +- + .../dc/link/protocols/link_dp_panel_replay.c | 36 +- + .../dc/link/protocols/link_dp_panel_replay.h | 2 +- + .../dc/link/protocols/link_edp_panel_control.c | 103 +- + .../dc/link/protocols/link_edp_panel_control.h | 6 + + .../amd/display/dc/link/protocols/link_hdmi_frl.c | 32 +- + .../gpu/drm/amd/display/dc/mpc/dcn10/dcn10_mpc.c | 1 + + .../gpu/drm/amd/display/dc/mpc/dcn20/dcn20_mpc.c | 1 + + .../gpu/drm/amd/display/dc/mpc/dcn30/dcn30_mpc.c | 1 + + .../gpu/drm/amd/display/dc/mpc/dcn32/dcn32_mpc.c | 1 + + .../gpu/drm/amd/display/dc/mpc/dcn401/dcn401_mpc.c | 1 + + .../gpu/drm/amd/display/dc/mpc/dcn42/dcn42_mpc.c | 680 +- + .../gpu/drm/amd/display/dc/mpc/dcn42/dcn42_mpc.h | 714 +- + .../gpu/drm/amd/display/dc/mpc/dcn60/dcn60_mpc.c | 41 +- + .../gpu/drm/amd/display/dc/mpc/dcn60/dcn60_mpc.h | 238 - + .../gpu/drm/amd/display/dc/opp/dcn20/dcn20_opp.c | 2 + + .../gpu/drm/amd/display/dc/optc/dcn31/dcn31_optc.c | 10 + + .../gpu/drm/amd/display/dc/optc/dcn31/dcn31_optc.h | 1 + + .../drm/amd/display/dc/optc/dcn314/dcn314_optc.c | 1 + + .../gpu/drm/amd/display/dc/optc/dcn35/dcn35_optc.c | 1 + + .../gpu/drm/amd/display/dc/optc/dcn42/dcn42_optc.c | 1 + + drivers/gpu/drm/amd/display/dc/os_types.h | 19 + + .../drm/amd/display/dc/pg/dcn35/dcn35_pg_cntl.c | 1 + + .../drm/amd/display/dc/pg/dcn42/dcn42_pg_cntl.c | 5 + + drivers/gpu/drm/amd/display/dc/resource/Makefile | 4 + + .../display/dc/resource/dce112/dce112_resource.c | 62 - + .../amd/display/dc/resource/dcn32/dcn32_resource.c | 2 +- + .../amd/display/dc/resource/dcn35/dcn35_resource.c | 2 +- + .../display/dc/resource/dcn351/dcn351_resource.c | 2 +- + .../amd/display/dc/resource/dcn36/dcn36_resource.c | 2 +- + .../display/dc/resource/dcn401/dcn401_resource.c | 2 +- + .../amd/display/dc/resource/dcn42/dcn42_resource.c | 62 +- + .../display/dc/resource/dcn42b/dcn42b_resource.c | 129 +- + .../amd/display/dc/resource/dcn60/dcn60_resource.c | 64 +- + .../amd/display/dc/resource/dcn60/dcn60_resource.h | 9 +- + drivers/gpu/drm/amd/display/dc/rmcm/Makefile | 44 + + .../gpu/drm/amd/display/dc/rmcm/dcn42/dcn42_rmcm.c | 867 + + .../gpu/drm/amd/display/dc/rmcm/dcn42/dcn42_rmcm.h | 717 + + .../gpu/drm/amd/display/dc/rmcm/dcn60/dcn60_rmcm.c | 251 + + .../{dc_edid_parser.h => rmcm/dcn60/dcn60_rmcm.h} | 23 +- + .../dcn401/dcn401_soc_and_ip_translator.c | 6 +- + .../dcn60/dcn60_soc_and_ip_translator.c | 6 +- + drivers/gpu/drm/amd/display/dc/sspl/Makefile | 5 +- + drivers/gpu/drm/amd/display/dc/sspl/dc_spl.c | 130 +- + .../amd/display/dc/sspl/dc_spl_isharp_filters.c | 10 +- + .../amd/display/dc/sspl/dc_spl_scl_easf_filters.c | 88 +- + .../drm/amd/display/dc/sspl/dc_spl_scl_filters.c | 24 +- + drivers/gpu/drm/amd/display/dc/sspl/dc_spl_types.h | 60 +- + .../gpu/drm/amd/display/dc/sspl/spl_custom_float.c | 152 - + .../gpu/drm/amd/display/dc/sspl/spl_custom_float.h | 29 - + .../gpu/drm/amd/display/dc/sspl/spl_fixpt31_32.c | 495 - + .../gpu/drm/amd/display/dc/sspl/spl_fixpt31_32.h | 526 - + .../gpu/drm/amd/display/dc/sspl/spl_namespace.h | 17 + + drivers/gpu/drm/amd/display/dc/sspl/spl_os_types.h | 43 +- + drivers/gpu/drm/amd/display/dmub/dmub_srv.h | 60 +- + drivers/gpu/drm/amd/display/dmub/inc/dmub_cmd.h | 189 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn20.c | 74 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn20.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn31.c | 82 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn31.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn32.c | 82 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn32.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn35.c | 82 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn35.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn401.c | 86 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn401.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn42.c | 86 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn42.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn60.c | 86 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_dcn60.h | 2 +- + drivers/gpu/drm/amd/display/dmub/src/dmub_srv.c | 63 +- + drivers/gpu/drm/amd/display/include/dal_asic_id.h | 5 + + .../drm/amd/display/modules/color/color_gamma.c | 3 +- + .../drm/amd/display/modules/inc/mod_info_packet.h | 4 + + .../gpu/drm/amd/display/modules/inc/mod_power.h | 4 + + .../amd/display/modules/info_packet/info_packet.c | 109 + + .../drm/amd/display/modules/power/power_replay.c | 13 + + drivers/gpu/drm/amd/include/amd_shared.h | 6 +- + .../drm/amd/include/asic_reg/gc/gc_12_1_0_offset.h | 2 + + .../amd/include/asic_reg/gc/gc_12_1_0_sh_mask.h | 4 +- + .../amd/include/asic_reg/nbio/nbio_7_11_5_offset.h | 11074 ++++ + .../include/asic_reg/nbio/nbio_7_11_5_sh_mask.h | 63248 +++++++++++++++++++ + drivers/gpu/drm/amd/include/discovery.h | 179 +- + .../amd/include/ivsrcid/mpnht/irqsrcs_mpnht_15_0.h | 30 + + drivers/gpu/drm/amd/include/kgd_kfd_interface.h | 4 +- + drivers/gpu/drm/amd/include/kgd_pp_interface.h | 17 + + drivers/gpu/drm/amd/include/v12_structs.h | 2 +- + drivers/gpu/drm/amd/pm/amdgpu_dpm.c | 26 +- + drivers/gpu/drm/amd/pm/amdgpu_pm.c | 56 +- + drivers/gpu/drm/amd/pm/inc/amdgpu_dpm.h | 1 + + drivers/gpu/drm/amd/pm/legacy-dpm/kv_dpm.c | 6 - + drivers/gpu/drm/amd/pm/legacy-dpm/si_dpm.c | 7 - + drivers/gpu/drm/amd/pm/powerplay/amd_powerplay.c | 6 - + .../gpu/drm/amd/pm/powerplay/hwmgr/ppatomctrl.c | 3 + + drivers/gpu/drm/amd/pm/swsmu/amdgpu_smu.c | 49 +- + drivers/gpu/drm/amd/pm/swsmu/inc/amdgpu_smu.h | 23 +- + .../pm/swsmu/inc/pmfw_if/smu15_driver_if_v15_0_0.h | 46 - + .../amd/pm/swsmu/inc/pmfw_if/smu_v13_0_12_ppsmc.h | 1 + + drivers/gpu/drm/amd/pm/swsmu/inc/smu_types.h | 1 + + drivers/gpu/drm/amd/pm/swsmu/inc/smu_v15_0.h | 6 - + drivers/gpu/drm/amd/pm/swsmu/smu11/smu_v11_0.c | 2 - + drivers/gpu/drm/amd/pm/swsmu/smu13/smu_v13_0.c | 6 +- + .../gpu/drm/amd/pm/swsmu/smu13/smu_v13_0_0_ppt.c | 2 +- + .../gpu/drm/amd/pm/swsmu/smu13/smu_v13_0_12_ppt.c | 24 + + .../gpu/drm/amd/pm/swsmu/smu13/smu_v13_0_6_ppt.c | 10 +- + .../gpu/drm/amd/pm/swsmu/smu13/smu_v13_0_6_ppt.h | 1 + + drivers/gpu/drm/amd/pm/swsmu/smu14/smu_v14_0.c | 8 +- + drivers/gpu/drm/amd/pm/swsmu/smu15/smu_v15_0.c | 174 +- + .../gpu/drm/amd/pm/swsmu/smu15/smu_v15_0_0_ppt.c | 270 +- + .../gpu/drm/amd/pm/swsmu/smu15/smu_v15_0_8_ppt.c | 16 +- + drivers/gpu/drm/amd/ras/core/Makefile | 9 +- + drivers/gpu/drm/amd/ras/core/aca.c | 353 +- + drivers/gpu/drm/amd/ras/core/aca.h | 44 +- + drivers/gpu/drm/amd/ras/core/aca_v1_0.c | 41 +- + drivers/gpu/drm/amd/ras/core/aca_v5_0.c | 448 + + drivers/gpu/drm/amd/ras/core/aca_v5_0.h | 62 + + drivers/gpu/drm/amd/ras/core/cmd.c | 147 +- + drivers/gpu/drm/amd/ras/core/cmd.h | 32 +- + drivers/gpu/drm/amd/ras/core/core.c | 309 +- + drivers/gpu/drm/amd/ras/core/eeprom.c | 504 +- + drivers/gpu/drm/amd/ras/core/eeprom.h | 27 +- + drivers/gpu/drm/amd/ras/core/eeprom_fw.c | 640 +- + drivers/gpu/drm/amd/ras/core/eeprom_fw.h | 67 +- + drivers/gpu/drm/amd/ras/core/log_ring.c | 56 +- + drivers/gpu/drm/amd/ras/core/log_ring.h | 46 +- + drivers/gpu/drm/amd/ras/core/ras.h | 138 +- + drivers/gpu/drm/amd/ras/core/ras_bert.c | 546 + + drivers/gpu/drm/amd/ras/core/ras_bert.h | 33 + + drivers/gpu/drm/amd/ras/core/ras_cper.c | 844 +- + drivers/gpu/drm/amd/ras/core/ras_cper.h | 177 +- + drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.c | 418 + + drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.h | 124 + + drivers/gpu/drm/amd/ras/core/ras_gfx.c | 10 +- + drivers/gpu/drm/amd/ras/core/ras_mce.c | 151 + + drivers/gpu/drm/amd/ras/core/ras_mce.h | 51 + + drivers/gpu/drm/amd/ras/core/ras_mp1.c | 169 +- + drivers/gpu/drm/amd/ras/core/ras_mp1.h | 87 +- + drivers/gpu/drm/amd/ras/core/ras_mp1_v13_0.c | 175 +- + drivers/gpu/drm/amd/ras/core/ras_mp1_v15_0.c | 249 + + drivers/gpu/drm/amd/ras/core/ras_mp1_v15_0.h | 30 + + drivers/gpu/drm/amd/ras/core/ras_nbio.c | 9 +- + drivers/gpu/drm/amd/ras/core/ras_process.c | 36 +- + drivers/gpu/drm/amd/ras/core/ras_psp.c | 599 +- + drivers/gpu/drm/amd/ras/core/ras_psp.h | 111 +- + drivers/gpu/drm/amd/ras/core/ras_psp_v13_0.c | 76 + + drivers/gpu/drm/amd/ras/core/ras_psp_v15_0.c | 133 + + drivers/gpu/drm/amd/ras/core/ras_psp_v15_0.h | 31 + + drivers/gpu/drm/amd/ras/core/ras_umc.c | 635 +- + drivers/gpu/drm/amd/ras/core/ras_umc.h | 52 +- + drivers/gpu/drm/amd/ras/core/ras_umc_v12_0.c | 143 +- + drivers/gpu/drm/amd/ras/core/ras_umc_v12_0.h | 3 - + drivers/gpu/drm/amd/ras/core/ras_umc_v15_0.c | 220 + + drivers/gpu/drm/amd/ras/core/ras_umc_v15_0.h | 69 + + drivers/gpu/drm/amd/ras/core/ta_if.h | 35 + + drivers/gpu/drm/amd/ras/ras_mgr/Makefile | 6 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_bert.c | 191 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_bert.h | 30 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_cmd.c | 38 + + .../drm/amd/ras/ras_mgr/amdgpu_ras_eeprom_i2c.c | 141 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.c | 267 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.h | 31 + + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.c | 357 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mgr.h | 8 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1.c | 105 + + .../{amdgpu_ras_mp1_v13_0.h => amdgpu_ras_mp1.h} | 8 +- + .../gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1_v13_0.c | 154 - + .../gpu/drm/amd/ras/ras_mgr/amdgpu_ras_process.c | 9 +- + drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_sys.c | 100 +- + .../gpu/drm/amd/ras/ras_mgr/amdgpu_virt_ras_cmd.c | 42 +- + drivers/gpu/drm/amd/ras/ras_mgr/ras_sys.h | 9 +- + .../gpu/drm/arm/display/include/malidp_product.h | 10 +- + drivers/gpu/drm/arm/display/komeda/komeda_crtc.c | 27 +- + drivers/gpu/drm/arm/display/komeda/komeda_plane.c | 18 +- + .../drm/arm/display/komeda/komeda_wb_connector.c | 33 + + drivers/gpu/drm/arm/hdlcd_crtc.c | 4 +- + drivers/gpu/drm/arm/malidp_crtc.c | 19 +- + drivers/gpu/drm/arm/malidp_planes.c | 18 +- + drivers/gpu/drm/armada/armada_crtc.c | 2 +- + drivers/gpu/drm/armada/armada_overlay.c | 39 +- + drivers/gpu/drm/armada/armada_plane.c | 15 +- + drivers/gpu/drm/armada/armada_plane.h | 2 +- + drivers/gpu/drm/aspeed/aspeed_gfx.h | 11 +- + drivers/gpu/drm/aspeed/aspeed_gfx_crtc.c | 203 +- + drivers/gpu/drm/aspeed/aspeed_gfx_drv.c | 3 +- + drivers/gpu/drm/ast/ast_dp.c | 24 +- + drivers/gpu/drm/ast/ast_mode.c | 20 +- + drivers/gpu/drm/atmel-hlcdc/atmel_hlcdc_crtc.c | 19 +- + drivers/gpu/drm/atmel-hlcdc/atmel_hlcdc_plane.c | 33 +- + drivers/gpu/drm/bridge/Kconfig | 7 +- + drivers/gpu/drm/bridge/analogix/analogix_dp_core.c | 88 +- + drivers/gpu/drm/bridge/analogix/analogix_dp_core.h | 4 +- + drivers/gpu/drm/bridge/analogix/analogix_dp_reg.c | 15 +- + drivers/gpu/drm/bridge/analogix/analogix_dp_reg.h | 4 + + .../gpu/drm/bridge/cadence/cdns-mhdp8546-core.c | 1 - + drivers/gpu/drm/bridge/chipone-icn6211.c | 8 +- + drivers/gpu/drm/bridge/ite-it6505.c | 4 +- + drivers/gpu/drm/bridge/lontium-lt8713sx.c | 2 +- + drivers/gpu/drm/bridge/lontium-lt9211.c | 2 +- + drivers/gpu/drm/bridge/lontium-lt9611.c | 6 +- + drivers/gpu/drm/bridge/lontium-lt9611uxc.c | 2 +- + drivers/gpu/drm/bridge/samsung-dsim.c | 4 +- + drivers/gpu/drm/bridge/sil-sii8620.c | 3 +- + drivers/gpu/drm/bridge/synopsys/dw-dp.c | 4 +- + drivers/gpu/drm/bridge/synopsys/dw-hdmi-qp.c | 209 +- + drivers/gpu/drm/bridge/synopsys/dw-hdmi.c | 2 +- + drivers/gpu/drm/bridge/tc358767.c | 4 +- + drivers/gpu/drm/bridge/ti-sn65dsi83.c | 79 +- + drivers/gpu/drm/clients/drm_fbdev_client.c | 23 +- + drivers/gpu/drm/clients/drm_log.c | 54 +- + drivers/gpu/drm/display/drm_bridge_connector.c | 186 +- + drivers/gpu/drm/display/drm_hdmi_helper.c | 304 + + drivers/gpu/drm/display/drm_hdmi_state_helper.c | 239 +- + drivers/gpu/drm/display/drm_scdc_helper.c | 359 +- + drivers/gpu/drm/drm_atomic.c | 36 +- + drivers/gpu/drm/drm_atomic_helper.c | 10 +- + drivers/gpu/drm/drm_atomic_state_helper.c | 86 - + drivers/gpu/drm/drm_atomic_uapi.c | 7 + + drivers/gpu/drm/drm_bridge.c | 27 +- + drivers/gpu/drm/drm_bridge_helper.c | 2 - + drivers/gpu/drm/drm_buddy.c | 3 +- + drivers/gpu/drm/drm_client_event.c | 18 + + drivers/gpu/drm/drm_colorop.c | 107 + + drivers/gpu/drm/drm_connector.c | 153 +- + drivers/gpu/drm/drm_crtc_internal.h | 2 - + drivers/gpu/drm/drm_debugfs.c | 157 - + drivers/gpu/drm/drm_drv.c | 34 +- + drivers/gpu/drm/drm_edid.c | 237 +- + drivers/gpu/drm/drm_gem.c | 19 +- + drivers/gpu/drm/drm_gem_atomic_helper.c | 100 +- + drivers/gpu/drm/drm_gpusvm.c | 410 +- + drivers/gpu/drm/drm_kms_helper_common.c | 14 + + drivers/gpu/drm/drm_mode_config.c | 21 +- + drivers/gpu/drm/drm_modeset_helper.c | 2 + + drivers/gpu/drm/drm_of.c | 38 +- + drivers/gpu/drm/drm_pagemap.c | 5 +- + drivers/gpu/drm/drm_panic.c | 927 +- + drivers/gpu/drm/drm_panic_helper.c | 919 + + .../{drm_panic_qr.rs => drm_panic_helper_qr.rs} | 4 +- + drivers/gpu/drm/drm_panic_internal.h | 69 + + drivers/gpu/drm/drm_probe_helper.c | 15 +- + drivers/gpu/drm/drm_ras.c | 278 +- + drivers/gpu/drm/drm_ras_nl.c | 33 + + drivers/gpu/drm/drm_ras_nl.h | 8 + + drivers/gpu/drm/drm_simple_kms_helper.c | 37 +- + drivers/gpu/drm/drm_sysfs.c | 32 - + drivers/gpu/drm/exynos/exynos_drm_crtc.c | 2 +- + drivers/gpu/drm/exynos/exynos_drm_dpi.c | 3 +- + drivers/gpu/drm/exynos/exynos_drm_plane.c | 22 +- + drivers/gpu/drm/exynos/exynos_drm_vidi.c | 3 +- + drivers/gpu/drm/exynos/exynos_hdmi.c | 3 +- + drivers/gpu/drm/fsl-dcu/fsl_dcu_drm_crtc.c | 2 +- + drivers/gpu/drm/fsl-dcu/fsl_dcu_drm_plane.c | 2 +- + drivers/gpu/drm/fsl-dcu/fsl_dcu_drm_rgb.c | 10 +- + drivers/gpu/drm/gma500/psb_intel_display.c | 34 +- + drivers/gpu/drm/gud/gud_drv.c | 2 +- + drivers/gpu/drm/hisilicon/hibmc/hibmc_drm_de.c | 18 +- + drivers/gpu/drm/hisilicon/hibmc/hibmc_drm_drv.c | 5 +- + drivers/gpu/drm/hisilicon/kirin/dw_drm_dsi.c | 9 +- + drivers/gpu/drm/hisilicon/kirin/kirin_drm_ade.c | 4 +- + drivers/gpu/drm/hyperv/hyperv_drm.h | 1 - + drivers/gpu/drm/hyperv/hyperv_drm_modeset.c | 4 +- + drivers/gpu/drm/hyperv/hyperv_drm_proto.c | 44 +- + drivers/gpu/drm/i915/display/i9xx_plane.c | 3 + + drivers/gpu/drm/i915/display/intel_atomic.c | 1 + + drivers/gpu/drm/i915/display/intel_audio.c | 153 + + drivers/gpu/drm/i915/display/intel_audio_regs.h | 16 +- + drivers/gpu/drm/i915/display/intel_cdclk.c | 326 +- + drivers/gpu/drm/i915/display/intel_cmtg.c | 2 +- + .../gpu/drm/i915/display/intel_crtc_state_dump.c | 3 + + drivers/gpu/drm/i915/display/intel_cursor.c | 296 +- + drivers/gpu/drm/i915/display/intel_cursor_regs.h | 8 +- + drivers/gpu/drm/i915/display/intel_ddi.c | 36 +- + drivers/gpu/drm/i915/display/intel_display.c | 76 +- + drivers/gpu/drm/i915/display/intel_display.h | 3 +- + .../drm/i915/display/intel_display_clock_gating.c | 67 +- + .../drm/i915/display/intel_display_clock_gating.h | 16 +- + .../gpu/drm/i915/display/intel_display_debugfs.c | 21 +- + .../drm/i915/display/intel_display_power_well.c | 6 +- + drivers/gpu/drm/i915/display/intel_display_regs.h | 18 +- + drivers/gpu/drm/i915/display/intel_display_types.h | 17 +- + drivers/gpu/drm/i915/display/intel_display_wa.c | 4 +- + drivers/gpu/drm/i915/display/intel_display_wa.h | 9 - + drivers/gpu/drm/i915/display/intel_dp.c | 52 +- + drivers/gpu/drm/i915/display/intel_dp_mst.c | 4 + + drivers/gpu/drm/i915/display/intel_dpll.c | 22 +- + drivers/gpu/drm/i915/display/intel_dpll_mgr.c | 221 +- + drivers/gpu/drm/i915/display/intel_dpll_mgr.h | 22 + + drivers/gpu/drm/i915/display/intel_frontbuffer.c | 3 +- + drivers/gpu/drm/i915/display/intel_hdmi.c | 53 +- + drivers/gpu/drm/i915/display/intel_hdmi.h | 11 +- + drivers/gpu/drm/i915/display/intel_hti.c | 3 - + .../gpu/drm/i915/display/intel_modeset_verify.c | 1 - + drivers/gpu/drm/i915/display/intel_parent.c | 10 +- + drivers/gpu/drm/i915/display/intel_parent.h | 3 +- + drivers/gpu/drm/i915/display/intel_psr.c | 18 +- + drivers/gpu/drm/i915/display/intel_snps_phy.c | 60 +- + drivers/gpu/drm/i915/display/intel_snps_phy.h | 2 + + drivers/gpu/drm/i915/display/intel_tdf.h | 25 - + drivers/gpu/drm/i915/display/intel_vrr.c | 395 +- + drivers/gpu/drm/i915/display/intel_vrr.h | 2 + + drivers/gpu/drm/i915/display/skl_universal_plane.c | 4 + + drivers/gpu/drm/i915/gvt/handlers.c | 2 +- + drivers/gpu/drm/i915/i915_dpt.c | 1 - + drivers/gpu/drm/i915/i915_reg.h | 2 +- + drivers/gpu/drm/i915/i915_switcheroo.c | 11 +- + drivers/gpu/drm/i915/intel_clock_gating.c | 30 +- + drivers/gpu/drm/i915/intel_gvt_mmio_table.c | 2 +- + drivers/gpu/drm/i915/intel_pcode.c | 14 +- + drivers/gpu/drm/i915/intel_pcode.h | 2 +- + drivers/gpu/drm/imagination/pvr_ccb.c | 43 +- + drivers/gpu/drm/imagination/pvr_device.c | 25 +- + drivers/gpu/drm/imagination/pvr_fw.c | 23 +- + drivers/gpu/drm/imagination/pvr_fw.h | 2 +- + drivers/gpu/drm/imagination/pvr_fw_meta.c | 18 +- + drivers/gpu/drm/imagination/pvr_fw_riscv.c | 40 +- + drivers/gpu/drm/imagination/pvr_fw_startstop.c | 22 +- + drivers/gpu/drm/imagination/pvr_fw_trace.c | 1 - + drivers/gpu/drm/imagination/pvr_rogue_cr_defs.h | 920 +- + drivers/gpu/drm/imagination/pvr_rogue_defs.h | 22 +- + drivers/gpu/drm/imagination/pvr_rogue_riscv.h | 2 - + drivers/gpu/drm/imagination/pvr_trace.h | 14 +- + drivers/gpu/drm/imx/dc/dc-crtc.c | 2 +- + drivers/gpu/drm/imx/dc/dc-kms.c | 8 +- + drivers/gpu/drm/imx/dc/dc-plane.c | 2 +- + drivers/gpu/drm/imx/dcss/dcss-crtc.c | 2 +- + drivers/gpu/drm/imx/dcss/dcss-plane.c | 2 +- + drivers/gpu/drm/imx/ipuv3/ipuv3-crtc.c | 18 +- + drivers/gpu/drm/imx/ipuv3/ipuv3-plane.c | 21 +- + drivers/gpu/drm/ingenic/ingenic-drm-drv.c | 4 +- + drivers/gpu/drm/ingenic/ingenic-ipu.c | 2 +- + drivers/gpu/drm/kmb/kmb_crtc.c | 2 +- + drivers/gpu/drm/kmb/kmb_dsi.c | 9 +- + drivers/gpu/drm/kmb/kmb_plane.c | 2 +- + drivers/gpu/drm/logicvc/logicvc_crtc.c | 2 +- + drivers/gpu/drm/logicvc/logicvc_layer.c | 2 +- + drivers/gpu/drm/loongson/lsdc_crtc.c | 30 +- + drivers/gpu/drm/loongson/lsdc_plane.c | 2 +- + drivers/gpu/drm/mcde/mcde_display.c | 272 +- + drivers/gpu/drm/mcde/mcde_drm.h | 12 +- + drivers/gpu/drm/mcde/mcde_drv.c | 3 +- + drivers/gpu/drm/mcde/mcde_dsi.c | 8 +- + drivers/gpu/drm/mediatek/mtk_crtc.c | 18 +- + drivers/gpu/drm/mediatek/mtk_dsi.c | 10 +- + drivers/gpu/drm/mediatek/mtk_plane.c | 21 +- + drivers/gpu/drm/meson/meson_crtc.c | 2 +- + drivers/gpu/drm/meson/meson_encoder_cvbs.c | 11 +- + drivers/gpu/drm/meson/meson_encoder_dsi.c | 11 +- + drivers/gpu/drm/meson/meson_encoder_hdmi.c | 11 +- + drivers/gpu/drm/meson/meson_overlay.c | 2 +- + drivers/gpu/drm/meson/meson_plane.c | 2 +- + drivers/gpu/drm/mgag200/mgag200_drv.h | 8 +- + drivers/gpu/drm/mgag200/mgag200_mode.c | 15 +- + drivers/gpu/drm/msm/disp/dpu1/dpu_crtc.c | 18 +- + drivers/gpu/drm/msm/disp/dpu1/dpu_plane.c | 18 +- + drivers/gpu/drm/msm/disp/mdp4/mdp4_crtc.c | 2 +- + drivers/gpu/drm/msm/disp/mdp4/mdp4_plane.c | 2 +- + drivers/gpu/drm/msm/disp/mdp5/mdp5_crtc.c | 20 +- + drivers/gpu/drm/msm/disp/mdp5/mdp5_plane.c | 16 +- + drivers/gpu/drm/msm/msm_gem_shrinker.c | 22 +- + drivers/gpu/drm/mxsfb/lcdif_kms.c | 19 +- + drivers/gpu/drm/mxsfb/mxsfb_kms.c | 6 +- + drivers/gpu/drm/nouveau/dispnv04/dfp.c | 5 +- + drivers/gpu/drm/nouveau/dispnv50/disp.c | 4 +- + drivers/gpu/drm/nouveau/dispnv50/head.c | 14 +- + drivers/gpu/drm/nouveau/dispnv50/headca7d.c | 21 +- + drivers/gpu/drm/nouveau/dispnv50/wndw.c | 17 +- + .../gpu/drm/nouveau/include/nvhw/class/clca7d.h | 4 + + drivers/gpu/drm/nouveau/include/nvkm/subdev/pci.h | 1 + + drivers/gpu/drm/nouveau/nouveau_abi16.c | 4 + + drivers/gpu/drm/nouveau/nouveau_acpi.c | 32 +- + drivers/gpu/drm/nouveau/nouveau_acpi.h | 10 +- + drivers/gpu/drm/nouveau/nouveau_bios.c | 2 +- + drivers/gpu/drm/nouveau/nouveau_connector.c | 152 +- + drivers/gpu/drm/nouveau/nouveau_connector.h | 12 +- + drivers/gpu/drm/nouveau/nouveau_dp.c | 6 +- + drivers/gpu/drm/nouveau/nouveau_drm.c | 114 +- + drivers/gpu/drm/nouveau/nouveau_exec.c | 2 +- + drivers/gpu/drm/nouveau/nouveau_fence.c | 2 +- + drivers/gpu/drm/nouveau/nouveau_led.c | 2 +- + drivers/gpu/drm/nouveau/nouveau_svm.c | 13 + + drivers/gpu/drm/nouveau/nouveau_vga.c | 28 +- + drivers/gpu/drm/nouveau/nvkm/engine/device/base.c | 2 +- + drivers/gpu/drm/nouveau/nvkm/subdev/bios/init.c | 2 +- + drivers/gpu/drm/nouveau/nvkm/subdev/clk/gk20a.h | 4 +- + drivers/gpu/drm/nouveau/nvkm/subdev/pci/Kbuild | 1 + + drivers/gpu/drm/nouveau/nvkm/subdev/pci/mcp79.c | 35 + + drivers/gpu/drm/omapdrm/dss/dsi.c | 361 +- + drivers/gpu/drm/omapdrm/dss/hdmi.h | 6 + + drivers/gpu/drm/omapdrm/dss/hdmi4.c | 29 +- + drivers/gpu/drm/omapdrm/dss/hdmi5.c | 37 +- + drivers/gpu/drm/omapdrm/dss/hdmi_common.c | 14 + + drivers/gpu/drm/omapdrm/omap_crtc.c | 17 +- + drivers/gpu/drm/omapdrm/omap_drv.c | 1 - + drivers/gpu/drm/omapdrm/omap_plane.c | 13 +- + drivers/gpu/drm/panel/Kconfig | 25 + + drivers/gpu/drm/panel/Makefile | 2 + + drivers/gpu/drm/panel/panel-edp.c | 5 + + drivers/gpu/drm/panel/panel-himax-hx83102.c | 24 +- + drivers/gpu/drm/panel/panel-himax-hx83112a.c | 26 +- + drivers/gpu/drm/panel/panel-himax-hx83112b.c | 23 +- + drivers/gpu/drm/panel/panel-himax-hx8394.c | 26 +- + drivers/gpu/drm/panel/panel-ilitek-ili7836a.c | 297 + + drivers/gpu/drm/panel/panel-ilitek-ili9805.c | 21 +- + drivers/gpu/drm/panel/panel-ilitek-ili9806e-core.c | 12 +- + drivers/gpu/drm/panel/panel-ilitek-ili9806e-core.h | 1 - + drivers/gpu/drm/panel/panel-ilitek-ili9806e-dsi.c | 16 +- + drivers/gpu/drm/panel/panel-ilitek-ili9806e-spi.c | 6 - + drivers/gpu/drm/panel/panel-ilitek-ili9881c.c | 15 +- + drivers/gpu/drm/panel/panel-ilitek-ili9882t.c | 24 +- + drivers/gpu/drm/panel/panel-jdi-fhd-r63452.c | 19 +- + drivers/gpu/drm/panel/panel-jdi-lt070me05000.c | 32 +- + drivers/gpu/drm/panel/panel-leadtek-ltk050h3146w.c | 20 +- + drivers/gpu/drm/panel/panel-leadtek-ltk500hd1829.c | 20 +- + drivers/gpu/drm/panel/panel-novatek-nt36532.c | 431 + + drivers/gpu/drm/panel/panel-samsung-dsi.h | 38 + + drivers/gpu/drm/panel/panel-samsung-s6d16d0.c | 74 +- + drivers/gpu/drm/panel/panel-samsung-s6d7aa0.c | 20 +- + drivers/gpu/drm/panel/panel-samsung-s6e3fa7.c | 20 +- + drivers/gpu/drm/panel/panel-samsung-s6e3fc2x01.c | 91 +- + drivers/gpu/drm/panel/panel-samsung-s6e3ha2.c | 30 +- + drivers/gpu/drm/panel/panel-samsung-s6e3ha8.c | 90 +- + drivers/gpu/drm/panel/panel-samsung-s6e63j0x03.c | 31 +- + drivers/gpu/drm/panel/panel-samsung-s6e63m0-dsi.c | 13 +- + drivers/gpu/drm/panel/panel-samsung-s6e63m0-spi.c | 6 - + drivers/gpu/drm/panel/panel-samsung-s6e63m0.c | 12 +- + drivers/gpu/drm/panel/panel-samsung-s6e63m0.h | 1 - + .../drm/panel/panel-samsung-s6e88a0-ams427ap24.c | 20 +- + .../drm/panel/panel-samsung-s6e88a0-ams452ef01.c | 26 +- + drivers/gpu/drm/panel/panel-samsung-s6e8aa0.c | 19 +- + .../gpu/drm/panel/panel-samsung-s6e8fc0-m1906f9.c | 46 +- + drivers/gpu/drm/panel/panel-samsung-sofef00.c | 35 +- + drivers/gpu/drm/panel/panel-sharp-ls043t1le01.c | 31 +- + drivers/gpu/drm/panel/panel-sharp-ls060t1sx01.c | 20 +- + drivers/gpu/drm/panel/panel-sony-td4353-jdi.c | 20 +- + .../gpu/drm/panel/panel-sony-tulip-truly-nt35521.c | 20 +- + drivers/gpu/drm/panel/panel-visionox-r66451.c | 23 +- + drivers/gpu/drm/panel/panel-visionox-rm69299.c | 21 +- + drivers/gpu/drm/panfrost/panfrost_devfreq.c | 2 +- + drivers/gpu/drm/panfrost/panfrost_device.c | 1 - + drivers/gpu/drm/panfrost/panfrost_drv.c | 2 +- + drivers/gpu/drm/panthor/panthor_device.h | 286 +- + drivers/gpu/drm/panthor/panthor_drv.c | 1 + + drivers/gpu/drm/panthor/panthor_fw.c | 22 +- + drivers/gpu/drm/panthor/panthor_gem.c | 14 +- + drivers/gpu/drm/panthor/panthor_gpu.c | 117 +- + drivers/gpu/drm/panthor/panthor_mmu.c | 44 +- + drivers/gpu/drm/panthor/panthor_mmu.h | 3 +- + drivers/gpu/drm/panthor/panthor_pwr.c | 24 +- + drivers/gpu/drm/panthor/panthor_sched.c | 591 +- + drivers/gpu/drm/panthor/panthor_sched.h | 5 + + drivers/gpu/drm/panthor/panthor_trace.h | 38 + + drivers/gpu/drm/pl111/pl111_display.c | 207 +- + drivers/gpu/drm/pl111/pl111_drm.h | 5 +- + drivers/gpu/drm/pl111/pl111_drv.c | 19 +- + drivers/gpu/drm/pl111/pl111_versatile.c | 18 - + drivers/gpu/drm/qxl/qxl_display.c | 6 +- + drivers/gpu/drm/radeon/atombios_encoders.c | 36 + + drivers/gpu/drm/radeon/radeon_device.c | 9 +- + drivers/gpu/drm/radeon/radeon_fence.c | 4 +- + drivers/gpu/drm/renesas/rcar-du/Kconfig | 12 + + drivers/gpu/drm/renesas/rcar-du/Makefile | 1 + + drivers/gpu/drm/renesas/rcar-du/rcar_dsc.c | 153 + + drivers/gpu/drm/renesas/rcar-du/rcar_du_crtc.c | 17 +- + drivers/gpu/drm/renesas/rcar-du/rcar_du_encoder.c | 17 +- + drivers/gpu/drm/renesas/rcar-du/rcar_du_plane.c | 19 +- + drivers/gpu/drm/renesas/rcar-du/rcar_du_vsp.c | 17 +- + drivers/gpu/drm/renesas/rcar-du/rcar_mipi_dsi.c | 51 +- + drivers/gpu/drm/renesas/rz-du/rzg2l_du_crtc.c | 15 +- + drivers/gpu/drm/renesas/rz-du/rzg2l_du_vsp.c | 15 +- + drivers/gpu/drm/renesas/shmobile/shmob_drm_crtc.c | 12 +- + drivers/gpu/drm/renesas/shmobile/shmob_drm_plane.c | 19 +- + drivers/gpu/drm/rockchip/dw-mipi-dsi-rockchip.c | 66 +- + drivers/gpu/drm/rockchip/rockchip_drm_vop.c | 20 +- + drivers/gpu/drm/rockchip/rockchip_drm_vop2.c | 20 +- + drivers/gpu/drm/scheduler/sched_entity.c | 10 +- + drivers/gpu/drm/scheduler/sched_fence.c | 3 +- + drivers/gpu/drm/scheduler/sched_internal.h | 3 +- + drivers/gpu/drm/scheduler/sched_main.c | 8 +- + drivers/gpu/drm/scheduler/tests/tests_basic.c | 2 +- + drivers/gpu/drm/sitronix/st7571.c | 2 +- + drivers/gpu/drm/sitronix/st7920.c | 24 +- + drivers/gpu/drm/solomon/ssd130x-spi.c | 25 +- + drivers/gpu/drm/solomon/ssd130x.c | 443 +- + drivers/gpu/drm/solomon/ssd130x.h | 10 +- + drivers/gpu/drm/sprd/sprd_dpu.c | 4 +- + drivers/gpu/drm/sti/sti_crtc.c | 2 +- + drivers/gpu/drm/sti/sti_cursor.c | 2 +- + drivers/gpu/drm/sti/sti_gdp.c | 2 +- + drivers/gpu/drm/sti/sti_hqvdp.c | 2 +- + drivers/gpu/drm/stm/ltdc.c | 6 +- + drivers/gpu/drm/sun4i/sun4i_crtc.c | 2 +- + drivers/gpu/drm/sun4i/sun4i_hdmi_enc.c | 3 +- + drivers/gpu/drm/sun4i/sun4i_layer.c | 21 +- + drivers/gpu/drm/sun4i/sun8i_ui_layer.c | 2 +- + drivers/gpu/drm/sun4i/sun8i_vi_layer.c | 2 +- + drivers/gpu/drm/sysfb/drm_sysfb_helper.h | 12 +- + drivers/gpu/drm/sysfb/drm_sysfb_modeset.c | 70 +- + drivers/gpu/drm/sysfb/efidrm.c | 17 +- + drivers/gpu/drm/sysfb/ofdrm.c | 18 +- + drivers/gpu/drm/sysfb/vesadrm.c | 18 +- + drivers/gpu/drm/tegra/dc.c | 18 +- + drivers/gpu/drm/tegra/dsi.c | 16 +- + drivers/gpu/drm/tegra/plane.c | 28 +- + drivers/gpu/drm/tegra/rgb.c | 15 +- + drivers/gpu/drm/tests/Makefile | 1 + + drivers/gpu/drm/tests/drm_connector_test.c | 42 +- + drivers/gpu/drm/tests/drm_hdmi_state_helper_test.c | 56 +- + drivers/gpu/drm/tests/drm_kunit_helpers.c | 4 +- + .../{drm_panic_test.c => drm_panic_helper_test.c} | 75 +- + drivers/gpu/drm/tidss/tidss_dispc.c | 38 +- + drivers/gpu/drm/tidss/tidss_dispc.h | 2 +- + drivers/gpu/drm/tidss/tidss_encoder.c | 10 +- + drivers/gpu/drm/tidss/tidss_plane.c | 2 + + drivers/gpu/drm/tilcdc/tilcdc_crtc.c | 7 +- + drivers/gpu/drm/tilcdc/tilcdc_plane.c | 2 +- + drivers/gpu/drm/tiny/appletbdrm.c | 14 +- + drivers/gpu/drm/tiny/arcpgu.c | 201 +- + drivers/gpu/drm/tiny/bochs.c | 6 +- + drivers/gpu/drm/tiny/cirrus-qemu.c | 2 +- + drivers/gpu/drm/tiny/gm12u320.c | 138 +- + drivers/gpu/drm/tiny/pixpaper.c | 2 +- + drivers/gpu/drm/tiny/repaper.c | 138 +- + drivers/gpu/drm/tiny/sharp-memory.c | 2 +- + drivers/gpu/drm/ttm/ttm_bo.c | 3 +- + drivers/gpu/drm/ttm/ttm_module.c | 1 - + drivers/gpu/drm/tve200/tve200_display.c | 220 +- + drivers/gpu/drm/tve200/tve200_drm.h | 6 +- + drivers/gpu/drm/tve200/tve200_drv.c | 12 +- + drivers/gpu/drm/udl/udl_modeset.c | 2 +- + drivers/gpu/drm/v3d/v3d_drv.h | 22 +- + drivers/gpu/drm/v3d/v3d_gem.c | 3 - + drivers/gpu/drm/v3d/v3d_perfmon.c | 13 +- + drivers/gpu/drm/v3d/v3d_sched.c | 4 - + drivers/gpu/drm/v3d/v3d_submit.c | 19 +- + drivers/gpu/drm/vboxvideo/vbox_mode.c | 4 +- + drivers/gpu/drm/vc4/tests/vc4_mock_crtc.c | 2 +- + drivers/gpu/drm/vc4/vc4_crtc.c | 14 +- + drivers/gpu/drm/vc4/vc4_drv.c | 1 - + drivers/gpu/drm/vc4/vc4_drv.h | 15 +- + drivers/gpu/drm/vc4/vc4_gem.c | 40 +- + drivers/gpu/drm/vc4/vc4_hdmi.c | 5 +- + drivers/gpu/drm/vc4/vc4_irq.c | 7 +- + drivers/gpu/drm/vc4/vc4_plane.c | 15 +- + drivers/gpu/drm/vc4/vc4_txp.c | 2 +- + drivers/gpu/drm/vc4/vc4_v3d.c | 37 +- + drivers/gpu/drm/verisilicon/vs_crtc.c | 2 +- + drivers/gpu/drm/verisilicon/vs_cursor_plane.c | 10 +- + drivers/gpu/drm/verisilicon/vs_plane.c | 34 +- + drivers/gpu/drm/verisilicon/vs_plane.h | 4 +- + drivers/gpu/drm/verisilicon/vs_primary_plane.c | 9 +- + drivers/gpu/drm/virtio/virtgpu_display.c | 14 +- + drivers/gpu/drm/virtio/virtgpu_plane.c | 14 +- + drivers/gpu/drm/vkms/tests/gen_yuv_conversion.py | 87 + + drivers/gpu/drm/vkms/tests/vkms_format_test.c | 40 +- + drivers/gpu/drm/vkms/vkms_colorop.c | 66 +- + drivers/gpu/drm/vkms/vkms_composer.c | 36 +- + drivers/gpu/drm/vkms/vkms_crtc.c | 18 +- + drivers/gpu/drm/vkms/vkms_drv.h | 2 +- + drivers/gpu/drm/vkms/vkms_formats.c | 112 +- + drivers/gpu/drm/vkms/vkms_formats.h | 2 +- + drivers/gpu/drm/vkms/vkms_plane.c | 91 +- + drivers/gpu/drm/vmwgfx/vmwgfx_kms.c | 39 +- + drivers/gpu/drm/vmwgfx/vmwgfx_kms.h | 4 +- + drivers/gpu/drm/vmwgfx/vmwgfx_ldu.c | 6 +- + drivers/gpu/drm/vmwgfx/vmwgfx_scrn.c | 6 +- + drivers/gpu/drm/vmwgfx/vmwgfx_stdu.c | 6 +- + drivers/gpu/drm/xe/Makefile | 4 +- + drivers/gpu/drm/xe/abi/guc_actions_slpc_abi.h | 1 + + drivers/gpu/drm/xe/abi/xe_log_abi.h | 200 + + drivers/gpu/drm/xe/abi/xe_sigid_abi.h | 172 + + drivers/gpu/drm/xe/display/xe_display.c | 22 +- + drivers/gpu/drm/xe/display/xe_display_pcode.c | 4 +- + drivers/gpu/drm/xe/display/xe_display_rpm.c | 2 - + drivers/gpu/drm/xe/display/xe_display_wa.c | 13 +- + drivers/gpu/drm/xe/display/xe_display_wa.h | 9 + + drivers/gpu/drm/xe/display/xe_dsb_buffer.c | 2 +- + drivers/gpu/drm/xe/display/xe_fb_pin.c | 2 +- + drivers/gpu/drm/xe/display/xe_panic.c | 3 +- + drivers/gpu/drm/xe/display/xe_tdf.c | 15 - + drivers/gpu/drm/xe/instructions/xe_mi_commands.h | 7 +- + drivers/gpu/drm/xe/regs/xe_gt_regs.h | 33 + + drivers/gpu/drm/xe/regs/xe_lrc_layout.h | 2 + + drivers/gpu/drm/xe/tests/Makefile | 1 + + drivers/gpu/drm/xe/tests/xe_any_kunit.c | 213 + + .../gpu/drm/xe/tests/xe_guc_klv_helpers_kunit.c | 303 + + drivers/gpu/drm/xe/tests/xe_kunit_helpers.c | 4 + + drivers/gpu/drm/xe/tests/xe_log_kunit.c | 553 + + drivers/gpu/drm/xe/tests/xe_pci.c | 25 +- + drivers/gpu/drm/xe/xe_any.h | 137 + + drivers/gpu/drm/xe/xe_bo.c | 76 +- + drivers/gpu/drm/xe/xe_bo.h | 3 +- + drivers/gpu/drm/xe/xe_bo_types.h | 8 + + drivers/gpu/drm/xe/xe_configfs.c | 72 +- + drivers/gpu/drm/xe/xe_configfs.h | 2 + + drivers/gpu/drm/xe/xe_debugfs.c | 66 + + drivers/gpu/drm/xe/xe_debugfs.h | 4 + + drivers/gpu/drm/xe/xe_defaults.h | 1 + + drivers/gpu/drm/xe/xe_devcoredump.c | 4 +- + drivers/gpu/drm/xe/xe_device.c | 150 +- + drivers/gpu/drm/xe/xe_device.h | 2 +- + drivers/gpu/drm/xe/xe_device_types.h | 42 +- + drivers/gpu/drm/xe/xe_dma_buf.c | 22 +- + drivers/gpu/drm/xe/xe_drm_ras.c | 65 + + drivers/gpu/drm/xe/xe_drm_ras.h | 3 + + drivers/gpu/drm/xe/xe_drm_ras_types.h | 3 + + drivers/gpu/drm/xe/xe_exec_queue.c | 78 +- + drivers/gpu/drm/xe/xe_exec_queue_types.h | 28 +- + drivers/gpu/drm/xe/xe_execlist.c | 4 +- + drivers/gpu/drm/xe/xe_ggtt.c | 74 +- + drivers/gpu/drm/xe/xe_gsc.c | 3 +- + drivers/gpu/drm/xe/xe_gt.c | 24 +- + drivers/gpu/drm/xe/xe_gt.h | 13 + + drivers/gpu/drm/xe/xe_gt_debugfs.c | 252 +- + drivers/gpu/drm/xe/xe_gt_idle.c | 84 +- + drivers/gpu/drm/xe/xe_gt_idle.h | 1 + + drivers/gpu/drm/xe/xe_gt_printk.h | 3 + + drivers/gpu/drm/xe/xe_gt_sriov_pf_config.c | 299 +- + drivers/gpu/drm/xe/xe_gt_sriov_pf_config.h | 10 + + drivers/gpu/drm/xe/xe_gt_sriov_printk.h | 3 + + drivers/gpu/drm/xe/xe_gt_stats.c | 7 + + drivers/gpu/drm/xe/xe_gt_stats_types.h | 22 + + drivers/gpu/drm/xe/xe_gt_types.h | 29 + + drivers/gpu/drm/xe/xe_guc.c | 18 +- + drivers/gpu/drm/xe/xe_guc_capture.c | 2 - + drivers/gpu/drm/xe/xe_guc_ct.c | 95 +- + drivers/gpu/drm/xe/xe_guc_ct.h | 38 +- + drivers/gpu/drm/xe/xe_guc_exec_queue_types.h | 32 +- + drivers/gpu/drm/xe/xe_guc_klv_helpers.c | 78 + + drivers/gpu/drm/xe/xe_guc_klv_helpers.h | 5 + + drivers/gpu/drm/xe/xe_guc_pagefault.c | 47 +- + drivers/gpu/drm/xe/xe_guc_pc.c | 15 +- + drivers/gpu/drm/xe/xe_guc_submit.c | 382 +- + drivers/gpu/drm/xe/xe_guc_submit.h | 5 + + drivers/gpu/drm/xe/xe_guc_tlb_inval.c | 29 + + drivers/gpu/drm/xe/xe_guc_types.h | 6 + + drivers/gpu/drm/xe/xe_hwmon.c | 95 +- + drivers/gpu/drm/xe/xe_i2c.c | 22 +- + drivers/gpu/drm/xe/xe_i2c.h | 1 + + drivers/gpu/drm/xe/xe_log.c | 247 + + drivers/gpu/drm/xe/xe_log.h | 194 + + drivers/gpu/drm/xe/xe_lrc.c | 39 +- + drivers/gpu/drm/xe/xe_lrc.h | 1 + + drivers/gpu/drm/xe/xe_mert.c | 2 +- + drivers/gpu/drm/xe/xe_migrate.c | 123 +- + drivers/gpu/drm/xe/xe_migrate.h | 6 + + drivers/gpu/drm/xe/xe_mmio_gem.c | 7 +- + drivers/gpu/drm/xe/xe_module.c | 4 + + drivers/gpu/drm/xe/xe_module.h | 1 + + drivers/gpu/drm/xe/xe_nvm.c | 1 + + drivers/gpu/drm/xe/xe_oa.c | 7 +- + drivers/gpu/drm/xe/xe_pagefault.c | 826 +- + drivers/gpu/drm/xe/xe_pagefault.h | 137 + + drivers/gpu/drm/xe/xe_pagefault_types.h | 201 +- + drivers/gpu/drm/xe/xe_pci.c | 74 +- + drivers/gpu/drm/xe/xe_pci_error.c | 13 +- + drivers/gpu/drm/xe/xe_pci_sriov.c | 27 +- + drivers/gpu/drm/xe/xe_pcode.c | 27 +- + drivers/gpu/drm/xe/xe_pcode.h | 6 +- + drivers/gpu/drm/xe/xe_pm.c | 23 + + drivers/gpu/drm/xe/xe_pm.h | 1 + + drivers/gpu/drm/xe/xe_printk.h | 3 + + drivers/gpu/drm/xe/xe_pt.c | 34 +- + drivers/gpu/drm/xe/xe_pxp_submit.c | 5 +- + drivers/gpu/drm/xe/xe_ras.c | 245 +- + drivers/gpu/drm/xe/xe_ras.h | 2 + + drivers/gpu/drm/xe/xe_ras_types.h | 50 + + drivers/gpu/drm/xe/xe_res_cursor.h | 5 +- + drivers/gpu/drm/xe/xe_ring_ops.c | 4 +- + drivers/gpu/drm/xe/xe_sriov_packet.c | 31 +- + drivers/gpu/drm/xe/xe_sriov_printk.h | 3 + + drivers/gpu/drm/xe/xe_sriov_vf_ccs.c | 1 - + drivers/gpu/drm/xe/xe_survivability_mode.c | 85 +- + drivers/gpu/drm/xe/xe_svm.c | 182 +- + drivers/gpu/drm/xe/xe_svm.h | 79 +- + drivers/gpu/drm/xe/xe_sysctrl.c | 92 + + drivers/gpu/drm/xe/xe_sysctrl.h | 2 + + drivers/gpu/drm/xe/xe_sysctrl_event.c | 28 +- + drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h | 49 + + drivers/gpu/drm/xe/xe_tile_printk.h | 3 + + drivers/gpu/drm/xe/xe_tile_sriov_printk.h | 3 + + drivers/gpu/drm/xe/xe_tile_types.h | 4 + + drivers/gpu/drm/xe/xe_tlb_inval.c | 42 +- + drivers/gpu/drm/xe/xe_tlb_inval.h | 1 + + drivers/gpu/drm/xe/xe_tlb_inval_types.h | 10 + + drivers/gpu/drm/xe/xe_ttm_vram_mgr.c | 600 +- + drivers/gpu/drm/xe/xe_ttm_vram_mgr.h | 3 + + drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h | 40 + + drivers/gpu/drm/xe/xe_userptr.c | 21 +- + drivers/gpu/drm/xe/xe_vm.c | 316 +- + drivers/gpu/drm/xe/xe_vm_types.h | 44 +- + drivers/gpu/drm/xe/xe_vram.c | 255 +- + drivers/gpu/drm/xe/xe_vram.h | 10 + + drivers/gpu/drm/xe/xe_wa.c | 4 +- + drivers/gpu/drm/xen/xen_drm_front.h | 6 +- + drivers/gpu/drm/xen/xen_drm_front_kms.c | 188 +- + drivers/gpu/drm/xlnx/zynqmp_kms.c | 18 +- + drivers/gpu/tests/gpu_buddy_test.c | 160 +- + drivers/gpu/vga/vga_switcheroo.c | 41 +- + drivers/video/fbdev/core/fbcon.c | 8 - + include/drm/display/drm_dp.h | 1 + + include/drm/display/drm_dp_helper.h | 6 + + include/drm/display/drm_hdmi_helper.h | 15 + + include/drm/display/drm_hdmi_state_helper.h | 11 +- + include/drm/display/drm_scdc.h | 21 +- + include/drm/display/drm_scdc_helper.h | 105 +- + include/drm/drm_atomic.h | 89 +- + include/drm/drm_atomic_state_helper.h | 6 - + include/drm/drm_bridge.h | 49 + + include/drm/drm_client.h | 14 + + include/drm/drm_client_event.h | 3 + + include/drm/drm_colorop.h | 127 + + include/drm/drm_connector.h | 216 +- + include/drm/drm_crtc.h | 12 - + include/drm/drm_device.h | 1 + + include/drm/drm_edid.h | 2 + + include/drm/drm_gem.h | 14 +- + include/drm/drm_gem_atomic_helper.h | 13 +- + include/drm/drm_gpusvm.h | 71 +- + include/drm/drm_mipi_dbi.h | 2 +- + include/drm/drm_mode_config.h | 4 +- + include/drm/drm_modeset_helper_vtables.h | 53 +- + include/drm/drm_panic.h | 117 +- + include/drm/drm_panic_helper.h | 41 + + include/drm/drm_plane.h | 69 +- + include/drm/drm_print.h | 3 + + include/drm/drm_ras.h | 34 + + include/drm/drm_simple_kms_helper.h | 7 +- + include/drm/drm_sysfs.h | 4 - + include/drm/intel/display_parent_interface.h | 15 +- + include/drm/ttm/ttm_resource.h | 2 +- + include/linux/dma-fence.h | 2 +- + include/linux/gpu_buddy.h | 101 +- + include/linux/hdmi.h | 12 + + include/linux/vga_switcheroo.h | 29 +- + include/sound/omap-hdmi-audio.h | 1 + + include/uapi/drm/amdgpu_drm.h | 41 + + include/uapi/drm/drm_mode.h | 12 + + include/uapi/drm/drm_ras.h | 18 + + include/uapi/drm/ivpu_accel.h | 17 +- + include/uapi/drm/xe_drm.h | 22 +- + kernel/cgroup/dmem.c | 4 +- + rust/bindings/bindings_helper.h | 4 +- + rust/kernel/drm/gem/mod.rs | 13 +- + rust/kernel/drm/gem/shmem.rs | 5 +- + sound/soc/ti/omap-hdmi.c | 51 +- + 1337 files changed, 149250 insertions(+), 25200 deletions(-) + create mode 100644 Documentation/devicetree/bindings/display/bridge/renesas,r8a779g0-dsc.yaml + create mode 100644 Documentation/devicetree/bindings/display/panel/ilitek,ili7836a.yaml + create mode 100644 Documentation/devicetree/bindings/display/panel/novatek,nt36532.yaml + create mode 100644 Documentation/devicetree/bindings/display/solomon,ssd1351.yaml + create mode 100644 Documentation/gpu/amdgpu/ualink.rst + create mode 100644 Documentation/gpu/xe/xe_sigid.rst + create mode 100644 drivers/gpu/drm/amd/amdgpu/amdgpu_sdma_types.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/amdgpu_ualink.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/amdgpu_vm_internal.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/nbio_v7_11_5.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/nbio_v7_11_5.h + create mode 100644 drivers/gpu/drm/amd/amdgpu/ualink_v1_0.c + create mode 100644 drivers/gpu/drm/amd/amdgpu/ualink_v1_0.h + create mode 100644 drivers/gpu/drm/amd/display/dc/clk_mgr/dcn10/dcn10_clk_mgr.c + create mode 100644 drivers/gpu/drm/amd/display/dc/clk_mgr/dcn10/dcn10_clk_mgr.h + create mode 100644 drivers/gpu/drm/amd/display/dc/clk_mgr/dcn60/dcn60_smu_driver_if.h + delete mode 100644 drivers/gpu/drm/amd/display/dc/dc_edid_parser.c + create mode 100644 drivers/gpu/drm/amd/display/dc/dc_memory_pool.c + create mode 100644 drivers/gpu/drm/amd/display/dc/dc_memory_pool.h + delete mode 100644 drivers/gpu/drm/amd/display/dc/dml2_0/dml21/dml21_wrapper.h + create mode 100644 drivers/gpu/drm/amd/display/dc/dml2_wrapper/Makefile + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_translation_helper.c (96%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_translation_helper.h (94%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_utils.c (97%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_utils.h (96%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_wrapper.c (100%) + create mode 100644 drivers/gpu/drm/amd/display/dc/dml2_wrapper/dml21_wrapper/dml21_wrapper.h + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_wrapper_fpu.c (96%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0/dml21 => dml2_wrapper/dml21_wrapper}/dml21_wrapper_fpu.h (95%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_dc_resource_mgmt.c (96%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_dc_resource_mgmt.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_dc_types.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_internal_types.h (99%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_mall_phantom.c (97%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_mall_phantom.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_policy.c (99%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_policy.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_translation_helper.c (99%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_translation_helper.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_utils.c (98%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_utils.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_wrapper.c (97%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_wrapper.h (100%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_wrapper_fpu.c (93%) + rename drivers/gpu/drm/amd/display/dc/{dml2_0 => dml2_wrapper}/dml2_wrapper_fpu.h (100%) + create mode 100644 drivers/gpu/drm/amd/display/dc/inc/hw/rmcm.h + create mode 100644 drivers/gpu/drm/amd/display/dc/rmcm/Makefile + create mode 100644 drivers/gpu/drm/amd/display/dc/rmcm/dcn42/dcn42_rmcm.c + create mode 100644 drivers/gpu/drm/amd/display/dc/rmcm/dcn42/dcn42_rmcm.h + create mode 100644 drivers/gpu/drm/amd/display/dc/rmcm/dcn60/dcn60_rmcm.c + rename drivers/gpu/drm/amd/display/dc/{dc_edid_parser.h => rmcm/dcn60/dcn60_rmcm.h} (71%) + delete mode 100644 drivers/gpu/drm/amd/display/dc/sspl/spl_custom_float.c + delete mode 100644 drivers/gpu/drm/amd/display/dc/sspl/spl_custom_float.h + delete mode 100644 drivers/gpu/drm/amd/display/dc/sspl/spl_fixpt31_32.c + delete mode 100644 drivers/gpu/drm/amd/display/dc/sspl/spl_fixpt31_32.h + create mode 100644 drivers/gpu/drm/amd/display/dc/sspl/spl_namespace.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/nbio/nbio_7_11_5_offset.h + create mode 100644 drivers/gpu/drm/amd/include/asic_reg/nbio/nbio_7_11_5_sh_mask.h + create mode 100644 drivers/gpu/drm/amd/include/ivsrcid/mpnht/irqsrcs_mpnht_15_0.h + create mode 100644 drivers/gpu/drm/amd/ras/core/aca_v5_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/aca_v5_0.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_bert.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_bert.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_eeprom_mgr.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_mce.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_mce.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_mp1_v15_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_mp1_v15_0.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_psp_v15_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_psp_v15_0.h + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_umc_v15_0.c + create mode 100644 drivers/gpu/drm/amd/ras/core/ras_umc_v15_0.h + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_bert.c + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_bert.h + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.c + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mce.h + create mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1.c + rename drivers/gpu/drm/amd/ras/ras_mgr/{amdgpu_ras_mp1_v13_0.h => amdgpu_ras_mp1.h} (86%) + delete mode 100644 drivers/gpu/drm/amd/ras/ras_mgr/amdgpu_ras_mp1_v13_0.c + create mode 100644 drivers/gpu/drm/drm_panic_helper.c + rename drivers/gpu/drm/{drm_panic_qr.rs => drm_panic_helper_qr.rs} (99%) + create mode 100644 drivers/gpu/drm/drm_panic_internal.h + delete mode 100644 drivers/gpu/drm/i915/display/intel_tdf.h + create mode 100644 drivers/gpu/drm/nouveau/nvkm/subdev/pci/mcp79.c + create mode 100644 drivers/gpu/drm/panel/panel-ilitek-ili7836a.c + create mode 100644 drivers/gpu/drm/panel/panel-novatek-nt36532.c + create mode 100644 drivers/gpu/drm/panel/panel-samsung-dsi.h + create mode 100644 drivers/gpu/drm/renesas/rcar-du/rcar_dsc.c + rename drivers/gpu/drm/tests/{drm_panic_test.c => drm_panic_helper_test.c} (77%) + create mode 100755 drivers/gpu/drm/vkms/tests/gen_yuv_conversion.py + create mode 100644 drivers/gpu/drm/xe/abi/xe_log_abi.h + create mode 100644 drivers/gpu/drm/xe/abi/xe_sigid_abi.h + create mode 100644 drivers/gpu/drm/xe/display/xe_display_wa.h + delete mode 100644 drivers/gpu/drm/xe/display/xe_tdf.c + create mode 100644 drivers/gpu/drm/xe/tests/xe_any_kunit.c + create mode 100644 drivers/gpu/drm/xe/tests/xe_log_kunit.c + create mode 100644 drivers/gpu/drm/xe/xe_any.h + create mode 100644 drivers/gpu/drm/xe/xe_log.c + create mode 100644 drivers/gpu/drm/xe/xe_log.h + create mode 100644 include/drm/drm_panic_helper.h +Merging drm-exynos/for-linux-next (3a8660878839f Linux 6.18-rc1) +$ git merge -m Merge branch 'for-linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/daeinki/drm-exynos.git drm-exynos/for-linux-next +Already up to date. +Merging drm-misc/for-linux-next (744f262401cc2 drm/dp/mst: reject DPCD read/write on ports with ddps=0) +$ git merge -m Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/misc/kernel.git drm-misc/for-linux-next +Auto-merging drivers/accel/ivpu/ivpu_job.c +Auto-merging drivers/gpu/drm/bridge/samsung-dsim.c +Merge made by the 'ort' strategy. + .../display/allwinner,sun6i-a31-mipi-dsi.yaml | 10 +- + .../display/bridge/fsl,imx8qxp-pxl2dpi.yaml | 4 +- + .../bindings/display/bridge/lontium,lt8912b.yaml | 8 +- + .../bridge/megachips,stdp2690-ge-b850v3-fw.yaml | 2 +- + .../bindings/display/bridge/nxp,tda998x.yaml | 2 +- + .../bindings/display/bridge/toshiba,tc358764.yaml | 2 +- + .../bindings/display/bridge/toshiba,tc358775.yaml | 12 +- + .../bindings/display/faraday,tve200.yaml | 2 +- + .../devicetree/bindings/display/himax,hx8357.yaml | 2 +- + .../bindings/display/imx/fsl,imx-lcdc.yaml | 2 +- + .../bindings/display/imx/fsl,imx6q-ldb.yaml | 54 +- + .../bindings/display/mediatek/mediatek,ethdr.yaml | 90 +-- + .../bindings/display/mediatek/mediatek,hdmi.yaml | 24 +- + .../bindings/display/panel/ilitek,il79900a.yaml | 13 +- + .../panel/panel-simple-lvds-dual-ports.yaml | 2 + + .../bindings/display/panel/panel-simple.yaml | 2 + + .../bindings/display/panel/samsung,ams495qa01.yaml | 2 +- + .../bindings/display/sitronix,st7567.yaml | 4 +- + .../devicetree/bindings/display/st,stm32-dsi.yaml | 46 +- + .../devicetree/bindings/display/st,stm32-ltdc.yaml | 6 +- + .../bindings/display/st,stm32mp25-lvds.yaml | 4 +- + Documentation/gpu/todo.rst | 33 +- + drivers/accel/amdxdna/aie2_ctx.c | 117 +++- + drivers/accel/amdxdna/aie2_msg_priv.h | 1 + + drivers/accel/amdxdna/aie2_pci.c | 15 +- + drivers/accel/amdxdna/aie2_pci.h | 6 +- + drivers/accel/amdxdna/amdxdna_mailbox.c | 2 +- + drivers/accel/amdxdna/npu4_regs.c | 1 + + drivers/accel/ivpu/ivpu_fw.c | 65 +- + drivers/accel/ivpu/ivpu_gem.c | 3 + + drivers/accel/ivpu/ivpu_ipc.c | 11 +- + drivers/accel/ivpu/ivpu_job.c | 6 +- + drivers/accel/ivpu/ivpu_ms.c | 17 +- + drivers/gpu/drm/Makefile | 3 +- + drivers/gpu/drm/adp/adp-mipi.c | 1 + + drivers/gpu/drm/arm/display/komeda/komeda_crtc.c | 1 + + drivers/gpu/drm/bridge/Kconfig | 18 +- + drivers/gpu/drm/bridge/analogix/Kconfig | 1 + + drivers/gpu/drm/bridge/analogix/analogix_dp_core.c | 40 +- + drivers/gpu/drm/bridge/aux-bridge.c | 1 + + drivers/gpu/drm/bridge/cadence/cdns-dsi-core.c | 1 + + drivers/gpu/drm/bridge/fsl-ldb.c | 19 +- + drivers/gpu/drm/bridge/imx/imx93-pdfc.c | 1 + + drivers/gpu/drm/bridge/panel.c | 563 ----------------- + drivers/gpu/drm/bridge/samsung-dsim.c | 24 +- + drivers/gpu/drm/bridge/ssd2825.c | 23 +- + drivers/gpu/drm/bridge/tc358767.c | 64 +- + drivers/gpu/drm/bridge/tc358768.c | 25 +- + drivers/gpu/drm/bridge/ti-tdp158.c | 1 + + drivers/gpu/drm/bridge/waveshare-dsi.c | 17 +- + drivers/gpu/drm/display/drm_bridge_connector.c | 1 + + drivers/gpu/drm/display/drm_dp_mst_topology.c | 6 + + drivers/gpu/drm/drm_bridge.c | 17 - + drivers/gpu/drm/drm_of.c | 63 -- + drivers/gpu/drm/drm_panel.c | 664 +++++++++++++++++++-- + drivers/gpu/drm/exynos/exynos_dp.c | 36 +- + drivers/gpu/drm/imagination/pvr_fw.c | 130 ++-- + drivers/gpu/drm/imagination/pvr_fw.h | 18 +- + drivers/gpu/drm/imx/dc/dc-kms.c | 1 + + drivers/gpu/drm/imx/dcss/Kconfig | 1 + + drivers/gpu/drm/ingenic/Kconfig | 1 + + drivers/gpu/drm/logicvc/Kconfig | 1 + + drivers/gpu/drm/mcde/Kconfig | 2 +- + drivers/gpu/drm/mcde/mcde_display.c | 1 + + drivers/gpu/drm/mcde/mcde_dsi.c | 45 +- + drivers/gpu/drm/msm/dp/dp_display.c | 1 + + drivers/gpu/drm/msm/dsi/dsi.c | 3 +- + drivers/gpu/drm/omapdrm/dss/omapdss.h | 1 - + drivers/gpu/drm/omapdrm/dss/output.c | 42 +- + drivers/gpu/drm/panel/Kconfig | 2 +- + drivers/gpu/drm/panel/panel-ebbg-ft8719.c | 23 +- + drivers/gpu/drm/panel/panel-ilitek-ili9882t.c | 234 +++++++- + drivers/gpu/drm/panel/panel-simple.c | 57 ++ + drivers/gpu/drm/pl111/Kconfig | 1 + + drivers/gpu/drm/rockchip/Kconfig | 2 + + drivers/gpu/drm/rockchip/analogix_dp-rockchip.c | 11 - + drivers/gpu/drm/scheduler/sched_main.c | 1 - + drivers/gpu/drm/solomon/ssd130x.c | 30 +- + drivers/gpu/drm/stm/Kconfig | 1 + + drivers/gpu/drm/tegra/rgb.c | 1 + + drivers/gpu/drm/tidss/Kconfig | 1 + + drivers/gpu/drm/tve200/Kconfig | 2 +- + drivers/gpu/drm/tve200/tve200_drm.h | 1 - + drivers/gpu/drm/tve200/tve200_drv.c | 31 +- + include/drm/bridge/analogix_dp.h | 1 - + include/drm/drm_bridge.h | 54 -- + include/drm/drm_of.h | 15 +- + include/drm/drm_panel.h | 99 ++- + 88 files changed, 1577 insertions(+), 1397 deletions(-) + delete mode 100644 drivers/gpu/drm/bridge/panel.c +Merging amdgpu/drm-next (76b9706e7fa1a drm/amd/display: Bump frame warning limit for all builds of dml) +$ git merge -m Merge branch 'drm-next' of https://gitlab.freedesktop.org/agd5f/linux.git amdgpu/drm-next +Already up to date. +Merging drm-intel/for-linux-next (c9e608d1247cd drm/i915/display: Update the CMN_SDP_TL in fastset path) +$ git merge -m Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/i915/kernel.git drm-intel/for-linux-next +Auto-merging drivers/gpu/drm/i915/display/intel_display_params.h +Auto-merging drivers/gpu/drm/i915/display/intel_display_types.h +Auto-merging drivers/gpu/drm/i915/display/intel_dp_link_training.c +Auto-merging drivers/gpu/drm/i915/display/intel_dp_mst.c +Auto-merging drivers/gpu/drm/i915/display/intel_psr.c +Auto-merging drivers/gpu/drm/i915/display/intel_vrr.c +Auto-merging drivers/gpu/drm/i915/display/skl_universal_plane.c +Auto-merging drivers/gpu/drm/xe/xe_device_types.h +Auto-merging drivers/gpu/drm/xe/xe_pci.c +CONFLICT (content): Merge conflict in drivers/gpu/drm/xe/xe_pci.c +Auto-merging drivers/gpu/drm/xe/xe_pm.c +Auto-merging drivers/gpu/drm/xe/xe_pm.h +Resolved 'drivers/gpu/drm/xe/xe_pci.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 4711f5c0956c1] Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/i915/kernel.git +$ git diff -M --stat --summary HEAD^.. + drivers/gpu/drm/i915/Makefile | 1 + + drivers/gpu/drm/i915/display/intel_alpm.c | 166 ++++++++++-- + drivers/gpu/drm/i915/display/intel_alpm.h | 3 + + drivers/gpu/drm/i915/display/intel_bios.c | 262 ++++++++++++++++++- + drivers/gpu/drm/i915/display/intel_bios.h | 14 ++ + drivers/gpu/drm/i915/display/intel_color.c | 76 ++++-- + .../gpu/drm/i915/display/intel_color_pipeline.c | 34 ++- + .../gpu/drm/i915/display/intel_crtc_state_dump.c | 9 + + drivers/gpu/drm/i915/display/intel_cx0_phy.c | 5 +- + drivers/gpu/drm/i915/display/intel_ddi.c | 37 ++- + drivers/gpu/drm/i915/display/intel_ddi_buf_trans.c | 90 ++++++- + drivers/gpu/drm/i915/display/intel_dip.c | 215 ++++++++++++++++ + drivers/gpu/drm/i915/display/intel_dip.h | 53 ++++ + drivers/gpu/drm/i915/display/intel_dip_regs.h | 38 +++ + drivers/gpu/drm/i915/display/intel_display.c | 60 ++++- + drivers/gpu/drm/i915/display/intel_display_core.h | 7 + + drivers/gpu/drm/i915/display/intel_display_irq.c | 18 +- + .../gpu/drm/i915/display/intel_display_limits.h | 1 + + .../gpu/drm/i915/display/intel_display_params.h | 5 + + drivers/gpu/drm/i915/display/intel_display_power.c | 9 + + .../gpu/drm/i915/display/intel_display_power_map.c | 42 +++- + drivers/gpu/drm/i915/display/intel_display_rpm.c | 7 + + drivers/gpu/drm/i915/display/intel_display_rpm.h | 1 + + drivers/gpu/drm/i915/display/intel_display_types.h | 20 ++ + drivers/gpu/drm/i915/display/intel_dp.c | 279 +++++++++++++++------ + .../gpu/drm/i915/display/intel_dp_aux_backlight.c | 78 +++--- + drivers/gpu/drm/i915/display/intel_dp_hdcp.c | 176 ++++++------- + .../gpu/drm/i915/display/intel_dp_link_training.c | 54 ++-- + drivers/gpu/drm/i915/display/intel_dp_mst.c | 2 +- + drivers/gpu/drm/i915/display/intel_dp_test.c | 77 +++--- + drivers/gpu/drm/i915/display/intel_fb.c | 2 +- + drivers/gpu/drm/i915/display/intel_fbc.c | 12 +- + drivers/gpu/drm/i915/display/intel_hotplug.c | 5 + + drivers/gpu/drm/i915/display/intel_lspcon.c | 82 +++--- + drivers/gpu/drm/i915/display/intel_plane.c | 53 +++- + drivers/gpu/drm/i915/display/intel_psr.c | 76 +++--- + drivers/gpu/drm/i915/display/intel_psr_regs.h | 2 + + drivers/gpu/drm/i915/display/intel_vrr.c | 14 +- + drivers/gpu/drm/i915/display/intel_vrr_regs.h | 6 - + drivers/gpu/drm/i915/display/skl_universal_plane.c | 57 +++-- + drivers/gpu/drm/i915/gt/intel_engine_cs.c | 2 +- + .../gpu/drm/i915/gt/intel_execlists_submission.c | 17 +- + drivers/gpu/drm/i915/gt/intel_reset.c | 8 +- + drivers/gpu/drm/i915/gt/selftest_engine_pm.c | 8 +- + drivers/gpu/drm/i915/gt/uc/intel_guc.h | 2 +- + drivers/gpu/drm/i915/gvt/aperture_gm.c | 4 +- + drivers/gpu/drm/i915/i915_dpt.c | 2 - + drivers/gpu/drm/i915/i915_drv.h | 2 - + drivers/gpu/drm/i915/i915_fb_pin.c | 6 - + drivers/gpu/drm/i915/i915_overlay.c | 5 - + drivers/gpu/drm/i915/i915_request.c | 2 - + drivers/gpu/drm/i915/selftests/igt_atomic.c | 7 + + drivers/gpu/drm/i915/vlv_iosf_sb.c | 6 +- + drivers/gpu/drm/xe/Makefile | 1 + + .../gpu/drm/xe/compat-i915-headers/i915_config.h | 16 -- + drivers/gpu/drm/xe/display/xe_display.c | 10 +- + drivers/gpu/drm/xe/display/xe_display_rpm.c | 6 + + drivers/gpu/drm/xe/xe_device_types.h | 13 + + drivers/gpu/drm/xe/xe_pci.c | 22 +- + drivers/gpu/drm/xe/xe_pm.c | 45 ++++ + drivers/gpu/drm/xe/xe_pm.h | 2 + + include/drm/intel/display_parent_interface.h | 1 + + 62 files changed, 1785 insertions(+), 550 deletions(-) + create mode 100644 drivers/gpu/drm/i915/display/intel_dip.c + create mode 100644 drivers/gpu/drm/i915/display/intel_dip.h + create mode 100644 drivers/gpu/drm/i915/display/intel_dip_regs.h + delete mode 100644 drivers/gpu/drm/xe/compat-i915-headers/i915_config.h +Merging drm-msm/msm-next (d33622598496c Merge tag 'qcom-drivers-for-7.4' into msm-next-merge-qcom-drivers-for-7.4) +$ git merge -m Merge branch 'msm-next' of https://gitlab.freedesktop.org/drm/msm.git drm-msm/msm-next +Merge made by the 'ort' strategy. +Merging drm-msm-lumag/msm-next-lumag (140b134753026 drm/msm: detach the ARM DMA mapping before attaching our own domain) +$ git merge -m Merge branch 'msm-next-lumag' of https://gitlab.freedesktop.org/lumag/msm.git drm-msm-lumag/msm-next-lumag +Already up to date. +Merging drm-xe/drm-xe-next (3fc93d311249d drm/xe/xe3p: Force non-compressible memory reads to 256B overfetches) +$ git merge -m Merge branch 'drm-xe-next' of https://gitlab.freedesktop.org/drm/xe/kernel.git drm-xe/drm-xe-next +Auto-merging drivers/gpu/drm/drm_pagemap.c +Auto-merging drivers/gpu/drm/xe/Makefile +Auto-merging drivers/gpu/drm/xe/xe_bo.h +Auto-merging drivers/gpu/drm/xe/xe_configfs.c +Auto-merging drivers/gpu/drm/xe/xe_device_types.h +Auto-merging drivers/gpu/drm/xe/xe_guc_ads.c +Auto-merging drivers/gpu/drm/xe/xe_pagefault.c +Auto-merging drivers/gpu/drm/xe/xe_pci.c +Auto-merging drivers/gpu/drm/xe/xe_vm.c +Auto-merging drivers/gpu/drm/xe/xe_wa_oob.rules +CONFLICT (content): Merge conflict in drivers/gpu/drm/xe/xe_wa_oob.rules +Resolved 'drivers/gpu/drm/xe/xe_wa_oob.rules' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master c2a5eaf3bbac2] Merge branch 'drm-xe-next' of https://gitlab.freedesktop.org/drm/xe/kernel.git +$ git diff -M --stat --summary HEAD^.. + .../ABI/testing/sysfs-driver-intel-xe-gpu | 10 + + Documentation/gpu/xe/xe_migrate.rst | 3 + + drivers/gpu/drm/drm_pagemap.c | 2 +- + drivers/gpu/drm/xe/Makefile | 1 + + drivers/gpu/drm/xe/abi/guc_klvs_abi.h | 1 + + drivers/gpu/drm/xe/regs/xe_gt_regs.h | 3 +- + drivers/gpu/drm/xe/regs/xe_regs.h | 2 + + drivers/gpu/drm/xe/xe_bo.c | 4 +- + drivers/gpu/drm/xe/xe_bo.h | 10 +- + drivers/gpu/drm/xe/xe_bo_types.h | 2 - + drivers/gpu/drm/xe/xe_configfs.c | 67 ++ + drivers/gpu/drm/xe/xe_configfs.h | 2 + + drivers/gpu/drm/xe/xe_cpu_bind.c | 296 ++++++++ + drivers/gpu/drm/xe/xe_cpu_bind.h | 111 +++ + drivers/gpu/drm/xe/xe_devcoredump.c | 46 +- + drivers/gpu/drm/xe/xe_devcoredump.h | 15 +- + drivers/gpu/drm/xe/xe_device.c | 14 + + drivers/gpu/drm/xe/xe_device_sysfs.c | 36 + + drivers/gpu/drm/xe/xe_device_types.h | 15 +- + drivers/gpu/drm/xe/xe_drm_client.c | 2 +- + drivers/gpu/drm/xe/xe_exec_queue.c | 161 ++--- + drivers/gpu/drm/xe/xe_exec_queue.h | 16 +- + drivers/gpu/drm/xe/xe_exec_queue_types.h | 20 +- + drivers/gpu/drm/xe/xe_gt_ccs_mode.c | 6 - + drivers/gpu/drm/xe/xe_guc.c | 16 + + drivers/gpu/drm/xe/xe_guc.h | 2 + + drivers/gpu/drm/xe/xe_guc_ads.c | 9 +- + drivers/gpu/drm/xe/xe_guc_ct.c | 6 +- + drivers/gpu/drm/xe/xe_guc_engine_activity.c | 6 +- + drivers/gpu/drm/xe/xe_guc_hwconfig.c | 2 +- + drivers/gpu/drm/xe/xe_guc_log.c | 7 +- + drivers/gpu/drm/xe/xe_guc_pc.c | 3 +- + drivers/gpu/drm/xe/xe_guc_submit.c | 76 +- + drivers/gpu/drm/xe/xe_guc_submit_types.h | 2 + + drivers/gpu/drm/xe/xe_i2c.c | 22 +- + drivers/gpu/drm/xe/xe_i2c.h | 1 - + drivers/gpu/drm/xe/xe_lrc.c | 73 ++ + drivers/gpu/drm/xe/xe_lrc.h | 4 + + drivers/gpu/drm/xe/xe_lrc_types.h | 4 + + drivers/gpu/drm/xe/xe_migrate.c | 769 +++++++++----------- + drivers/gpu/drm/xe/xe_migrate.h | 95 +-- + drivers/gpu/drm/xe/xe_pagefault.c | 3 + + drivers/gpu/drm/xe/xe_pci.c | 5 + + drivers/gpu/drm/xe/xe_pci_sriov.c | 3 + + drivers/gpu/drm/xe/xe_pci_types.h | 4 +- + drivers/gpu/drm/xe/xe_pt.c | 779 ++++++++++++--------- + drivers/gpu/drm/xe/xe_pt.h | 12 +- + drivers/gpu/drm/xe/xe_pt_types.h | 49 +- + drivers/gpu/drm/xe/xe_ring_ops.c | 80 ++- + drivers/gpu/drm/xe/xe_ring_ops_types.h | 24 + + drivers/gpu/drm/xe/xe_sched_job.c | 101 ++- + drivers/gpu/drm/xe/xe_sched_job.h | 56 ++ + drivers/gpu/drm/xe/xe_sched_job_types.h | 49 +- + drivers/gpu/drm/xe/xe_sriov_pf_provision.c | 4 +- + drivers/gpu/drm/xe/xe_sync.c | 20 +- + drivers/gpu/drm/xe/xe_sysctrl_mailbox.c | 7 +- + drivers/gpu/drm/xe/xe_sysctrl_mailbox_types.h | 3 + + drivers/gpu/drm/xe/xe_tlb_inval.c | 54 ++ + drivers/gpu/drm/xe/xe_tlb_inval_job.c | 28 +- + drivers/gpu/drm/xe/xe_tlb_inval_job.h | 4 +- + drivers/gpu/drm/xe/xe_tlb_inval_types.h | 17 + + drivers/gpu/drm/xe/xe_trace.h | 2 +- + drivers/gpu/drm/xe/xe_tuning.c | 4 + + drivers/gpu/drm/xe/xe_uc_fw.c | 1 + + drivers/gpu/drm/xe/xe_vm.c | 242 +++---- + drivers/gpu/drm/xe/xe_vm.h | 3 + + drivers/gpu/drm/xe/xe_vm_madvise.c | 6 - + drivers/gpu/drm/xe/xe_vm_types.h | 12 +- + drivers/gpu/drm/xe/xe_wa_oob.rules | 2 + + include/drm/intel/pciids.h | 1 + + 70 files changed, 2238 insertions(+), 1279 deletions(-) + create mode 100644 Documentation/ABI/testing/sysfs-driver-intel-xe-gpu + create mode 100644 drivers/gpu/drm/xe/xe_cpu_bind.c + create mode 100644 drivers/gpu/drm/xe/xe_cpu_bind.h +$ git am -3 ../patches/0001-drm-xe-Fix-up-merge-issue.patch +Applying: drm: xe: Fix up merge issue +Using index info to reconstruct a base tree... +M drivers/gpu/drm/xe/xe_ttm_vram_mgr.c +Falling back to patching base and 3-way merge... +Auto-merging drivers/gpu/drm/xe/xe_ttm_vram_mgr.c +No changes -- Patch already applied. +Merging drm-rust/for-linux-next (658c5f0042e6e gpu: nova-core: use a single try_pin_init!() block in probe()) +$ git merge -m Merge branch 'for-linux-next' of https://gitlab.freedesktop.org/drm/rust/kernel.git drm-rust/for-linux-next +Auto-merging MAINTAINERS +Auto-merging rust/bindings/bindings_helper.h +Auto-merging rust/helpers/helpers.c +Merge made by the 'ort' strategy. + Documentation/gpu/nova/core/fsp.rst | 2 + + Documentation/gpu/nova/core/pramin.rst | 128 +++ + Documentation/gpu/nova/core/todo.rst | 2 +- + Documentation/gpu/nova/index.rst | 1 + + MAINTAINERS | 7 + + drivers/gpu/drm/nova/driver.rs | 19 +- + drivers/gpu/drm/nova/file.rs | 119 ++- + drivers/gpu/drm/tyr/driver.rs | 1 + + drivers/gpu/drm/tyr/fw.rs | 5 +- + drivers/gpu/drm/tyr/gpu.rs | 6 +- + drivers/gpu/drm/tyr/regs.rs | 50 +- + drivers/gpu/nova-core/Kconfig | 10 + + drivers/gpu/nova-core/api.rs | 68 ++ + drivers/gpu/nova-core/driver.rs | 109 ++- + drivers/gpu/nova-core/falcon.rs | 164 ++-- + drivers/gpu/nova-core/falcon/fsp.rs | 63 +- + drivers/gpu/nova-core/falcon/gsp.rs | 51 +- + drivers/gpu/nova-core/falcon/hal/ga102.rs | 59 +- + drivers/gpu/nova-core/falcon/hal/tu102.rs | 9 +- + drivers/gpu/nova-core/falcon/sec2.rs | 37 +- + drivers/gpu/nova-core/fb.rs | 22 +- + drivers/gpu/nova-core/fb/hal/gb100.rs | 59 +- + drivers/gpu/nova-core/fb/regs.rs | 34 +- + drivers/gpu/nova-core/firmware.rs | 3 +- + drivers/gpu/nova-core/firmware/booter.rs | 2 +- + drivers/gpu/nova-core/firmware/fwsec/bootloader.rs | 30 +- + drivers/gpu/nova-core/firmware/gsp.rs | 119 +-- + .../gpu/nova-core/firmware/{fsp.rs => gsp_fmc.rs} | 59 +- + drivers/gpu/nova-core/firmware/radix3.rs | 144 +++ + drivers/gpu/nova-core/firmware/riscv.rs | 8 +- + drivers/gpu/nova-core/fsp.rs | 45 +- + drivers/gpu/nova-core/gpu.rs | 231 +++-- + drivers/gpu/nova-core/gpu/regs.rs | 86 ++ + drivers/gpu/nova-core/gsp.rs | 39 +- + drivers/gpu/nova-core/gsp/boot.rs | 26 +- + drivers/gpu/nova-core/gsp/cmdq.rs | 98 +- + drivers/gpu/nova-core/gsp/commands.rs | 40 +- + drivers/gpu/nova-core/gsp/fw.rs | 18 +- + drivers/gpu/nova-core/gsp/fw/commands.rs | 28 + + drivers/gpu/nova-core/gsp/hal.rs | 16 +- + drivers/gpu/nova-core/gsp/hal/gh100.rs | 16 +- + drivers/gpu/nova-core/gsp/hal/tu102.rs | 41 +- + drivers/gpu/nova-core/gsp/regs.rs | 9 +- + drivers/gpu/nova-core/gsp/sequencer.rs | 6 +- + drivers/gpu/nova-core/mctp.rs | 16 +- + drivers/gpu/nova-core/mm.rs | 335 +++++++ + drivers/gpu/nova-core/mm/bar_user.rs | 420 ++++++++ + drivers/gpu/nova-core/mm/hal.rs | 56 ++ + drivers/gpu/nova-core/mm/hal/gb100.rs | 35 + + drivers/gpu/nova-core/mm/hal/gh100.rs | 35 + + drivers/gpu/nova-core/mm/hal/tu102.rs | 37 + + drivers/gpu/nova-core/mm/pagetable.rs | 424 ++++++++ + drivers/gpu/nova-core/mm/pagetable/map.rs | 345 +++++++ + drivers/gpu/nova-core/mm/pagetable/ver2.rs | 275 ++++++ + drivers/gpu/nova-core/mm/pagetable/ver3.rs | 421 ++++++++ + drivers/gpu/nova-core/mm/pagetable/walk.rs | 244 +++++ + drivers/gpu/nova-core/mm/pramin.rs | 312 ++++++ + drivers/gpu/nova-core/mm/regs.rs | 70 ++ + drivers/gpu/nova-core/mm/tlb.rs | 120 +++ + drivers/gpu/nova-core/mm/vmm.rs | 346 +++++++ + drivers/gpu/nova-core/nova_core.rs | 4 + + drivers/gpu/nova-core/num.rs | 2 +- + drivers/gpu/nova-core/regs.rs | 257 ++--- + drivers/gpu/nova-core/selftest.rs | 64 ++ + drivers/gpu/nova-core/vbios.rs | 11 +- + include/uapi/drm/nova_drm.h | 128 +++ + rust/bindings/bindings_helper.h | 1 + + rust/helpers/dma_fence.c | 49 + + rust/helpers/helpers.c | 1 + + rust/helpers/pci.c | 6 + + rust/kernel/auxiliary.rs | 17 +- + rust/kernel/bitfield.rs | 9 + + rust/kernel/debugfs.rs | 26 +- + rust/kernel/debugfs/entry.rs | 4 +- + rust/kernel/debugfs/file_ops.rs | 29 +- + rust/kernel/device_id.rs | 3 +- + rust/kernel/dma.rs | 141 ++- + rust/kernel/dma_buf/dma_fence.rs | 1022 ++++++++++++++++++++ + rust/kernel/dma_buf/mod.rs | 14 + + rust/kernel/io.rs | 220 +++-- + rust/kernel/io/register.rs | 803 ++++----------- + rust/kernel/io/resource.rs | 8 + + rust/kernel/lib.rs | 3 + + rust/kernel/maple_tree.rs | 30 +- + rust/kernel/mem.rs | 234 +++++ + rust/kernel/pci.rs | 14 + + rust/kernel/sync/atomic.rs | 4 +- + rust/kernel/uaccess.rs | 18 +- + rust/macros/io/mod.rs | 3 + + rust/macros/io/register.rs | 296 ++++++ + rust/macros/lib.rs | 9 + + samples/rust/rust_dma.rs | 30 +- + samples/rust/rust_driver_pci.rs | 4 + + 93 files changed, 7452 insertions(+), 1592 deletions(-) + create mode 100644 Documentation/gpu/nova/core/pramin.rst + create mode 100644 drivers/gpu/nova-core/api.rs + rename drivers/gpu/nova-core/firmware/{fsp.rs => gsp_fmc.rs} (67%) + create mode 100644 drivers/gpu/nova-core/firmware/radix3.rs + create mode 100644 drivers/gpu/nova-core/gpu/regs.rs + create mode 100644 drivers/gpu/nova-core/mm.rs + create mode 100644 drivers/gpu/nova-core/mm/bar_user.rs + create mode 100644 drivers/gpu/nova-core/mm/hal.rs + create mode 100644 drivers/gpu/nova-core/mm/hal/gb100.rs + create mode 100644 drivers/gpu/nova-core/mm/hal/gh100.rs + create mode 100644 drivers/gpu/nova-core/mm/hal/tu102.rs + create mode 100644 drivers/gpu/nova-core/mm/pagetable.rs + create mode 100644 drivers/gpu/nova-core/mm/pagetable/map.rs + create mode 100644 drivers/gpu/nova-core/mm/pagetable/ver2.rs + create mode 100644 drivers/gpu/nova-core/mm/pagetable/ver3.rs + create mode 100644 drivers/gpu/nova-core/mm/pagetable/walk.rs + create mode 100644 drivers/gpu/nova-core/mm/pramin.rs + create mode 100644 drivers/gpu/nova-core/mm/regs.rs + create mode 100644 drivers/gpu/nova-core/mm/tlb.rs + create mode 100644 drivers/gpu/nova-core/mm/vmm.rs + create mode 100644 drivers/gpu/nova-core/selftest.rs + create mode 100644 rust/helpers/dma_fence.c + create mode 100644 rust/kernel/dma_buf/dma_fence.rs + create mode 100644 rust/kernel/dma_buf/mod.rs + create mode 100644 rust/kernel/mem.rs + create mode 100644 rust/macros/io/mod.rs + create mode 100644 rust/macros/io/register.rs +Merging drm-nova/nova-next (93296e9d9528f gpu: nova-core: vbios: store reference to Device where relevant) +$ git merge -m Merge branch 'nova-next' of https://gitlab.freedesktop.org/drm/nova.git drm-nova/nova-next +Already up to date. +Merging etnaviv/etnaviv/next (6bde14ba5f7ef drm/etnaviv: add optional reset support) +$ git merge -m Merge branch 'etnaviv/next' of https://git.pengutronix.de/git/lst/linux etnaviv/etnaviv/next +Already up to date. +Merging fbdev/for-next (9f4c6043f33c9 fbdev: atafb: avoid cast when assigning buffer address pointer) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/deller/linux-fbdev.git fbdev/for-next +Merge made by the 'ort' strategy. + drivers/video/fbdev/acornfb.c | 57 ++++++------------------- + drivers/video/fbdev/acornfb.h | 9 +--- + drivers/video/fbdev/atafb.c | 28 +----------- + drivers/video/fbdev/aty/aty128fb.c | 2 +- + drivers/video/fbdev/aty/atyfb_base.c | 4 +- + drivers/video/fbdev/aty/mach64_cursor.c | 2 +- + drivers/video/fbdev/aty/radeon_base.c | 4 +- + drivers/video/fbdev/aty/radeon_monitor.c | 4 +- + drivers/video/fbdev/aty/radeonfb.h | 2 +- + drivers/video/fbdev/au1100fb.c | 2 +- + drivers/video/fbdev/core/Kconfig | 6 --- + drivers/video/fbdev/efifb.c | 4 +- + drivers/video/fbdev/grvga.c | 2 +- + drivers/video/fbdev/gxt4500.c | 1 + + drivers/video/fbdev/kyro/STG4000OverlayDevice.c | 2 +- + drivers/video/fbdev/kyro/fbdev.c | 4 +- + drivers/video/fbdev/maxinefb.c | 2 +- + drivers/video/fbdev/mmp/core.c | 2 +- + drivers/video/fbdev/mmp/hw/mmp_ctrl.c | 4 +- + drivers/video/fbdev/mmp/hw/mmp_ctrl.h | 4 +- + drivers/video/fbdev/mmp/hw/mmp_spi.c | 2 +- + drivers/video/fbdev/nvidia/nvidia.c | 1 + + drivers/video/fbdev/ocfb.c | 2 +- + drivers/video/fbdev/omap2/omapfb/dss/dispc.c | 4 +- + drivers/video/fbdev/pm2fb.c | 2 +- + drivers/video/fbdev/s1d13xxxfb.c | 6 +-- + drivers/video/fbdev/sh_mobile_lcdcfb.c | 37 ++++++++++------ + drivers/video/fbdev/skeletonfb.c | 6 +-- + drivers/video/fbdev/sstfb.c | 5 ++- + drivers/video/fbdev/tgafb.c | 6 +-- + drivers/video/fbdev/tridentfb.c | 2 +- + drivers/video/fbdev/udlfb.c | 5 +++ + drivers/video/fbdev/via/dvi.c | 6 +-- + drivers/video/fbdev/via/hw.c | 2 +- + drivers/video/fbdev/via/viafbdev.c | 4 +- + drivers/video/sticore.c | 2 +- + include/video/mmp_disp.h | 6 +-- + 37 files changed, 100 insertions(+), 143 deletions(-) +$ git am -3 ../patches/0001-fix-up-for-drm-hyperv-Remove-reference-to-hyperv_fb-.patch +Applying: fix up for "drm/hyperv: Remove reference to hyperv_fb driver" +$ git reset HEAD^ +Unstaged changes after reset: +M drivers/gpu/drm/hyperv/Kconfig +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master 5bee4e96fdc0d] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/deller/linux-fbdev.git + Date: Wed Sep 30 13:14:38 2026 +0100 +Merging regmap/for-next (117e6a5fd98fb Merge regmap/for-7.4 into regmap-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regmap.git regmap/for-next +Merge made by the 'ort' strategy. + drivers/base/regmap/internal.h | 12 +++ + drivers/base/regmap/regcache-rbtree.c | 4 +- + drivers/base/regmap/regcache.c | 28 ++---- + drivers/base/regmap/regmap-debugfs.c | 24 ++--- + drivers/base/regmap/regmap-kunit.c | 98 +++++++++++++++++- + drivers/base/regmap/regmap-ram.c | 26 +++-- + drivers/base/regmap/regmap-raw-ram.c | 21 ++-- + drivers/base/regmap/regmap.c | 184 +++++++++++----------------------- + 8 files changed, 219 insertions(+), 178 deletions(-) +Merging sound/for-next (d4febca239be3 ALSA: control: Fix UAF in snd_ctl_elem_add() on card disconnect) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tiwai/sound.git sound/for-next +Auto-merging MAINTAINERS +Auto-merging drivers/acpi/scan.c +Auto-merging sound/core/pcm_native.c +Merge made by the 'ort' strategy. + Documentation/sound/alsa-configuration.rst | 6 + + Documentation/sound/cards/hdspm.rst | 109 ++--- + MAINTAINERS | 7 + + drivers/acpi/scan.c | 1 + + drivers/platform/x86/serial-multi-instantiate.c | 1 + + include/sound/control.h | 2 +- + include/sound/emu10k1.h | 4 +- + include/sound/emux_legacy.h | 2 +- + include/sound/rawmidi.h | 1 + + include/uapi/sound/asound.h | 3 +- + sound/core/control.c | 2 + + sound/core/init.c | 12 +- + sound/core/pcm_native.c | 6 +- + sound/core/rawmidi.c | 83 +++- + sound/core/seq/Kconfig | 7 +- + sound/core/seq/oss/seq_oss.c | 3 + + sound/core/seq/oss/seq_oss_init.c | 2 +- + sound/core/timer.c | 1 + + sound/core/timer_compat.c | 1 + + sound/drivers/dummy.c | 10 +- + sound/drivers/mpu401/mpu401_uart.c | 8 +- + sound/drivers/serial-generic.c | 10 +- + sound/firewire/bebob/bebob_hwdep.c | 2 +- + sound/hda/codecs/conexant.c | 67 +++ + sound/hda/codecs/hdmi/hdmi.c | 2 +- + sound/hda/codecs/realtek/alc269.c | 147 +++++- + sound/hda/codecs/realtek/alc882.c | 2 +- + sound/hda/codecs/side-codecs/cs35l41_hda_i2c.c | 3 + + .../hda/codecs/side-codecs/cs35l41_hda_property.c | 28 ++ + sound/hda/codecs/sigmatel.c | 6 +- + sound/hda/common/codec.c | 2 +- + sound/hda/controllers/acpi.c | 41 ++ + sound/hda/core/bus.c | 2 +- + sound/hda/core/component.c | 5 +- + sound/hda/core/controller.c | 5 +- + sound/i2c/other/ak4113.c | 2 +- + sound/i2c/other/ak4114.c | 2 +- + sound/i2c/other/ak4xxx-adda.c | 2 +- + sound/isa/opti9xx/opti92x-ad1848.c | 2 +- + sound/mips/hal2.c | 4 +- + sound/mips/sgio2audio.c | 13 +- + sound/oss/dmasound/dmasound_q40.c | 65 ++- + sound/pci/ac97/ac97_codec.c | 2 +- + sound/pci/ac97/ac97_patch.c | 2 +- + sound/pci/au88x0/au88x0.h | 2 +- + sound/pci/au88x0/au88x0_mpu401.c | 2 +- + sound/pci/aw2/aw2-saa7146.c | 4 +- + sound/pci/cs4281.c | 2 +- + sound/pci/cs46xx/cs46xx_lib.c | 6 +- + sound/pci/cs46xx/dsp_spos_scb_lib.c | 8 +- + sound/pci/echoaudio/echoaudio.c | 4 +- + sound/pci/echoaudio/echoaudio.h | 4 +- + sound/pci/lx6464es/lx_core.c | 2 +- + sound/pci/lx6464es/lx_defs.h | 2 +- + sound/pci/oxygen/oxygen_pcm.c | 37 +- + sound/pci/pcxhr/pcxhr_core.c | 4 +- + sound/pci/riptide/riptide.c | 4 +- + sound/pci/rme32.c | 2 +- + sound/pci/rme9652/hdsp.c | 2 +- + sound/pci/trident/trident_memory.c | 2 +- + sound/pci/via82xx.c | 11 +- + sound/pci/vx222/vx222_ops.c | 2 +- + sound/pcmcia/vx/vxp_ops.c | 2 +- + sound/ppc/tumbler.c | 2 +- + sound/soc/codecs/tas675x.c | 10 +- + sound/usb/Makefile | 1 + + sound/usb/caiaq/Makefile | 2 +- + sound/usb/caiaq/device.c | 12 + + sound/usb/caiaq/device.h | 13 + + sound/usb/caiaq/input.c | 29 +- + sound/usb/caiaq/lcd.c | 258 +++++++++++ + sound/usb/caiaq/lcd.h | 7 + + sound/usb/clock.c | 27 ++ + sound/usb/midi.c | 69 +-- + sound/usb/mixer.c | 4 +- + sound/usb/mixer_evo.c | 505 +++++++++++++++++++++ + sound/usb/mixer_evo.h | 12 + + sound/usb/mixer_maps.c | 13 + + sound/usb/mixer_quirks.c | 5 + + sound/usb/mixer_us16x08.h | 2 +- + sound/usb/quirks-table.h | 253 ++++++++++- + sound/usb/quirks.c | 235 +++++++++- + sound/usb/usbaudio.h | 6 + + tools/testing/selftests/alsa/mixer-test.c | 255 +++++++++++ + tools/testing/selftests/alsa/utimer-test.c | 6 + + 85 files changed, 2215 insertions(+), 300 deletions(-) + create mode 100644 sound/usb/caiaq/lcd.c + create mode 100644 sound/usb/caiaq/lcd.h + create mode 100644 sound/usb/mixer_evo.c + create mode 100644 sound/usb/mixer_evo.h +Merging ieee1394/for-next (a6e7c3836b812 firewire: cdev: use kzalloc_flex() to allocate structure with byte array) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ieee1394/linux1394.git ieee1394/for-next +Merge made by the 'ort' strategy. + drivers/firewire/.kunitconfig | 1 + + drivers/firewire/Kconfig | 16 ++ + drivers/firewire/config-rom-generator-test.c | 407 +++++++++++++++++++++++++++ + drivers/firewire/config-rom-parser-test.c | 355 +++++++++++++++++++++++ + drivers/firewire/core-card.c | 11 +- + drivers/firewire/core-cdev.c | 299 ++++++++++---------- + drivers/firewire/core-device.c | 4 + + drivers/firewire/core-transaction.c | 146 +++++----- + drivers/firewire/device-attribute-test.c | 4 +- + drivers/firewire/ohci-serdes-test.c | 4 +- + drivers/firewire/ohci.c | 267 ++++++++++++------ + drivers/firewire/ohci.h | 2 +- + drivers/firewire/uapi-test.c | 20 ++ + include/linux/firewire.h | 50 ++-- + include/uapi/linux/firewire-cdev.h | 4 +- + tools/firewire/nosy-dump.c | 10 +- + 16 files changed, 1252 insertions(+), 348 deletions(-) + create mode 100644 drivers/firewire/config-rom-generator-test.c + create mode 100644 drivers/firewire/config-rom-parser-test.c +Merging sound-asoc/for-next (0fbfaeecfd6f5 Merge asoc/for-7.4 into asoc-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/sound.git sound-asoc/for-next +Auto-merging Documentation/devicetree/bindings/vendor-prefixes.yaml +Auto-merging MAINTAINERS +Auto-merging sound/soc/amd/acp-es8336.c +Auto-merging sound/soc/amd/acp/acp3x-es83xx/acp3x-es83xx.c +Auto-merging sound/soc/intel/boards/sof_cirrus_common.c +Merge made by the 'ort' strategy. + .../devicetree/bindings/sound/adi,adau1372.yaml | 10 +- + .../devicetree/bindings/sound/adi,adau7118.yaml | 22 +- + .../bindings/sound/airoha,an7581-afe.yaml | 41 + + .../bindings/sound/airoha,an7581-wm8960.yaml | 71 + + .../bindings/sound/amlogic,gx-sound-card.yaml | 20 +- + .../bindings/sound/asahi-kasei,ak4619.yaml | 16 +- + .../bindings/sound/audio-graph-port.yaml | 10 +- + .../devicetree/bindings/sound/cirrus,cs35l45.yaml | 60 +- + .../devicetree/bindings/sound/cirrus,cs4270.yaml | 12 +- + .../devicetree/bindings/sound/cirrus,cs42l42.yaml | 42 +- + .../devicetree/bindings/sound/cirrus,cs42l84.yaml | 18 +- + .../devicetree/bindings/sound/cirrus,cs42xx8.yaml | 56 +- + .../bindings/sound/davinci-evm-audio.txt | 49 - + .../devicetree/bindings/sound/dialog,da7219.yaml | 62 +- + .../bindings/sound/esstech,es9039q2m.yaml | 94 + + .../devicetree/bindings/sound/everest,es8316.yaml | 5 + + .../bindings/sound/foursemi,fs2105s.yaml | 2 +- + .../devicetree/bindings/sound/fsl,easrc.yaml | 45 + + .../devicetree/bindings/sound/fsl,imx-asrc.yaml | 45 +- + .../devicetree/bindings/sound/fsl,sai.yaml | 14 +- + .../bindings/sound/hisilicon,hi6210-i2s.yaml | 4 + + .../devicetree/bindings/sound/imx-audio-card.yaml | 4 +- + .../devicetree/bindings/sound/imx-audmux.yaml | 8 +- + .../bindings/sound/invensense,ics43432.yaml | 8 +- + .../bindings/sound/loongson,ls-audio-card.yaml | 4 +- + .../bindings/sound/mediatek,mt2701-audio.yaml | 10 +- + .../bindings/sound/mediatek,mt8173-afe-pcm.yaml | 22 +- + .../bindings/sound/mediatek,mt8183-audio.yaml | 82 +- + .../sound/mt8186-mt6366-da7219-max98357.yaml | 36 +- + .../sound/mt8186-mt6366-rt1019-rt5682s.yaml | 36 +- + .../sound/mt8192-mt6359-rt1015-rt5682.yaml | 46 +- + .../devicetree/bindings/sound/mt8195-mt6359.yaml | 50 +- + .../devicetree/bindings/sound/nuvoton,nau8360.yaml | 115 + + .../bindings/sound/qcom,q6apm-lpass-dais.yaml | 12 +- + .../bindings/sound/qcom,q6dsp-lpass-ports.yaml | 16 +- + .../devicetree/bindings/sound/qcom,sm8250.yaml | 1 + + .../bindings/sound/qcom,wcd9378-sdw.yaml | 105 + + .../devicetree/bindings/sound/realtek,rt5677.yaml | 10 + + .../devicetree/bindings/sound/renesas,rsnd.yaml | 16 +- + .../devicetree/bindings/sound/renesas,rz-ssi.yaml | 32 +- + .../bindings/sound/stericsson,ux500-msp-i2s.yaml | 96 + + .../bindings/sound/ti,da830-evm-audio.yaml | 83 + + .../devicetree/bindings/sound/ti,tas2781.yaml | 4 +- + .../devicetree/bindings/sound/ti,tas5805m.yaml | 10 +- + .../bindings/sound/ti,tlv320dac3100.yaml | 2 +- + .../devicetree/bindings/sound/ti,tpa6130a2.yaml | 2 +- + .../devicetree/bindings/sound/ux500-mop500.txt | 39 - + .../devicetree/bindings/sound/ux500-msp.txt | 42 - + .../devicetree/bindings/vendor-prefixes.yaml | 2 + + MAINTAINERS | 18 + + drivers/firmware/cirrus/cs_dsp.c | 2 +- + drivers/gpib/fmh_gpib/fmh_gpib.c | 10 +- + include/dt-bindings/sound/qcom,q6dsp-lpass-ports.h | 52 + + include/linux/dmaengine.h | 9 - + include/linux/firmware/cirrus/cs_dsp.h | 2 +- + include/sound/sdca.h | 23 +- + include/sound/sdca_class.h | 70 + + include/sound/simple_card_utils.h | 2 + + include/sound/soc-component.h | 14 +- + include/sound/soc-dai.h | 179 +- + include/sound/soc.h | 16 +- + include/sound/soc_sdw_utils.h | 20 + + include/sound/tas2781-dsp.h | 13 + + include/sound/tas2781.h | 24 + + sound/hda/core/intel-nhlt.c | 4 +- + sound/soc/amd/acp-es8336.c | 3 +- + sound/soc/amd/acp/Kconfig | 4 +- + sound/soc/amd/acp/acp3x-es83xx/acp3x-es83xx.c | 12 + + sound/soc/amd/acp/soc_amd_sdw_common.h | 13 + + sound/soc/atmel/atmel-pcm-dma.c | 2 +- + sound/soc/bcm/cygnus-ssp.h | 2 +- + sound/soc/codecs/88pm860x-codec.c | 6 + + sound/soc/codecs/Kconfig | 49 + + sound/soc/codecs/Makefile | 8 + + sound/soc/codecs/ab8500-codec.c | 157 +- + sound/soc/codecs/adau1372.c | 12 + + sound/soc/codecs/adau1373.c | 12 + + sound/soc/codecs/adau1701.c | 15 +- + sound/soc/codecs/adau17x1.c | 17 +- + sound/soc/codecs/adau1977.c | 13 + + sound/soc/codecs/adau7118.c | 12 + + sound/soc/codecs/ak4104.c | 7 + + sound/soc/codecs/ak4118.c | 7 + + sound/soc/codecs/ak4458.c | 9 + + sound/soc/codecs/ak4535.c | 6 + + sound/soc/codecs/ak4619.c | 32 + + sound/soc/codecs/ak4642.c | 14 +- + sound/soc/codecs/ak4671.c | 7 + + sound/soc/codecs/ak5386.c | 6 + + sound/soc/codecs/ak5558.c | 7 + + sound/soc/codecs/alc5623.c | 11 + + sound/soc/codecs/alc5632.c | 13 + + sound/soc/codecs/arizona-jack.c | 1 + + sound/soc/codecs/arizona.c | 23 + + sound/soc/codecs/cpcap.c | 87 +- + sound/soc/codecs/cs35l33.c | 8 +- + sound/soc/codecs/cs35l34.c | 2 +- + sound/soc/codecs/cs35l35.c | 7 + + sound/soc/codecs/cs35l36.c | 56 + + sound/soc/codecs/cs35l41.c | 10 + + sound/soc/codecs/cs35l45-tables.c | 4 + + sound/soc/codecs/cs35l45.c | 97 +- + sound/soc/codecs/cs35l45.h | 23 + + sound/soc/codecs/cs35l56-shared-test.c | 56 + + sound/soc/codecs/cs35l56.c | 10 + + sound/soc/codecs/cs40l50-codec.c | 9 + + sound/soc/codecs/cs4234.c | 9 + + sound/soc/codecs/cs4265.c | 12 + + sound/soc/codecs/cs4270.c | 6 + + sound/soc/codecs/cs4271.c | 6 + + sound/soc/codecs/cs42l42.c | 11 +- + sound/soc/codecs/cs42l43.c | 12 + + sound/soc/codecs/cs42l51.c | 7 + + sound/soc/codecs/cs42l52.c | 22 +- + sound/soc/codecs/cs42l56.c | 17 +- + sound/soc/codecs/cs42l73.c | 11 +- + sound/soc/codecs/cs42l84.c | 6 + + sound/soc/codecs/cs42xx8.c | 8 + + sound/soc/codecs/cs43130.c | 14 + + sound/soc/codecs/cs4341.c | 8 + + sound/soc/codecs/cs4349.c | 12 + + sound/soc/codecs/cs48l32.c | 23 + + sound/soc/codecs/cs530x.c | 9 + + sound/soc/codecs/cs53l30.c | 8 + + sound/soc/codecs/cx20442.c | 2 +- + sound/soc/codecs/cx2072x.c | 12 + + sound/soc/codecs/da7210.c | 9 +- + sound/soc/codecs/da7218.c | 141 +- + sound/soc/codecs/da7219.c | 12 + + sound/soc/codecs/da732x.c | 17 + + sound/soc/codecs/da9055.c | 8 + + sound/soc/codecs/es7134.c | 6 + + sound/soc/codecs/es7241.c | 7 + + sound/soc/codecs/es8311.c | 27 +- + sound/soc/codecs/es8316.c | 27 +- + sound/soc/codecs/es8323.c | 32 +- + sound/soc/codecs/es8326.c | 181 +- + sound/soc/codecs/es8326.h | 4 + + sound/soc/codecs/es8328.c | 20 +- + sound/soc/codecs/es8375.c | 30 +- + sound/soc/codecs/es8389.c | 65 +- + sound/soc/codecs/es9039q2m.c | 2008 +++++++++++++++++ + sound/soc/codecs/fs210x.c | 5 +- + sound/soc/codecs/hda.c | 7 +- + sound/soc/codecs/hdac_hda.c | 2 +- + sound/soc/codecs/hdac_hdmi.c | 2 +- + sound/soc/codecs/inno_rk3036.c | 12 + + sound/soc/codecs/isabelle.c | 13 + + sound/soc/codecs/lm49453.c | 15 + + sound/soc/codecs/lpass-rx-macro.c | 6 +- + sound/soc/codecs/lpass-tx-macro.c | 4 +- + sound/soc/codecs/lpass-va-macro.c | 4 +- + sound/soc/codecs/lpass-wsa-macro.c | 48 +- + sound/soc/codecs/max98088.c | 12 + + sound/soc/codecs/max98090.c | 35 +- + sound/soc/codecs/max98095.c | 24 +- + sound/soc/codecs/max98371.c | 7 + + sound/soc/codecs/max98373-i2c.c | 10 + + sound/soc/codecs/max98388.c | 10 + + sound/soc/codecs/max98390.c | 10 + + sound/soc/codecs/max98396.c | 12 + + sound/soc/codecs/max9850.c | 11 + + sound/soc/codecs/max98520.c | 10 + + sound/soc/codecs/max9860.c | 23 + + sound/soc/codecs/max9867.c | 10 + + sound/soc/codecs/max98925.c | 8 + + sound/soc/codecs/max98926.c | 8 + + sound/soc/codecs/max98927.c | 14 +- + sound/soc/codecs/mc13783.c | 12 + + sound/soc/codecs/ml26124.c | 6 + + sound/soc/codecs/mt6359.c | 8 +- + sound/soc/codecs/nau8325.c | 11 + + sound/soc/codecs/nau8360-dsp.c | 634 ++++++ + sound/soc/codecs/nau8360-dsp.h | 122 ++ + sound/soc/codecs/nau8360.c | 2309 ++++++++++++++++++++ + sound/soc/codecs/nau8360.h | 917 ++++++++ + sound/soc/codecs/nau8540.c | 15 +- + sound/soc/codecs/nau8810.c | 12 + + sound/soc/codecs/nau8821.c | 11 + + sound/soc/codecs/nau8822.c | 27 +- + sound/soc/codecs/nau8824.c | 11 + + sound/soc/codecs/nau8825.c | 29 +- + sound/soc/codecs/ntp8835.c | 7 + + sound/soc/codecs/ntp8918.c | 7 + + sound/soc/codecs/pcm5102a.c | 30 +- + sound/soc/codecs/pcm6240.c | 8 +- + sound/soc/codecs/rk3308_codec.c | 11 + + sound/soc/codecs/rk3328_codec.c | 9 + + sound/soc/codecs/rt1011.c | 44 +- + sound/soc/codecs/rt1015.c | 91 +- + sound/soc/codecs/rt1016.c | 14 +- + sound/soc/codecs/rt1019.c | 10 + + sound/soc/codecs/rt1305.c | 14 +- + sound/soc/codecs/rt1308.c | 14 +- + sound/soc/codecs/rt1318.c | 13 +- + sound/soc/codecs/rt1320-sdw.c | 12 +- + sound/soc/codecs/rt274.c | 8 + + sound/soc/codecs/rt286.c | 8 + + sound/soc/codecs/rt298.c | 8 + + sound/soc/codecs/rt5514-spi.c | 37 +- + sound/soc/codecs/rt5514.c | 39 +- + sound/soc/codecs/rt5616.c | 23 +- + sound/soc/codecs/rt5631.c | 10 + + sound/soc/codecs/rt5631.h | 8 +- + sound/soc/codecs/rt5640.c | 22 +- + sound/soc/codecs/rt5645.c | 23 +- + sound/soc/codecs/rt5651.c | 12 +- + sound/soc/codecs/rt5659.c | 14 +- + sound/soc/codecs/rt5660.c | 14 +- + sound/soc/codecs/rt5663.c | 10 + + sound/soc/codecs/rt5665.c | 14 +- + sound/soc/codecs/rt5668.c | 22 + + sound/soc/codecs/rt5670.c | 12 +- + sound/soc/codecs/rt5677-spi.c | 21 +- + sound/soc/codecs/rt5677.c | 10 + + sound/soc/codecs/rt5682.c | 22 + + sound/soc/codecs/rt5682s.c | 34 +- + sound/soc/codecs/rt712-sdca-dmic.c | 4 +- + sound/soc/codecs/rt712-sdca-sdw.c | 1 + + sound/soc/codecs/rt712-sdca-sdw.h | 3 + + sound/soc/codecs/rt712-sdca.c | 136 ++ + sound/soc/codecs/rt712-sdca.h | 6 + + sound/soc/codecs/rt721-sdca-sdw.c | 15 + + sound/soc/codecs/rt721-sdca.c | 498 +++-- + sound/soc/codecs/rt721-sdca.h | 17 + + sound/soc/codecs/rt766-sdca.c | 2 +- + sound/soc/codecs/rt9120.c | 9 + + sound/soc/codecs/rt9123.c | 13 + + sound/soc/codecs/rtq9124.c | 9 + + sound/soc/codecs/rtq9128.c | 9 + + sound/soc/codecs/sgtl5000.c | 11 + + sound/soc/codecs/si476x.c | 18 + + sound/soc/codecs/sma1303.c | 13 + + sound/soc/codecs/sma1307.c | 27 +- + sound/soc/codecs/sn624x-sdca-sdw.c | 1888 ++++++++++++++++ + sound/soc/codecs/sn624x-sdca.h | 152 ++ + sound/soc/codecs/src4xxx.c | 8 + + sound/soc/codecs/ssm2518.c | 13 + + sound/soc/codecs/ssm2602.c | 13 + + sound/soc/codecs/ssm3515.c | 10 + + sound/soc/codecs/ssm4567.c | 13 + + sound/soc/codecs/sta32x.c | 116 +- + sound/soc/codecs/sta350.c | 15 +- + sound/soc/codecs/sta529.c | 7 + + sound/soc/codecs/tas2552.c | 14 + + sound/soc/codecs/tas2562.c | 10 + + sound/soc/codecs/tas2764.c | 14 +- + sound/soc/codecs/tas2770.c | 14 +- + sound/soc/codecs/tas2780.c | 10 + + sound/soc/codecs/tas2781-comlib-i2c.c | 7 +- + sound/soc/codecs/tas2781-comlib.c | 22 +- + sound/soc/codecs/tas2781-fmwlib.c | 40 +- + sound/soc/codecs/tas2781-i2c.c | 320 ++- + sound/soc/codecs/tas2783-sdw.c | 218 +- + sound/soc/codecs/tas2783.h | 4 +- + sound/soc/codecs/tas5086.c | 7 + + sound/soc/codecs/tas571x.c | 7 + + sound/soc/codecs/tas5720.c | 9 + + sound/soc/codecs/tas6424.c | 9 + + sound/soc/codecs/tfa9879.c | 9 + + sound/soc/codecs/tlv320adc3xxx.c | 14 + + sound/soc/codecs/tlv320adcx140.c | 12 + + sound/soc/codecs/tlv320aic23.c | 9 + + sound/soc/codecs/tlv320aic26.c | 8 + + sound/soc/codecs/tlv320aic31xx.c | 41 +- + sound/soc/codecs/tlv320aic32x4-clk.c | 2 +- + sound/soc/codecs/tlv320aic32x4.c | 102 +- + sound/soc/codecs/tlv320aic3x.c | 42 +- + sound/soc/codecs/tlv320dac33.c | 10 +- + sound/soc/codecs/tscs454.c | 15 + + sound/soc/codecs/twl4030.c | 80 +- + sound/soc/codecs/uda1334.c | 6 + + sound/soc/codecs/uda1342.c | 7 + + sound/soc/codecs/uda1380.c | 11 + + sound/soc/codecs/wcd-mbhc-v2.c | 8 +- + sound/soc/codecs/wcd9335.c | 25 +- + sound/soc/codecs/wcd934x.c | 2 +- + sound/soc/codecs/wcd9378-sdca.c | 1080 +++++++++ + sound/soc/codecs/wcd9378-sdca.h | 20 + + sound/soc/codecs/wcd9378-sdw.c | 51 + + sound/soc/codecs/wm2200.c | 10 + + sound/soc/codecs/wm5100.c | 10 + + sound/soc/codecs/wm8350.c | 24 + + sound/soc/codecs/wm8400.c | 9 + + sound/soc/codecs/wm8510.c | 12 + + sound/soc/codecs/wm8523.c | 13 + + sound/soc/codecs/wm8524.c | 6 + + sound/soc/codecs/wm8580.c | 20 + + sound/soc/codecs/wm8711.c | 13 + + sound/soc/codecs/wm8728.c | 9 + + sound/soc/codecs/wm8731.c | 13 + + sound/soc/codecs/wm8737.c | 11 + + sound/soc/codecs/wm8741.c | 13 + + sound/soc/codecs/wm8750.c | 13 + + sound/soc/codecs/wm8753.c | 20 + + sound/soc/codecs/wm8770.c | 11 + + sound/soc/codecs/wm8776.c | 13 + + sound/soc/codecs/wm8804.c | 15 +- + sound/soc/codecs/wm8900.c | 18 + + sound/soc/codecs/wm8903.c | 18 + + sound/soc/codecs/wm8904.c | 20 +- + sound/soc/codecs/wm8940.c | 13 + + sound/soc/codecs/wm8955.c | 42 +- + sound/soc/codecs/wm8958-dsp2.c | 8 +- + sound/soc/codecs/wm8960.c | 13 + + sound/soc/codecs/wm8961.c | 18 + + sound/soc/codecs/wm8962.c | 74 +- + sound/soc/codecs/wm8971.c | 13 + + sound/soc/codecs/wm8974.c | 17 + + sound/soc/codecs/wm8978.c | 12 + + sound/soc/codecs/wm8983.c | 12 + + sound/soc/codecs/wm8985.c | 53 +- + sound/soc/codecs/wm8988.c | 13 + + sound/soc/codecs/wm8990.c | 9 + + sound/soc/codecs/wm8991.c | 9 + + sound/soc/codecs/wm8993.c | 18 + + sound/soc/codecs/wm8994.c | 23 + + sound/soc/codecs/wm8995.c | 61 +- + sound/soc/codecs/wm8996.c | 14 +- + sound/soc/codecs/wm9081.c | 18 + + sound/soc/codecs/wm9713.c | 13 + + sound/soc/codecs/wm_hubs.c | 2 +- + sound/soc/codecs/zl38060.c | 6 + + sound/soc/dwc/dwc-i2s.c | 9 + + sound/soc/fsl/Kconfig | 2 +- + sound/soc/fsl/fsl_asrc.c | 110 +- + sound/soc/fsl/fsl_asrc_common.h | 7 +- + sound/soc/fsl/fsl_asrc_dma.c | 38 +- + sound/soc/fsl/fsl_asrc_m2m.c | 17 +- + sound/soc/fsl/fsl_easrc.c | 125 +- + sound/soc/fsl/fsl_mqs.c | 14 + + sound/soc/fsl/fsl_xcvr.c | 28 +- + sound/soc/generic/audio-graph-card.c | 1 + + sound/soc/generic/audio-graph-card2.c | 4 +- + sound/soc/generic/simple-card-utils.c | 23 +- + sound/soc/generic/simple-card.c | 13 +- + sound/soc/hisilicon/hi6210-i2s.c | 7 + + sound/soc/img/img-i2s-in.c | 12 +- + sound/soc/img/img-i2s-out.c | 13 +- + sound/soc/img/img-parallel-out.c | 8 +- + sound/soc/intel/atom/sst-mfld-dsp.h | 2 +- + sound/soc/intel/atom/sst/sst.h | 2 +- + sound/soc/intel/atom/sst/sst_acpi.c | 14 +- + sound/soc/intel/avs/boards/hdaudio.c | 8 +- + sound/soc/intel/avs/pcm.c | 11 +- + sound/soc/intel/avs/topology.c | 25 - + sound/soc/intel/avs/topology.h | 3 - + sound/soc/intel/boards/Kconfig | 1 + + sound/soc/intel/boards/sof_cirrus_common.c | 2 +- + sound/soc/intel/common/soc-acpi-intel-arl-match.c | 71 + + sound/soc/intel/common/soc-acpi-intel-lnl-match.c | 71 + + sound/soc/intel/common/soc-acpi-intel-mtl-match.c | 71 + + sound/soc/intel/common/soc-acpi-intel-ptl-match.c | 85 + + sound/soc/intel/common/sof-function-topology-lib.c | 289 ++- + sound/soc/intel/common/sof-function-topology-lib.h | 3 + + sound/soc/jz4740/jz4740-i2s.c | 7 + + sound/soc/kirkwood/kirkwood-i2s.c | 7 + + sound/soc/loongson/loongson_i2s.c | 6 + + sound/soc/mediatek/Kconfig | 27 +- + sound/soc/mediatek/Makefile | 1 + + sound/soc/mediatek/an7581/Makefile | 9 + + sound/soc/mediatek/an7581/an7581-afe-common.h | 49 + + sound/soc/mediatek/an7581/an7581-afe-pcm.c | 515 +++++ + sound/soc/mediatek/an7581/an7581-dai-etdm.c | 433 ++++ + sound/soc/mediatek/an7581/an7581-reg.h | 114 + + sound/soc/mediatek/an7581/an7581-wm8960.c | 161 ++ + sound/soc/mediatek/common/mtk-afe-fe-dai.c | 14 +- + sound/soc/mediatek/common/mtk-base-afe.h | 2 + + sound/soc/mediatek/mt2701/mt2701-afe-clock-ctrl.c | 61 +- + sound/soc/mediatek/mt2701/mt2701-afe-pcm.c | 4 +- + sound/soc/mediatek/mt2701/mt2701-cs42448.c | 4 +- + sound/soc/mediatek/mt2701/mt2701-wm8960.c | 4 +- + sound/soc/mediatek/mt6797/mt6797-afe-clk.c | 23 +- + sound/soc/mediatek/mt6797/mt6797-afe-pcm.c | 8 +- + sound/soc/mediatek/mt7986/mt7986-afe-pcm.c | 8 +- + sound/soc/mediatek/mt7986/mt7986-dai-etdm.c | 2 +- + sound/soc/mediatek/mt7986/mt7986-wm8960.c | 4 +- + sound/soc/mediatek/mt8173/mt8173-afe-pcm.c | 26 +- + sound/soc/mediatek/mt8183/mt8183-afe-clk.c | 31 +- + sound/soc/mediatek/mt8183/mt8183-afe-pcm.c | 21 +- + sound/soc/mediatek/mt8186/mt8186-afe-clk.c | 95 +- + sound/soc/mediatek/mt8186/mt8186-afe-gpio.c | 6 +- + sound/soc/mediatek/mt8186/mt8186-afe-pcm.c | 23 +- + sound/soc/mediatek/mt8188/mt8188-afe-clk.c | 115 +- + sound/soc/mediatek/mt8188/mt8188-afe-pcm.c | 36 +- + sound/soc/mediatek/mt8188/mt8188-audsys-clk.c | 12 +- + sound/soc/mediatek/mt8188/mt8188-dai-adda.c | 5 +- + sound/soc/mediatek/mt8188/mt8188-dai-dmic.c | 10 +- + sound/soc/mediatek/mt8189/mt8189-afe-clk.c | 134 +- + sound/soc/mediatek/mt8189/mt8189-afe-pcm.c | 61 +- + sound/soc/mediatek/mt8189/mt8189-dai-i2s.c | 18 +- + sound/soc/mediatek/mt8189/mt8189-dai-tdm.c | 19 +- + sound/soc/mediatek/mt8192/mt8192-afe-clk.c | 163 +- + sound/soc/mediatek/mt8192/mt8192-afe-gpio.c | 94 +- + sound/soc/mediatek/mt8192/mt8192-afe-pcm.c | 23 +- + sound/soc/mediatek/mt8192/mt8192-dai-adda.c | 44 +- + sound/soc/mediatek/mt8192/mt8192-dai-i2s.c | 20 +- + sound/soc/mediatek/mt8192/mt8192-dai-tdm.c | 18 +- + .../mediatek/mt8192/mt8192-mt6359-rt1015-rt5682.c | 37 +- + sound/soc/mediatek/mt8195/mt8195-afe-pcm.c | 2 +- + sound/soc/mediatek/mt8196/mt8196-afe-pcm.c | 14 +- + sound/soc/mxs/mxs-saif.c | 10 + + sound/soc/pxa/mmp-sspa.c | 6 + + sound/soc/pxa/pxa-ssp.c | 11 + + sound/soc/pxa/pxa2xx-i2s.c | 6 + + sound/soc/pxa/pxa2xx-pcm-lib.c | 9 +- + sound/soc/qcom/common.c | 61 +- + sound/soc/qcom/common.h | 5 +- + sound/soc/qcom/qdsp6/q6apm-lpass-dais.c | 2 + + sound/soc/qcom/qdsp6/q6asm-dai.c | 2 +- + sound/soc/qcom/qdsp6/q6dsp-lpass-ports.c | 54 + + sound/soc/qcom/sc8280xp.c | 132 +- + sound/soc/qcom/sdm845.c | 49 +- + sound/soc/renesas/rz-ssi.c | 9 + + sound/soc/renesas/siu_dai.c | 6 + + sound/soc/renesas/ssi.c | 13 + + sound/soc/rockchip/rk3399_gru_sound.c | 4 +- + sound/soc/rockchip/rockchip_i2s.c | 25 +- + sound/soc/rockchip/rockchip_i2s_tdm.c | 27 +- + sound/soc/rockchip/rockchip_pdm.c | 14 +- + sound/soc/rockchip/rockchip_sai.c | 31 +- + sound/soc/samsung/aries_wm8994.c | 14 +- + sound/soc/samsung/pcm.c | 19 +- + sound/soc/samsung/spdif.c | 8 +- + sound/soc/samsung/tm2_wm5110.c | 12 +- + sound/soc/sdca/Kconfig | 6 +- + sound/soc/sdca/sdca_asoc.c | 4 +- + sound/soc/sdca/sdca_class.c | 148 +- + sound/soc/sdca/sdca_class.h | 35 - + sound/soc/sdca/sdca_class_function.c | 50 +- + sound/soc/sdca/sdca_device.c | 4 + + sound/soc/sdca/sdca_fdl.c | 18 +- + sound/soc/sdca/sdca_functions.c | 15 +- + sound/soc/sdw_utils/Makefile | 4 +- + sound/soc/sdw_utils/soc_sdw_rt711.c | 1 + + sound/soc/sdw_utils/soc_sdw_rt_mf_sdca.c | 6 + + sound/soc/sdw_utils/soc_sdw_rt_sdca_jack_common.c | 8 + + sound/soc/sdw_utils/soc_sdw_senary_amp.c | 80 + + sound/soc/sdw_utils/soc_sdw_senary_dmic.c | 49 + + sound/soc/sdw_utils/soc_sdw_senary_sdca.c | 63 + + .../sdw_utils/soc_sdw_senary_sdca_jack_common.c | 197 ++ + sound/soc/sdw_utils/soc_sdw_utils.c | 384 ++++ + sound/soc/soc-component.c | 8 +- + sound/soc/soc-compress.c | 5 +- + sound/soc/soc-core.c | 175 +- + sound/soc/soc-dai.c | 418 +++- + sound/soc/soc-dapm.c | 9 +- + sound/soc/soc-generic-dmaengine-pcm.c | 10 +- + sound/soc/soc-internal.h | 39 + + sound/soc/soc-ops.c | 4 +- + sound/soc/soc-pcm.c | 144 +- + sound/soc/sof/amd/Kconfig | 1 + + sound/soc/sof/amd/acp-common.c | 17 + + sound/soc/sof/amd/acp-dsp-offset.h | 17 + + sound/soc/sof/amd/acp.c | 391 +++- + sound/soc/sof/amd/acp.h | 16 + + sound/soc/sof/amd/acp7x.h | 37 + + sound/soc/sof/amd/pci-acp7x.c | 2 + + sound/soc/sof/imx/imx-common.c | 1 + + sound/soc/sof/imx/imx8.c | 26 +- + sound/soc/sof/intel/apl.c | 1 + + sound/soc/sof/intel/cnl.c | 1 + + sound/soc/sof/intel/hda-dai.c | 26 +- + sound/soc/sof/intel/icl.c | 1 + + sound/soc/sof/intel/mtl.c | 1 + + sound/soc/sof/intel/skl.c | 1 + + sound/soc/sof/intel/tgl.c | 1 + + sound/soc/sof/ipc4-priv.h | 10 +- + sound/soc/sof/ipc4-topology.c | 65 +- + sound/soc/sof/mediatek/mt8186/mt8186.c | 2 +- + sound/soc/sof/mediatek/mt8195/mt8195.c | 2 +- + sound/soc/sof/topology.c | 3 +- + sound/soc/sprd/sprd-pcm-compress.c | 7 +- + sound/soc/sprd/sprd-pcm-dma.c | 10 +- + sound/soc/sti/uniperif_player.c | 29 +- + sound/soc/sti/uniperif_reader.c | 27 +- + sound/soc/stm/stm32_i2s.c | 12 + + sound/soc/stm/stm32_sai_sub.c | 15 + + sound/soc/stm/stm32_spdifrx.c | 2 +- + sound/soc/sunxi/sun4i-i2s.c | 13 + + sound/soc/sunxi/sun8i-codec.c | 13 + + sound/soc/tegra/tegra20_i2s.c | 10 + + sound/soc/tegra/tegra210_adx.c | 12 +- + sound/soc/tegra/tegra210_adx.h | 2 +- + sound/soc/tegra/tegra210_i2s.c | 13 + + sound/soc/tegra/tegra30_i2s.c | 10 + + sound/soc/ti/davinci-i2s.c | 16 +- + sound/soc/ti/davinci-mcasp.c | 17 +- + sound/soc/ti/omap-mcbsp.c | 14 +- + sound/soc/uniphier/aio-cpu.c | 9 + + sound/soc/uniphier/aio-dma.c | 5 +- + sound/soc/uniphier/aio.h | 2 +- + sound/soc/ux500/Kconfig | 23 +- + sound/soc/ux500/Makefile | 3 - + sound/soc/ux500/mop500.c | 167 -- + sound/soc/ux500/mop500_ab8500.c | 437 ---- + sound/soc/ux500/mop500_ab8500.h | 17 - + sound/soc/ux500/ux500_msp_dai.c | 33 +- + sound/soc/ux500/ux500_msp_i2s.c | 2 +- + sound/soc/xilinx/xlnx_formatter_pcm.c | 79 +- + sound/soc/xtensa/xtfpga-i2s.c | 12 +- + 501 files changed, 20889 insertions(+), 3581 deletions(-) + create mode 100644 Documentation/devicetree/bindings/sound/airoha,an7581-afe.yaml + create mode 100644 Documentation/devicetree/bindings/sound/airoha,an7581-wm8960.yaml + delete mode 100644 Documentation/devicetree/bindings/sound/davinci-evm-audio.txt + create mode 100644 Documentation/devicetree/bindings/sound/esstech,es9039q2m.yaml + create mode 100644 Documentation/devicetree/bindings/sound/nuvoton,nau8360.yaml + create mode 100644 Documentation/devicetree/bindings/sound/qcom,wcd9378-sdw.yaml + create mode 100644 Documentation/devicetree/bindings/sound/stericsson,ux500-msp-i2s.yaml + create mode 100644 Documentation/devicetree/bindings/sound/ti,da830-evm-audio.yaml + delete mode 100644 Documentation/devicetree/bindings/sound/ux500-mop500.txt + delete mode 100644 Documentation/devicetree/bindings/sound/ux500-msp.txt + create mode 100644 include/sound/sdca_class.h + create mode 100644 sound/soc/codecs/es9039q2m.c + create mode 100644 sound/soc/codecs/nau8360-dsp.c + create mode 100644 sound/soc/codecs/nau8360-dsp.h + create mode 100644 sound/soc/codecs/nau8360.c + create mode 100644 sound/soc/codecs/nau8360.h + create mode 100644 sound/soc/codecs/sn624x-sdca-sdw.c + create mode 100644 sound/soc/codecs/sn624x-sdca.h + create mode 100644 sound/soc/codecs/wcd9378-sdca.c + create mode 100644 sound/soc/codecs/wcd9378-sdca.h + create mode 100644 sound/soc/codecs/wcd9378-sdw.c + create mode 100644 sound/soc/mediatek/an7581/Makefile + create mode 100644 sound/soc/mediatek/an7581/an7581-afe-common.h + create mode 100644 sound/soc/mediatek/an7581/an7581-afe-pcm.c + create mode 100644 sound/soc/mediatek/an7581/an7581-dai-etdm.c + create mode 100644 sound/soc/mediatek/an7581/an7581-reg.h + create mode 100644 sound/soc/mediatek/an7581/an7581-wm8960.c + delete mode 100644 sound/soc/sdca/sdca_class.h + create mode 100644 sound/soc/sdw_utils/soc_sdw_senary_amp.c + create mode 100644 sound/soc/sdw_utils/soc_sdw_senary_dmic.c + create mode 100644 sound/soc/sdw_utils/soc_sdw_senary_sdca.c + create mode 100644 sound/soc/sdw_utils/soc_sdw_senary_sdca_jack_common.c + create mode 100644 sound/soc/soc-internal.h + create mode 100644 sound/soc/sof/amd/acp7x.h + delete mode 100644 sound/soc/ux500/mop500.c + delete mode 100644 sound/soc/ux500/mop500_ab8500.c + delete mode 100644 sound/soc/ux500/mop500_ab8500.h +Merging modules/modules-next (d6bfb4a05affc module: Bring includes in linux/kmod.h up to date) +$ git merge -m Merge branch 'modules-next' of https://git.kernel.org/pub/scm/linux/kernel/git/modules/linux.git modules/modules-next +Auto-merging fs/coredump.c +Auto-merging fs/nfsd/nfs4layouts.c +Auto-merging fs/nfsd/nfs4recover.c +Auto-merging kernel/reboot.c +Auto-merging kernel/time/jiffies.c +Merge made by the 'ort' strategy. + arch/x86/kernel/cpu/mce/dev-mcelog.c | 2 +- + drivers/block/drbd/drbd_nl.c | 1 + + drivers/greybus/svc_watchdog.c | 1 + + drivers/macintosh/windfarm_core.c | 1 + + drivers/pnp/pnpbios/core.c | 2 +- + drivers/video/fbdev/uvesafb.c | 1 + + fs/coredump.c | 2 +- + fs/nfs/cache_lib.c | 2 +- + fs/nfsd/nfs4layouts.c | 2 +- + fs/nfsd/nfs4recover.c | 1 + + fs/ocfs2/stackglue.c | 1 + + include/linux/kmod.h | 12 ++------ + include/linux/module_symbol.h | 7 +++-- + kernel/cgroup/cgroup-v1.c | 2 +- + kernel/module/kallsyms.c | 59 +++++++++++++++++++----------------- + kernel/module/kmod.c | 3 +- + kernel/power/process.c | 2 +- + kernel/reboot.c | 2 +- + kernel/time/jiffies.c | 1 + + kernel/umh.c | 2 +- + lib/kobject_uevent.c | 2 +- + net/bridge/br_stp_if.c | 2 +- + scripts/faddr2line | 2 +- + scripts/mod/modpost.h | 2 +- + security/keys/request_key.c | 2 +- + security/tomoyo/common.h | 2 +- + tools/perf/util/symbol.h | 4 +-- + 27 files changed, 63 insertions(+), 59 deletions(-) +Merging input/next (daae2ab46e0cb Input: gscps2 - drop busy-wait and manual interrupt pump on transmit) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/dtor/input.git input/next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../bindings/input/touchscreen/sis,9200-ts.yaml | 61 +++++ + .../bindings/input/touchscreen/sis_i2c.txt | 31 --- + MAINTAINERS | 2 +- + drivers/input/keyboard/snvs_pwrkey.c | 9 +- + drivers/input/keyboard/st-keyscan.c | 13 +- + drivers/input/matrix-keymap.c | 7 + + drivers/input/serio/gscps2.c | 293 +++++++++++---------- + 7 files changed, 225 insertions(+), 191 deletions(-) + create mode 100644 Documentation/devicetree/bindings/input/touchscreen/sis,9200-ts.yaml + delete mode 100644 Documentation/devicetree/bindings/input/touchscreen/sis_i2c.txt +Merging block/for-next (0183ac11c13ab Merge branch 'block-7.3' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux.git block/for-next +Auto-merging drivers/nvme/host/pci.c +Auto-merging io_uring/io-wq.c +Merge made by the 'ort' strategy. + arch/m68k/mac/config.c | 31 +- + block/bdev.c | 2 +- + block/blk-cgroup.c | 5 +- + block/blk-core.c | 7 +- + block/blk-mq.c | 63 ++- + block/blk-settings.c | 18 +- + block/blk-zoned.c | 872 ++++++++++++++++++++------------- + block/blk.h | 6 + + block/fops.c | 45 +- + block/mq-deadline.c | 2 +- + drivers/block/drbd/drbd_nl_gen.c | 4 - + drivers/block/rnull/configfs.rs | 44 +- + drivers/block/rnull/rnull.rs | 15 +- + drivers/block/swim.c | 457 +++++++++-------- + drivers/block/swim3.c | 8 +- + drivers/block/swim_asm.S | 340 +++++++------ + drivers/block/virtio_blk.c | 1 + + drivers/nvme/host/core.c | 63 ++- + drivers/nvme/host/fc.c | 2 +- + drivers/nvme/host/ioctl.c | 19 +- + drivers/nvme/host/multipath.c | 19 +- + drivers/nvme/host/nvme.h | 9 +- + drivers/nvme/host/pci.c | 2 + + drivers/nvme/host/sysfs.c | 4 +- + drivers/nvme/host/tcp.c | 39 +- + drivers/nvme/target/configfs.c | 9 +- + drivers/nvme/target/core.c | 35 +- + drivers/nvme/target/fabrics-cmd-auth.c | 2 +- + drivers/nvme/target/nvmet.h | 2 + + drivers/nvme/target/pci-epf.c | 3 + + include/linux/blkdev.h | 11 +- + include/linux/io_uring.h | 32 ++ + include/linux/io_uring/cmd.h | 34 +- + include/linux/io_uring_types.h | 2 + + io_uring/bpf_filter.c | 1 + + io_uring/cancel.c | 89 +++- + io_uring/cancel.h | 13 +- + io_uring/cmd_net.c | 19 +- + io_uring/fdinfo.c | 11 +- + io_uring/io-wq.c | 12 +- + io_uring/io_uring.c | 241 +++++++-- + io_uring/io_uring.h | 31 -- + io_uring/loop.c | 5 + + io_uring/notif.c | 2 + + io_uring/refs.h | 27 + + io_uring/rsrc.c | 42 +- + io_uring/rw.c | 4 + + io_uring/timeout.c | 18 + + io_uring/timeout.h | 2 + + io_uring/tw.c | 3 + + io_uring/uring_cmd.c | 23 +- + io_uring/uring_cmd.h | 2 +- + io_uring/zcrx.c | 9 +- + rust/kernel/block/mq/gen_disk.rs | 72 +-- + rust/kernel/block/mq/operations.rs | 16 +- + rust/kernel/block/mq/request.rs | 16 +- + rust/kernel/block/mq/tag_set.rs | 28 +- + tools/testing/selftests/ublk/kublk.c | 2 +- + 58 files changed, 1795 insertions(+), 1100 deletions(-) +$ git am -3 ../patches/0001-Revert-block-remove-bio_last_bvec_all.patch +Applying: Revert "block: remove bio_last_bvec_all" +$ git reset HEAD^ +Unstaged changes after reset: +M Documentation/block/biovecs.rst +M include/linux/bio.h +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master ac7093d359896] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux.git + Date: Wed Sep 30 13:14:56 2026 +0100 +$ git am -3 ../patches/0001-Fixup-for-blk-zoned-mismerge.patch +Applying: Fixup for blk-zoned mismerge +Using index info to reconstruct a base tree... +M block/blk-zoned.c +Falling back to patching base and 3-way merge... +No changes -- Patch already applied. +Merging device-mapper/for-next (bec2fa7bdf06d dm-verity: add DM_VERITY_VERIFY_ROOTHASH_SIG_FORCE) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/device-mapper/linux-dm.git device-mapper/for-next +Merge made by the 'ort' strategy. + Documentation/admin-guide/device-mapper/verity.rst | 5 +++ + drivers/md/Kconfig | 14 ++++++++ + drivers/md/dm-cache-metadata.c | 2 +- + drivers/md/dm-clone-metadata.c | 3 -- + drivers/md/dm-crypt.c | 23 +++++++------ + drivers/md/dm-delay.c | 3 +- + drivers/md/dm-exception-store.h | 2 +- + drivers/md/dm-mpath.c | 5 +-- + drivers/md/dm-pcache/backing_dev.c | 2 +- + drivers/md/dm-raid.c | 2 +- + drivers/md/dm-raid1.c | 2 +- + drivers/md/dm-unstripe.c | 2 +- + drivers/md/dm-vdo/block-map.c | 2 +- + drivers/md/dm-vdo/dm-vdo-target.c | 2 +- + drivers/md/dm-vdo/indexer/chapter-index.c | 15 ++++++--- + drivers/md/dm-vdo/indexer/delta-index.c | 17 ++++++++-- + drivers/md/dm-vdo/indexer/volume.c | 6 ---- + drivers/md/dm-verity-verify-sig.c | 4 +-- + drivers/md/dm-writecache.c | 39 ++++++++++------------ + drivers/md/dm-zone.c | 6 ++-- + drivers/md/persistent-data/dm-bitset.c | 5 +-- + drivers/md/persistent-data/dm-btree-remove.c | 2 +- + 22 files changed, 92 insertions(+), 71 deletions(-) +Merging libata/for-next (cfce1dc635041 ata: libata: Fix scsi_done() documentation) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/libata/linux libata/for-next +Auto-merging drivers/ata/ahci.c +Auto-merging drivers/ata/libahci.c +Auto-merging drivers/ata/libahci_platform.c +Auto-merging drivers/ata/libata-core.c +Auto-merging drivers/ata/libata-scsi.c +Auto-merging include/linux/libata.h +Merge made by the 'ort' strategy. + Documentation/driver-api/libata.rst | 4 +- + drivers/ata/Kconfig | 17 +- + drivers/ata/Makefile | 2 +- + drivers/ata/ahci.c | 19 +- + drivers/ata/ahci_brcm.c | 1 + + drivers/ata/ahci_ceva.c | 1 + + drivers/ata/ahci_da850.c | 6 +- + drivers/ata/ahci_qoriq.c | 1 + + drivers/ata/ahci_st.c | 78 +-- + drivers/ata/ata_generic.c | 16 + + drivers/ata/libahci.c | 29 +- + drivers/ata/libahci_platform.c | 4 + + drivers/ata/libata-core.c | 81 ++- + drivers/ata/libata-scsi.c | 7 +- + drivers/ata/libata-sff.c | 13 +- + drivers/ata/pata_arasan_cf.c | 6 +- + drivers/ata/pata_cswarp.c | 183 +++++++ + drivers/ata/pata_parport/pata_parport.c | 9 +- + drivers/ata/sata_dwc_460ex.c | 5 +- + drivers/ata/sata_fsl.c | 26 +- + drivers/ata/sata_inic162x.c | 903 -------------------------------- + drivers/ata/sata_qstor.c | 8 +- + include/linux/libata.h | 1 + + 23 files changed, 416 insertions(+), 1004 deletions(-) + create mode 100644 drivers/ata/pata_cswarp.c + delete mode 100644 drivers/ata/sata_inic162x.c +Merging pcmcia/pcmcia-next (b3c26ea81ccc5 pcmcia: remove obsolete host controller drivers) +$ git merge -m Merge branch 'pcmcia-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brodo/linux.git pcmcia/pcmcia-next +Already up to date. +Merging mmc/next (c2063613bceaa mmc: Merge branch fixes into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/ulfh/mmc.git mmc/next +Auto-merging CREDITS +Auto-merging Documentation/devicetree/bindings/mmc/qcom,sdhci-msm.yaml +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + CREDITS | 5 + + .../bindings/mmc/amlogic,meson-gx-mmc.yaml | 29 +- + .../devicetree/bindings/mmc/cdns,sd6hc.yaml | 105 + + .../bindings/mmc/mmc-controller-common.yaml | 10 + + .../devicetree/bindings/mmc/qcom,sdhci-msm.yaml | 2 + + Documentation/driver-api/mmc/mmc-async-req.rst | 64 +- + Documentation/driver-api/mmc/mmc-tools.rst | 1 + + MAINTAINERS | 12 +- + drivers/memstick/core/ms_block.c | 2 +- + drivers/memstick/host/r592.c | 2 +- + drivers/mmc/core/block.c | 2 +- + drivers/mmc/core/core.c | 8 +- + drivers/mmc/core/core.h | 10 +- + drivers/mmc/core/crypto.c | 2 +- + drivers/mmc/core/host.c | 39 +- + drivers/mmc/core/mmc.c | 2 +- + drivers/mmc/core/mmc_ops.c | 22 +- + drivers/mmc/core/sdio_cis.c | 5 +- + drivers/mmc/host/Kconfig | 31 - + drivers/mmc/host/Makefile | 2 +- + drivers/mmc/host/atmel-mci.c | 5 +- + drivers/mmc/host/bcm2835.c | 2 +- + drivers/mmc/host/cavium-thunderx.c | 5 +- + drivers/mmc/host/davinci_mmc.c | 6 +- + drivers/mmc/host/dw_mmc-rockchip.c | 2 +- + drivers/mmc/host/dw_mmc.c | 316 ++- + drivers/mmc/host/dw_mmc.h | 18 +- + drivers/mmc/host/meson-gx-mmc.c | 31 +- + drivers/mmc/host/mvsdio.c | 2 +- + drivers/mmc/host/rtsx_usb_sdmmc.c | 5 + + .../host/{sdhci-cadence.c => sdhci-cadence-core.c} | 303 ++- + drivers/mmc/host/sdhci-cadence-phy-v6.c | 975 ++++++++ + drivers/mmc/host/sdhci-cadence.h | 118 + + drivers/mmc/host/sdhci-msm.c | 34 +- + drivers/mmc/host/sdhci-of-arasan.c | 4 +- + drivers/mmc/host/sdhci-of-dwcmshc.c | 13 + + drivers/mmc/host/sdhci-of-esdhc.c | 2 +- + drivers/mmc/host/sdhci-omap.c | 2 +- + drivers/mmc/host/sdhci-pci-core.c | 8 +- + drivers/mmc/host/sdhci-pxav3.c | 13 +- + drivers/mmc/host/sdhci.h | 2 +- + drivers/mmc/host/sunplus-mmc.c | 5 + + drivers/mmc/host/ushc.c | 5 +- + drivers/mmc/host/vub300.c | 2493 -------------------- + drivers/staging/greybus/sdio.c | 11 +- + include/linux/mmc/host.h | 6 +- + 46 files changed, 1823 insertions(+), 2918 deletions(-) + create mode 100644 Documentation/devicetree/bindings/mmc/cdns,sd6hc.yaml + rename drivers/mmc/host/{sdhci-cadence.c => sdhci-cadence-core.c} (67%) + create mode 100644 drivers/mmc/host/sdhci-cadence-phy-v6.c + create mode 100644 drivers/mmc/host/sdhci-cadence.h + delete mode 100644 drivers/mmc/host/vub300.c +Merging mfd/for-mfd-next (319633b06ff2b dt-bindings: mfd: qcom,tcsr: Add compatible for MSM8952) +$ git merge -m Merge branch 'for-mfd-next' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/mfd.git mfd/for-mfd-next +Auto-merging MAINTAINERS +Auto-merging drivers/gpio/Kconfig +Auto-merging drivers/gpio/Makefile +Merge made by the 'ort' strategy. + .../devicetree/bindings/input/cpcap-pwrbutton.txt | 20 - + .../bindings/input/motorola,cpcap-pwrbutton.yaml | 32 ++ + .../leds/backlight/ti,lm3533-backlight.yaml | 69 +++ + .../devicetree/bindings/leds/ti,lm3533-leds.yaml | 67 +++ + .../devicetree/bindings/leds/ti,lm3533.yaml | 169 +++++++ + .../devicetree/bindings/mfd/motorola,cpcap.yaml | 408 +++++++++++++++ + .../devicetree/bindings/mfd/motorola-cpcap.txt | 78 --- + .../devicetree/bindings/mfd/qcom,spmi-pmic.yaml | 1 + + .../devicetree/bindings/mfd/qcom,tcsr.yaml | 1 + + .../devicetree/bindings/mfd/rohm,bd71815-pmic.yaml | 9 +- + .../devicetree/bindings/mfd/rohm,bd71828-pmic.yaml | 9 +- + .../devicetree/bindings/mfd/rohm,bd72720-pmic.yaml | 29 +- + .../devicetree/bindings/mfd/rohm,bd73800-pmic.yaml | 218 ++++++++ + .../devicetree/bindings/mfd/rohm,pmic-pins.yaml | 72 +++ + .../devicetree/bindings/mfd/stericsson,ab8500.yaml | 150 +++++- + .../bindings/mfd/ti,keystone-devctrl.yaml | 90 ++++ + .../devicetree/bindings/mfd/ti,tps61050.yaml | 92 ++++ + .../devicetree/bindings/mfd/ti,tps65910.yaml | 8 + + .../devicetree/bindings/mfd/ti,twl6040.yaml | 156 ++++++ + .../bindings/mfd/ti-keystone-devctrl.txt | 19 - + Documentation/devicetree/bindings/mfd/tps6105x.txt | 62 --- + Documentation/devicetree/bindings/mfd/twl6040.txt | 67 --- + .../bindings/regulator/rohm,bd73800-regulator.yaml | 99 ++++ + MAINTAINERS | 2 + + drivers/clk/clk-bd718x7.c | 8 + + drivers/gpio/Kconfig | 12 + + drivers/gpio/Makefile | 1 + + drivers/gpio/gpio-bd73800.c | 209 ++++++++ + drivers/iio/light/lm3533-als.c | 185 +++---- + drivers/leds/leds-lm3533.c | 121 ++--- + drivers/mfd/Kconfig | 17 +- + drivers/mfd/ab8500-core.c | 2 +- + drivers/mfd/cs42l43-i2c.c | 10 + + drivers/mfd/cs42l43-sdw.c | 9 + + drivers/mfd/cs42l43.c | 30 +- + drivers/mfd/cs42l43.h | 1 + + drivers/mfd/da903x.c | 34 +- + drivers/mfd/da9062-core.c | 24 +- + drivers/mfd/da9150-core.c | 8 +- + drivers/mfd/intel-lpss.c | 24 +- + drivers/mfd/intel_quark_i2c_gpio.c | 52 +- + drivers/mfd/iqs62x.c | 8 +- + drivers/mfd/khadas-mcu.c | 104 +++- + drivers/mfd/lm3533-core.c | 386 ++++++-------- + drivers/mfd/lm3533-ctrlbank.c | 33 +- + drivers/mfd/max77843.c | 8 +- + drivers/mfd/motorola-cpcap.c | 143 +++--- + drivers/mfd/mt6360-core.c | 8 +- + drivers/mfd/rk8xx-core.c | 2 +- + drivers/mfd/rk8xx-i2c.c | 4 +- + drivers/mfd/rn5t618.c | 8 +- + drivers/mfd/rohm-bd71828.c | 147 +++++- + drivers/mfd/ti_am335x_tscadc.c | 8 +- + drivers/mfd/tps6586x.c | 8 +- + drivers/mfd/twl-core.c | 31 +- + drivers/mfd/wcd934x.c | 8 +- + drivers/mfd/wm831x-auxadc.c | 2 +- + drivers/regulator/Kconfig | 4 +- + drivers/regulator/bd71828-regulator.c | 558 ++++++++++++++++++++- + drivers/rtc/rtc-bd70528.c | 8 + + drivers/thermal/khadas_mcu_fan.c | 108 +++- + drivers/video/backlight/lm3533_bl.c | 188 +++---- + include/linux/mfd/cs42l43-regs.h | 1 + + include/linux/mfd/da9150/core.h | 1 + + include/linux/mfd/khadas-mcu.h | 19 +- + include/linux/mfd/lm3533.h | 75 +-- + include/linux/mfd/motorola-cpcap.h | 7 + + include/linux/mfd/rohm-bd73800.h | 306 +++++++++++ + include/linux/mfd/rohm-generic.h | 1 + + 69 files changed, 3747 insertions(+), 1111 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/input/cpcap-pwrbutton.txt + create mode 100644 Documentation/devicetree/bindings/input/motorola,cpcap-pwrbutton.yaml + create mode 100644 Documentation/devicetree/bindings/leds/backlight/ti,lm3533-backlight.yaml + create mode 100644 Documentation/devicetree/bindings/leds/ti,lm3533-leds.yaml + create mode 100644 Documentation/devicetree/bindings/leds/ti,lm3533.yaml + create mode 100644 Documentation/devicetree/bindings/mfd/motorola,cpcap.yaml + delete mode 100644 Documentation/devicetree/bindings/mfd/motorola-cpcap.txt + create mode 100644 Documentation/devicetree/bindings/mfd/rohm,bd73800-pmic.yaml + create mode 100644 Documentation/devicetree/bindings/mfd/rohm,pmic-pins.yaml + create mode 100644 Documentation/devicetree/bindings/mfd/ti,keystone-devctrl.yaml + create mode 100644 Documentation/devicetree/bindings/mfd/ti,tps61050.yaml + create mode 100644 Documentation/devicetree/bindings/mfd/ti,twl6040.yaml + delete mode 100644 Documentation/devicetree/bindings/mfd/ti-keystone-devctrl.txt + delete mode 100644 Documentation/devicetree/bindings/mfd/tps6105x.txt + delete mode 100644 Documentation/devicetree/bindings/mfd/twl6040.txt + create mode 100644 Documentation/devicetree/bindings/regulator/rohm,bd73800-regulator.yaml + create mode 100644 drivers/gpio/gpio-bd73800.c + create mode 100644 include/linux/mfd/rohm-bd73800.h +Merging backlight/for-backlight-next (5e1631df673f3 backlight: ili9320/ktd253: Fix typos in comments) +$ git merge -m Merge branch 'for-backlight-next' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/backlight.git backlight/for-backlight-next +Merge made by the 'ort' strategy. + drivers/video/backlight/ili9320.c | 2 +- + drivers/video/backlight/ktd253-backlight.c | 2 +- + 2 files changed, 2 insertions(+), 2 deletions(-) +Merging battery/for-next (4fc88ba435dad power: supply: cros_charge-control: adopt EC charge state on probe) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-power-supply.git battery/for-next +Auto-merging drivers/power/supply/max17042_battery.c +Auto-merging include/linux/power_supply.h +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-class-power | 39 ++++- + .../ABI/testing/sysfs-class-power-ltc4162l | 2 + + Documentation/ABI/testing/sysfs-class-power-mp2629 | 2 +- + Documentation/ABI/testing/sysfs-class-power-rt9467 | 2 + + Documentation/ABI/testing/sysfs-class-power-rt9471 | 2 + + .../devicetree/bindings/power/supply/bq27xxx.yaml | 2 +- + drivers/power/reset/keystone-reset.c | 3 +- + drivers/power/supply/bq24190_charger.c | 71 +++++++++- + drivers/power/supply/bq24257_charger.c | 43 +++++- + drivers/power/supply/bq2515x_charger.c | 15 +- + drivers/power/supply/bq25630_charger.c | 84 +++++++++++ + drivers/power/supply/bq25890_charger.c | 2 +- + drivers/power/supply/bq27xxx_battery.c | 18 ++- + drivers/power/supply/bq27xxx_battery_i2c.c | 2 + + drivers/power/supply/charger-manager.c | 2 +- + drivers/power/supply/cros_charge-control.c | 157 ++++++++++++++++++--- + drivers/power/supply/da9030_battery.c | 2 +- + drivers/power/supply/ltc4162-l-charger.c | 56 +++++++- + drivers/power/supply/max17042_battery.c | 2 +- + drivers/power/supply/mm8013.c | 1 + + drivers/power/supply/power_supply_sysfs.c | 17 +++ + drivers/power/supply/qcom_smbx.c | 98 ++++++++++--- + drivers/power/supply/rt9467-charger.c | 54 ++++++- + drivers/power/supply/rt9471.c | 54 ++++++- + drivers/power/supply/s2mu005-battery.c | 4 +- + drivers/power/supply/sbs-charger.c | 4 +- + drivers/power/supply/smb347-charger.c | 4 +- + include/linux/power/bq27xxx_battery.h | 1 + + include/linux/power_supply.h | 18 ++- + 29 files changed, 671 insertions(+), 90 deletions(-) +Merging regulator/for-next (26f25a65cda68 Merge regulator/for-7.4 into regulator-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/regulator.git regulator/for-next +Auto-merging Documentation/devicetree/bindings/vendor-prefixes.yaml +Auto-merging MAINTAINERS +Auto-merging drivers/regulator/Kconfig +Merge made by the 'ort' strategy. + .../regulator/maxim,max77620-regulator.yaml | 12 +- + .../bindings/regulator/maxim,max77826.yaml | 4 +- + .../bindings/regulator/maxim,max77838.yaml | 4 +- + .../regulator/mediatek,mt6380-regulator.yaml | 117 ++++ + .../devicetree/bindings/regulator/mps,mp5416.yaml | 38 +- + .../devicetree/bindings/regulator/mps,mp886x.yaml | 30 +- + .../devicetree/bindings/regulator/mps,mpq4210.yaml | 68 ++ + .../devicetree/bindings/regulator/mps,mpq7920.yaml | 46 +- + .../bindings/regulator/mt6380-regulator.txt | 89 --- + .../bindings/regulator/nexperia,nex10000ub.yaml | 69 ++ + .../devicetree/bindings/regulator/nxp,pf0900.yaml | 2 +- + .../bindings/regulator/onnn,fan53880.yaml | 4 +- + .../bindings/regulator/pwm-regulator.yaml | 2 +- + .../bindings/regulator/qcom,rpmh-regulator.yaml | 16 + + .../regulator/richtek,rtmv20-regulator.yaml | 6 +- + .../bindings/regulator/silergy,sy8824x.yaml | 8 +- + .../bindings/regulator/silergy,sy8827n.yaml | 4 +- + .../devicetree/bindings/regulator/ti,tps65023.yaml | 90 +++ + .../devicetree/bindings/regulator/ti,tps6586x.yaml | 196 ++++++ + .../devicetree/bindings/regulator/tps65023.txt | 60 -- + .../devicetree/bindings/regulator/tps6586x.txt | 135 ---- + .../bindings/soc/mediatek/mediatek,pwrap.yaml | 3 + + .../devicetree/bindings/vendor-prefixes.yaml | 2 + + MAINTAINERS | 6 + + drivers/regulator/88pm886-regulator.c | 26 + + drivers/regulator/Kconfig | 21 +- + drivers/regulator/Makefile | 2 + + drivers/regulator/ab8500-ext.c | 2 +- + drivers/regulator/ab8500.c | 748 +++++++++++++++++++-- + drivers/regulator/axp20x-regulator.c | 6 +- + drivers/regulator/core.c | 10 +- + drivers/regulator/fixed.c | 4 + + drivers/regulator/mp886x.c | 34 +- + drivers/regulator/mpq4210.c | 243 +++++++ + drivers/regulator/nex10000ub-regulator.c | 147 ++++ + drivers/regulator/pca9450-regulator.c | 81 ++- + drivers/regulator/pf1550-regulator.c | 19 +- + drivers/regulator/pv88080-regulator.c | 4 +- + drivers/regulator/qcom-rpmh-regulator.c | 55 +- + drivers/regulator/rt6190-regulator.c | 18 +- + drivers/regulator/tps6105x-regulator.c | 2 +- + drivers/regulator/tps65185.c | 6 +- + include/linux/mfd/88pm886.h | 7 + + include/linux/regulator/driver.h | 2 +- + include/linux/regulator/pca9450.h | 4 +- + kernel/irq/irqdesc.c | 10 +- + kernel/softirq.c | 4 +- + 47 files changed, 1979 insertions(+), 487 deletions(-) + create mode 100644 Documentation/devicetree/bindings/regulator/mediatek,mt6380-regulator.yaml + create mode 100644 Documentation/devicetree/bindings/regulator/mps,mpq4210.yaml + delete mode 100644 Documentation/devicetree/bindings/regulator/mt6380-regulator.txt + create mode 100644 Documentation/devicetree/bindings/regulator/nexperia,nex10000ub.yaml + create mode 100644 Documentation/devicetree/bindings/regulator/ti,tps65023.yaml + create mode 100644 Documentation/devicetree/bindings/regulator/ti,tps6586x.yaml + delete mode 100644 Documentation/devicetree/bindings/regulator/tps65023.txt + delete mode 100644 Documentation/devicetree/bindings/regulator/tps6586x.txt + create mode 100644 drivers/regulator/mpq4210.c + create mode 100644 drivers/regulator/nex10000ub-regulator.c +Merging security/next (881f19c2ffbc7 lsm: remove redundant NULL check in security_init()) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git security/next +Auto-merging fs/namei.c +Auto-merging fs/namespace.c +Auto-merging include/linux/lsm_hook_defs.h +CONFLICT (content): Merge conflict in include/linux/lsm_hook_defs.h +Auto-merging include/linux/ns/ns_common_types.h +CONFLICT (content): Merge conflict in include/linux/ns/ns_common_types.h +Auto-merging include/linux/sched.h +Auto-merging include/linux/security.h +CONFLICT (content): Merge conflict in include/linux/security.h +Auto-merging security/security.c +Auto-merging security/selinux/hooks.c +Auto-merging security/smack/smack_lsm.c +Auto-merging tools/testing/selftests/bpf/progs/lsm.c +Auto-merging tools/testing/selftests/landlock/fs_test.c +Resolved 'include/linux/lsm_hook_defs.h' using previous resolution. +Resolved 'include/linux/ns/ns_common_types.h' using previous resolution. +Resolved 'include/linux/security.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master fb6b2da489e27] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git +$ git diff -M --stat --summary HEAD^.. + Documentation/admin-guide/LSM/index.rst | 20 +- + fs/cachefiles/security.c | 4 +- + fs/namei.c | 18 +- + fs/namespace.c | 3 +- + include/linux/cred.h | 17 +- + include/linux/lsm_audit.h | 7 +- + include/linux/lsm_hook_defs.h | 28 +-- + include/linux/lsm_hooks.h | 1 + + include/linux/ns/ns_common_types.h | 3 + + include/linux/sched.h | 8 +- + include/linux/security.h | 78 +++++--- + include/uapi/linux/nsfs.h | 1 + + init/init_task.c | 2 +- + kernel/auditsc.c | 5 +- + kernel/cred.c | 5 +- + kernel/nscommon.c | 17 +- + kernel/nsproxy.c | 6 + + rust/kernel/security.rs | 3 +- + security/lsm_audit.c | 8 +- + security/lsm_init.c | 9 +- + security/security.c | 115 +++++++++-- + security/selinux/hooks.c | 25 ++- + security/smack/smack_lsm.c | 9 +- + tools/testing/selftests/bpf/prog_tests/test_lsm.c | 231 ++++++++++++++++++++++ + tools/testing/selftests/bpf/progs/lsm.c | 79 ++++++++ + tools/testing/selftests/landlock/fs_test.c | 10 +- + 26 files changed, 600 insertions(+), 112 deletions(-) +$ git am -3 ../patches/0001-security-Fix-up-mismerge-and-additional-semantic-iss.patch +Applying: security: Fix up mismerge and additional semantic issues with vfs-brauner +Using index info to reconstruct a base tree... +M include/linux/lsm_hook_defs.h +M security/security.c +Falling back to patching base and 3-way merge... +Auto-merging security/security.c +$ git reset HEAD^ +Unstaged changes after reset: +M include/linux/security.h +M security/apparmor/af_unix.c +M security/apparmor/file.c +M security/apparmor/include/af_unix.h +M security/apparmor/include/file.h +M security/apparmor/include/net.h +M security/apparmor/lsm.c +M security/apparmor/net.c +M security/security.c +M security/selinux/hooks.c +M security/smack/smack_lsm.c +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master 635e4a5f0833b] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git + Date: Wed Sep 30 13:15:18 2026 +0100 +$ git am -3 ../patches/0001-security-More-merge-fixup-from-the-constification-of.patch +Applying: security: More merge fixup from the constification of idmap +$ git reset HEAD^ +Unstaged changes after reset: +M security/security.c +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master 65c77b8cc47d9] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/lsm.git + Date: Wed Sep 30 13:15:18 2026 +0100 +Merging apparmor/apparmor-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'apparmor-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jj/linux-apparmor apparmor/apparmor-next +Already up to date. +Merging integrity/next-integrity (1dc317438308c integrity: Replace integrity_audit_message() with integrity_audit_msg()) +$ git merge -m Merge branch 'next-integrity' of https://git.kernel.org/pub/scm/linux/kernel/git/zohar/linux-integrity integrity/next-integrity +Auto-merging security/integrity/evm/evm_main.c +Auto-merging security/integrity/ima/ima_api.c +Auto-merging security/integrity/ima/ima_appraise.c +Auto-merging security/integrity/ima/ima_main.c +Auto-merging security/integrity/ima/ima_policy.c +Merge made by the 'ort' strategy. + security/integrity/evm/evm_main.c | 9 +++++---- + security/integrity/ima/ima_api.c | 8 ++++---- + security/integrity/ima/ima_appraise.c | 6 +++--- + security/integrity/ima/ima_fs.c | 7 ++++--- + security/integrity/ima/ima_init.c | 2 +- + security/integrity/ima/ima_main.c | 13 +++++++------ + security/integrity/ima/ima_policy.c | 11 ++++++----- + security/integrity/ima/ima_queue.c | 2 +- + security/integrity/ima/ima_queue_keys.c | 8 ++++---- + security/integrity/ima/ima_template_lib.c | 2 +- + security/integrity/integrity.h | 18 +++--------------- + security/integrity/integrity_audit.c | 12 ++---------- + 12 files changed, 41 insertions(+), 57 deletions(-) +Merging selinux/next (f2b37cf384b01 Automated merge of 'dev' into 'next') +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/selinux.git selinux/next +Auto-merging security/selinux/selinuxfs.c +Merge made by the 'ort' strategy. + security/selinux/avc.c | 42 ++++++++++++++++------------- + security/selinux/include/avc.h | 7 ++--- + security/selinux/selinuxfs.c | 59 ++++++++++++++++++++++++----------------- + security/selinux/ss/mls.c | 38 -------------------------- + security/selinux/ss/mls.h | 3 --- + security/selinux/ss/mls_types.h | 3 --- + security/selinux/ss/policydb.c | 51 ++++++++++++++++++++++++++--------- + 7 files changed, 101 insertions(+), 102 deletions(-) +Merging smack/next (fedc88e38ce97 smack: fix cred UAF in smack_file_send_sigiotask()) +$ git merge -m Merge branch 'next' of https://github.com/cschaufler/smack-next smack/next +Already up to date. +Merging tomoyo/master (72d3fcf802c45 Linux 7.3-rc5) +$ git merge -m Merge branch 'master' of git://git.code.sf.net/p/tomoyo/tomoyo.git tomoyo/master +Already up to date. +Merging tpmdd-tpm/for-next-tpm (015fb29a74834 tpm: Disable TPM on null key name mismatch) +$ git merge -m Merge branch 'for-next-tpm' of https://git.kernel.org/pub/scm/linux/kernel/git/jarkko/linux-tpmdd.git tpmdd-tpm/for-next-tpm +Merge made by the 'ort' strategy. + drivers/char/tpm/tpm-interface.c | 2 +- + drivers/char/tpm/tpm2-cmd.c | 6 ++++-- + drivers/char/tpm/tpm2-sessions.c | 1 + + 3 files changed, 6 insertions(+), 3 deletions(-) +Merging tpmdd-keys/for-next-keys (e7ac8fd885298 assoc_array: discard shortcut when collapsing a leaf-only node) +$ git merge -m Merge branch 'for-next-keys' of https://git.kernel.org/pub/scm/linux/kernel/git/jarkko/linux-tpmdd.git tpmdd-keys/for-next-keys +Auto-merging security/keys/trusted-keys/trusted_tpm1.c +Merge made by the 'ort' strategy. + lib/assoc_array.c | 36 ++++++++++++-------- + security/keys/persistent.c | 55 ++++++++++++++++++------------- + security/keys/trusted-keys/trusted_tpm1.c | 4 ++- + security/keys/trusted-keys/trusted_tpm2.c | 4 ++- + 4 files changed, 60 insertions(+), 39 deletions(-) +Merging watchdog/watchdog-next (8b5a9f09037e3 dt-bindings: watchdog: apple,wdt: Add t8140 compatible) +$ git merge -m Merge branch 'watchdog-next' of https://git.kernel.org/pub/scm/linux/kernel/git/groeck/linux-staging.git watchdog/watchdog-next +Auto-merging drivers/watchdog/hpwdt.c +Auto-merging drivers/watchdog/mtk_wdt.c +Merge made by the 'ort' strategy. + .../devicetree/bindings/watchdog/apple,wdt.yaml | 1 + + .../bindings/watchdog/mediatek,mtk-wdt.yaml | 2 + + .../bindings/watchdog/renesas,r9a09g057-wdt.yaml | 29 ++++- + .../devicetree/bindings/watchdog/samsung-wdt.yaml | 23 +++- + Documentation/watchdog/watchdog-parameters.rst | 2 + + drivers/watchdog/acquirewdt.c | 3 +- + drivers/watchdog/advantech_ec_wdt.c | 3 +- + drivers/watchdog/advantechwdt.c | 5 +- + drivers/watchdog/airoha_wdt.c | 5 +- + drivers/watchdog/alim1535_wdt.c | 5 +- + drivers/watchdog/alim7101_wdt.c | 5 +- + drivers/watchdog/arm_smc_wdt.c | 3 +- + drivers/watchdog/armada_37xx_wdt.c | 3 +- + drivers/watchdog/aspeed_wdt.c | 3 +- + drivers/watchdog/at91rm9200_wdt.c | 5 +- + drivers/watchdog/at91sam9_wdt.c | 5 +- + drivers/watchdog/atcwdt200_wdt.c | 5 +- + drivers/watchdog/ath79_wdt.c | 5 +- + drivers/watchdog/bcm2835_wdt.c | 3 +- + drivers/watchdog/bcm47xx_wdt.c | 5 +- + drivers/watchdog/bcm7038_wdt.c | 3 +- + drivers/watchdog/booke_wdt.c | 3 +- + drivers/watchdog/cadence_wdt.c | 13 +-- + drivers/watchdog/cgbc_wdt.c | 7 +- + drivers/watchdog/da9052_wdt.c | 5 +- + drivers/watchdog/da9055_wdt.c | 3 +- + drivers/watchdog/da9062_wdt.c | 8 +- + drivers/watchdog/davinci_wdt.c | 7 +- + drivers/watchdog/db8500_wdt.c | 5 +- + drivers/watchdog/dw_wdt.c | 3 +- + drivers/watchdog/ebc-c384_wdt.c | 5 +- + drivers/watchdog/eurotechwdt.c | 3 +- + drivers/watchdog/exar_wdt.c | 5 +- + drivers/watchdog/f71808e_wdt.c | 9 +- + drivers/watchdog/gef_wdt.c | 3 +- + drivers/watchdog/geodewdt.c | 5 +- + drivers/watchdog/gpio_wdt.c | 3 +- + drivers/watchdog/hpwdt.c | 3 +- + drivers/watchdog/i6300esb.c | 28 +++-- + drivers/watchdog/iTCO_wdt.c | 5 +- + drivers/watchdog/ib700wdt.c | 5 +- + drivers/watchdog/ibmasr.c | 3 +- + drivers/watchdog/ie6xx_wdt.c | 5 +- + drivers/watchdog/imgpdc_wdt.c | 5 +- + drivers/watchdog/imx2_wdt.c | 5 +- + drivers/watchdog/imx7ulp_wdt.c | 3 +- + drivers/watchdog/imx_sc_wdt.c | 3 +- + drivers/watchdog/indydog.c | 3 +- + drivers/watchdog/intel_oc_wdt.c | 5 +- + drivers/watchdog/it87_wdt.c | 7 +- + drivers/watchdog/jz4740_wdt.c | 7 +- + drivers/watchdog/keembay_wdt.c | 13 +-- + drivers/watchdog/kempld_wdt.c | 7 +- + drivers/watchdog/lenovo_se10_wdt.c | 5 +- + drivers/watchdog/lenovo_se30_wdt.c | 5 +- + drivers/watchdog/lenovo_se30g2_se60_wdt.c | 5 +- + drivers/watchdog/loongson1_wdt.c | 5 +- + drivers/watchdog/lpc18xx_wdt.c | 5 +- + drivers/watchdog/ma35d1_wdt.c | 3 +- + drivers/watchdog/max63xx_wdt.c | 9 +- + drivers/watchdog/max77620_wdt.c | 3 +- + drivers/watchdog/mena21_wdt.c | 3 +- + drivers/watchdog/menf21bmc_wdt.c | 3 +- + drivers/watchdog/menz69_wdt.c | 3 +- + drivers/watchdog/meson_gxbb_wdt.c | 5 +- + drivers/watchdog/meson_wdt.c | 3 +- + drivers/watchdog/mixcomwd.c | 3 +- + drivers/watchdog/mpc8xxx_wdt.c | 5 +- + drivers/watchdog/msc313e_wdt.c | 12 +-- + drivers/watchdog/mt7621_wdt.c | 3 +- + drivers/watchdog/mtk_wdt.c | 119 ++++++++++++++++++--- + drivers/watchdog/nct6694_wdt.c | 3 +- + drivers/watchdog/ni903x_wdt.c | 5 +- + drivers/watchdog/nic7018_wdt.c | 5 +- + drivers/watchdog/nv_tco.c | 5 +- + drivers/watchdog/octeon-wdt-main.c | 5 +- + drivers/watchdog/of_xilinx_wdt.c | 8 +- + drivers/watchdog/omap_wdt.c | 3 +- + drivers/watchdog/orion_wdt.c | 3 +- + drivers/watchdog/pc87413_wdt.c | 9 +- + drivers/watchdog/pcwd.c | 5 +- + drivers/watchdog/pcwd_pci.c | 7 +- + drivers/watchdog/pcwd_usb.c | 7 +- + drivers/watchdog/pika_wdt.c | 5 +- + drivers/watchdog/pm8916_wdt.c | 8 +- + drivers/watchdog/pnx4008_wdt.c | 5 +- + drivers/watchdog/pseries-wdt.c | 7 +- + drivers/watchdog/rc32434_wdt.c | 5 +- + drivers/watchdog/renesas_wdt.c | 3 +- + drivers/watchdog/rn5t618_wdt.c | 3 +- + drivers/watchdog/rt2880_wdt.c | 3 +- + drivers/watchdog/rti_wdt.c | 5 +- + drivers/watchdog/rzg2l_wdt.c | 3 +- + drivers/watchdog/rzv2h_wdt.c | 3 +- + drivers/watchdog/s32g_wdt.c | 5 +- + drivers/watchdog/s3c2410_wdt.c | 23 +++- + drivers/watchdog/sama5d4_wdt.c | 5 +- + drivers/watchdog/sbc60xxwdt.c | 5 +- + drivers/watchdog/sbc7240_wdt.c | 5 +- + drivers/watchdog/sbc8360.c | 3 +- + drivers/watchdog/sbc_epx_c3.c | 3 +- + drivers/watchdog/sbsa_gwdt.c | 24 ++++- + drivers/watchdog/sc1200wdt.c | 3 +- + drivers/watchdog/sch311x_wdt.c | 5 +- + drivers/watchdog/shwdt.c | 7 +- + drivers/watchdog/simatic-ipc-wdt.c | 3 +- + drivers/watchdog/sl28cpld_wdt.c | 3 +- + drivers/watchdog/smsc37b787_wdt.c | 3 +- + drivers/watchdog/softdog.c | 5 +- + drivers/watchdog/sp5100_tco.c | 68 ++++++++++-- + drivers/watchdog/sp805_wdt.c | 14 +-- + drivers/watchdog/starfive-wdt.c | 7 +- + drivers/watchdog/stmp3xxx_rtc_wdt.c | 13 +-- + drivers/watchdog/stpmic1_wdt.c | 3 +- + drivers/watchdog/sun4v_wdt.c | 5 +- + drivers/watchdog/sunplus_wdt.c | 3 +- + drivers/watchdog/sunxi_wdt.c | 3 +- + drivers/watchdog/tegra_wdt.c | 5 +- + drivers/watchdog/tqmx86_wdt.c | 3 +- + drivers/watchdog/ts4800_wdt.c | 3 +- + drivers/watchdog/twl4030_wdt.c | 3 +- + drivers/watchdog/txx9wdt.c | 7 +- + drivers/watchdog/uniphier_wdt.c | 5 +- + drivers/watchdog/via_wdt.c | 5 +- + drivers/watchdog/visconti_wdt.c | 3 +- + drivers/watchdog/w83627hf_wdt.c | 5 +- + drivers/watchdog/w83877f_wdt.c | 5 +- + drivers/watchdog/w83977f_wdt.c | 5 +- + drivers/watchdog/wafer5823wdt.c | 5 +- + drivers/watchdog/watchdog_core.c | 5 +- + drivers/watchdog/watchdog_dev.c | 9 +- + drivers/watchdog/wdat_wdt.c | 5 +- + drivers/watchdog/wdt.c | 9 +- + drivers/watchdog/wdt977.c | 7 +- + drivers/watchdog/wdt_pci.c | 9 +- + drivers/watchdog/wm831x_wdt.c | 3 +- + drivers/watchdog/wm8350_wdt.c | 3 +- + drivers/watchdog/xen_wdt.c | 5 +- + drivers/watchdog/xilinx_wwdt.c | 5 +- + drivers/watchdog/ziirave_wdt.c | 3 +- + include/dt-bindings/reset/mediatek,mt8167-wdt.h | 21 ++++ + 141 files changed, 689 insertions(+), 298 deletions(-) + create mode 100644 include/dt-bindings/reset/mediatek,mt8167-wdt.h +Merging iommu/next (cec2dc7663e71 Merge branches 'apple/dart', 'samsung/exynos', 'riscv', 'broadcom', 'intel/vt-d', 'amd/amd-vi' and 'core' into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/iommu/linux.git iommu/next +Auto-merging MAINTAINERS +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu.h +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_device.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c +Merge made by the 'ort' strategy. + .../bindings/display/brcm,bcm2835-hvs.yaml | 3 + + .../devicetree/bindings/iommu/apple,dart.yaml | 12 +- + .../bindings/iommu/brcm,bcm2712-iommu.yaml | 54 ++ + .../bindings/iommu/brcm,bcm2712-iommuc.yaml | 40 ++ + MAINTAINERS | 12 + + drivers/gpu/drm/amd/amdgpu/amdgpu.h | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_device.c | 50 ++ + drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c | 12 + + drivers/iommu/Kconfig | 17 +- + drivers/iommu/Makefile | 1 + + drivers/iommu/amd/amd_iommu.h | 3 + + drivers/iommu/amd/amd_iommu_types.h | 7 + + drivers/iommu/amd/init.c | 169 ++++--- + drivers/iommu/amd/iommu.c | 460 ++++++++++++----- + drivers/iommu/apple-dart.c | 74 ++- + drivers/iommu/bcm2712-iommu-cache.c | 84 ++++ + drivers/iommu/bcm2712-iommu-cache.h | 9 + + drivers/iommu/bcm2712-iommu.c | 550 +++++++++++++++++++++ + drivers/iommu/exynos-iommu.c | 37 +- + drivers/iommu/generic_pt/.kunitconfig | 1 + + drivers/iommu/generic_pt/Kconfig | 10 + + drivers/iommu/generic_pt/fmt/Makefile | 2 + + drivers/iommu/generic_pt/fmt/bcm2712.h | 288 +++++++++++ + drivers/iommu/generic_pt/fmt/defs_bcm2712.h | 18 + + drivers/iommu/generic_pt/fmt/iommu_bcm2712.c | 6 + + drivers/iommu/generic_pt/kunit_iommu_pt.h | 2 +- + drivers/iommu/intel/dmar.c | 72 ++- + drivers/iommu/intel/iommu.c | 156 +++++- + drivers/iommu/intel/iommu.h | 17 +- + drivers/iommu/intel/nested.c | 2 + + drivers/iommu/iommu-sva.c | 14 +- + drivers/iommu/iommu.c | 4 + + drivers/iommu/riscv/iommu.c | 12 + + include/linux/amd-iommu.h | 13 +- + include/linux/generic_pt/common.h | 6 + + include/linux/generic_pt/iommu.h | 12 + + 36 files changed, 1995 insertions(+), 235 deletions(-) + create mode 100644 Documentation/devicetree/bindings/iommu/brcm,bcm2712-iommu.yaml + create mode 100644 Documentation/devicetree/bindings/iommu/brcm,bcm2712-iommuc.yaml + create mode 100644 drivers/iommu/bcm2712-iommu-cache.c + create mode 100644 drivers/iommu/bcm2712-iommu-cache.h + create mode 100644 drivers/iommu/bcm2712-iommu.c + create mode 100644 drivers/iommu/generic_pt/fmt/bcm2712.h + create mode 100644 drivers/iommu/generic_pt/fmt/defs_bcm2712.h + create mode 100644 drivers/iommu/generic_pt/fmt/iommu_bcm2712.c +Merging audit/next (88bd5e852addf Automated merge of 'dev' into 'next') +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/pcmoore/audit.git audit/next +Auto-merging kernel/auditsc.c +Merge made by the 'ort' strategy. + include/linux/audit.h | 6 ++-- + kernel/audit.c | 2 +- + kernel/audit.h | 20 ++++++++---- + kernel/audit_tree.c | 37 ++++++++++++++++------ + kernel/audit_watch.c | 2 ++ + kernel/auditfilter.c | 86 +++++++++++++++++++++++++-------------------------- + kernel/auditsc.c | 4 ++- + 7 files changed, 92 insertions(+), 65 deletions(-) +Merging devicetree/for-next (833aaa4790eee of/irq: Stop the MSI walk at the first msi-parent) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/robh/linux.git devicetree/for-next +Auto-merging Documentation/devicetree/bindings/i2c/xlnx,xps-iic-2.00.a.yaml +Auto-merging Documentation/devicetree/bindings/mfd/ti,tps65910.yaml +Auto-merging Documentation/devicetree/bindings/trivial-devices.yaml +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../devicetree/bindings/arm/arm,cci-400.yaml | 6 +- + .../devicetree/bindings/arm/arm,coresight-cti.yaml | 2 +- + .../bindings/arm/mediatek/mediatek,audsys.yaml | 2 +- + .../bindings/arm/mediatek/mediatek,g3dsys.txt | 30 - + Documentation/devicetree/bindings/arm/omap/mpu.txt | 54 -- + .../devicetree/bindings/arm/ti/ti,omap-mpu.yaml | 58 ++ + .../devicetree/bindings/arm/vexpress-config.yaml | 4 + + .../devicetree/bindings/bus/omap-ocp2scp.txt | 29 - + .../devicetree/bindings/bus/ti,omap-ocp2scp.yaml | 74 ++ + .../bindings/display/bridge/sil,sii9022.yaml | 2 +- + .../display/mediatek/mediatek,hdmi-ddc.yaml | 11 +- + .../devicetree/bindings/dts-coding-style.rst | 5 +- + .../bindings/firmware/nvidia,tegra186-bpmp.yaml | 6 +- + .../devicetree/bindings/gpio/delta,tn48m-gpio.yaml | 2 +- + .../devicetree/bindings/gpu/arm,mali-utgard.yaml | 25 +- + .../devicetree/bindings/hwmon/adi,ltc2991.yaml | 2 +- + .../devicetree/bindings/hwmon/vexpress.txt | 23 - + .../bindings/i2c/xlnx,xps-iic-2.00.a.yaml | 4 +- + .../bindings/iio/light/upisemi,us5182.yaml | 2 +- + .../devicetree/bindings/input/matrix-keymap.yaml | 3 +- + .../devicetree/bindings/input/ti,tca8418.yaml | 21 +- + .../interrupt-controller/chrp,open-pic.yaml | 7 +- + .../bindings/interrupt-controller/qcom,pdc.yaml | 1 + + .../mailbox/allwinner,sun6i-a31-msgbox.yaml | 3 - + .../devicetree/bindings/mailbox/altera-mailbox.txt | 12 +- + .../bindings/mailbox/hisilicon,hi3660-mailbox.txt | 2 +- + .../bindings/mailbox/hisilicon,hi6220-mailbox.txt | 2 +- + .../devicetree/bindings/mailbox/mailbox.txt | 60 -- + .../bindings/mailbox/ti,omap-mailbox.yaml | 3 +- + .../bindings/media/amlogic,c3-mipi-csi2.yaml | 4 +- + .../devicetree/bindings/media/arm,mali-c55.yaml | 2 +- + .../devicetree/bindings/media/atmel,isc.yaml | 14 +- + .../bindings/media/brcm,bcm2835-unicam.yaml | 10 +- + .../devicetree/bindings/media/cdns,csi2rx.yaml | 58 +- + .../bindings/media/i2c/galaxycore,gc05a2.yaml | 2 +- + .../devicetree/bindings/media/i2c/ovti,ov2680.yaml | 26 +- + .../devicetree/bindings/media/i2c/ovti,ov2732.yaml | 6 +- + .../bindings/media/img,e5010-jpeg-enc.yaml | 22 +- + .../bindings/media/mediatek,vcodec-encoder.yaml | 2 +- + .../bindings/media/mediatek-jpeg-decoder.yaml | 4 +- + .../bindings/media/mediatek-jpeg-encoder.yaml | 2 +- + .../bindings/media/microchip,csi2dc.yaml | 52 +- + .../devicetree/bindings/media/microchip,xisc.yaml | 14 +- + .../devicetree/bindings/media/nxp,imx8-jpeg.yaml | 4 +- + .../bindings/media/qcom,msm8996-venus.yaml | 40 +- + .../bindings/media/qcom,x1e80100-camss.yaml | 26 +- + .../bindings/media/raspberrypi,pispbe.yaml | 12 +- + .../devicetree/bindings/media/samsung,fimc.yaml | 14 +- + .../devicetree/bindings/media/st,stm32-dcmi.yaml | 16 +- + .../devicetree/bindings/media/st,stm32-dcmipp.yaml | 14 +- + .../devicetree/bindings/media/ti,cal.yaml | 48 +- + .../devicetree/bindings/mfd/ti,tps65910.yaml | 2 +- + Documentation/devicetree/bindings/mux/reg-mux.yaml | 2 +- + .../bindings/nvmem/zii,rave-sp-eeprom.yaml | 2 +- + .../bindings/pci/hisilicon,kirin-pcie.yaml | 4 +- + .../bindings/pinctrl/sunplus,sp7021-pinctrl.yaml | 2 +- + .../bindings/power/reset/xlnx,zynqmp-power.yaml | 9 +- + .../devicetree/bindings/regulator/vexpress.txt | 32 - + .../soc/hisilicon/hisilicon,hip05-cpld.yaml | 35 + + .../soc/mediatek/mediatek,mt2701-g3dsys.yaml | 58 ++ + .../bindings/soc/nuvoton/nuvoton,npcm-gcr.yaml | 18 +- + .../bindings/soc/qcom/qcom,aoss-qmp.yaml | 2 +- + .../bindings/spi/aspeed,ast2600-fmc.yaml | 6 +- + .../devicetree/bindings/thermal/thermal-idle.yaml | 2 +- + .../devicetree/bindings/trivial-devices.yaml | 4 +- + Documentation/devicetree/of_unittest.rst | 2 +- + MAINTAINERS | 2 + + drivers/of/fdt_address.c | 2 +- + drivers/of/irq.c | 39 +- + drivers/of/property.c | 23 +- + include/dt-bindings/clock/agilex-clock.h | 1 + + scripts/dtc/dt-check-style | 972 ++++++++++++--------- + .../dtc/dt-style-selftest/bad/dts-blank-lines.dts | 37 + + .../bad/dts-child-name-order.dtso | 33 + + .../dtc/dt-style-selftest/bad/dts-cont-align.dts | 27 + + .../dt-style-selftest/bad/dts-digit-node-order.dts | 40 + + .../bad/dts-digit-node-order.dtso | 41 + + .../bad/dts-extend-node-child-name-order.dtso | 26 + + .../bad/dts-extend-node-digit-node-order.dtso | 34 + + .../dtc/dt-style-selftest/bad/dts-line-length.dts | 22 + + .../dtc/dt-style-selftest/bad/dts-node-name.dts | 60 ++ + .../dt-style-selftest/bad/dts-property-name.dts | 28 + + .../dt-style-selftest/bad/dts-property-order.dts | 18 + + .../dt-style-selftest/bad/dts-property-order.dtso | 62 ++ + .../bad/dts-redundant-ws-strict.dts | 27 + + .../dtc/dt-style-selftest/bad/dts-redundant-ws.dts | 28 + + .../dt-style-selftest/bad/dts-redundant-ws.dtso | 9 + + .../dtc/dt-style-selftest/bad/dts-trailing-ws.dts | 16 + + .../dtc/dt-style-selftest/bad/dts-unused-label.dts | 21 + + .../dt-style-selftest/bad/yaml-blank-lines.yaml | 54 ++ + .../bad/yaml-child-addr-order.yaml | 2 +- + .../bad/yaml-child-name-order.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-cont-align.yaml | 9 +- + .../bad/yaml-digit-node-order.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-hex-case.yaml | 7 +- + .../dt-style-selftest/bad/yaml-indent-strict.yaml | 2 +- + .../bad/yaml-label-in-string.yaml | 2 +- + .../dt-style-selftest/bad/yaml-line-length.yaml | 5 +- + .../dt-style-selftest/bad/yaml-mixed-indent.yaml | 4 +- + .../dt-style-selftest/bad/yaml-multi-close.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-node-close.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-node-name.yaml | 54 ++ + .../bad/yaml-prop-order-device-type.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-prop-order.yaml | 7 +- + .../dt-style-selftest/bad/yaml-prop-pairing.yaml | 2 +- + .../dt-style-selftest/bad/yaml-property-name.yaml | 46 + + .../bad/yaml-redundant-ws-strict.yaml | 31 + + .../dt-style-selftest/bad/yaml-redundant-ws.yaml | 35 + + .../dt-style-selftest/bad/yaml-required-blank.yaml | 2 +- + scripts/dtc/dt-style-selftest/bad/yaml-tab.yaml | 2 +- + .../bad/yaml-trailing-comment.yaml | 2 +- + .../dt-style-selftest/bad/yaml-trailing-ws.yaml | 7 +- + .../bad/yaml-unclosed-comment.yaml | 2 +- + .../bad/yaml-unit-addr-prefix.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-unit-addr.yaml | 2 +- + .../dt-style-selftest/bad/yaml-unused-label.yaml | 2 +- + .../bad/yaml-value-ws-multiline.yaml | 2 +- + .../dtc/dt-style-selftest/bad/yaml-value-ws.yaml | 2 +- + .../expected/dts-blank-lines.dts.txt | 6 + + .../expected/dts-child-name-order.dts.txt | 1 + + .../expected/dts-child-name-order.dtso.txt | 3 + + .../expected/dts-cont-align.dts.txt | 11 + + .../expected/dts-digit-node-order.dts.txt | 2 + + .../expected/dts-digit-node-order.dtso.txt | 2 + + .../dts-extend-node-child-name-order.dtso.txt | 2 + + .../dts-extend-node-digit-node-order.dtso.txt | 2 + + .../expected/dts-line-length.dts.txt | 3 + + .../expected/dts-mixed-indent.dts.txt | 1 + + .../expected/dts-node-name.dts.txt | 13 + + .../expected/dts-property-name.dts.txt | 14 + + .../expected/dts-property-order.dts.txt | 16 +- + .../expected/dts-property-order.dtso.txt | 12 + + .../expected/dts-redundant-ws-strict.dts.txt | 13 + + .../expected/dts-redundant-ws.dts.txt | 10 + + .../expected/dts-redundant-ws.dtso.txt | 2 + + .../expected/dts-trailing-ws.dts.txt | 3 + + .../expected/dts-unused-label.dts.txt | 2 + + .../expected/yaml-blank-lines.yaml.txt | 6 + + .../expected/yaml-cont-align.yaml.txt | 4 +- + .../expected/yaml-hex-case.yaml.txt | 2 + + .../expected/yaml-line-length.yaml.txt | 1 + + .../expected/yaml-mixed-indent.yaml.txt | 1 + + .../expected/yaml-node-name.yaml.txt | 7 + + .../expected/yaml-prop-order.yaml.txt | 1 + + .../expected/yaml-prop-pairing.yaml.txt | 4 +- + .../expected/yaml-property-name.yaml.txt | 14 + + .../expected/yaml-redundant-ws-strict.yaml.txt | 5 + + .../expected/yaml-redundant-ws.yaml.txt | 4 + + .../expected/yaml-trailing-ws.yaml.txt | 2 + + .../expected/yaml-value-ws-multiline.yaml.txt | 1 + + .../good/dts-child-name-order.dtso | 44 + + .../dtc/dt-style-selftest/good/dts-cont-align.dts | 14 +- + .../good/dts-digit-node-order.dts | 3 - + .../good/dts-digit-node-order.dtso | 59 ++ + .../good/dts-extend-node-child-name-order.dtso | 26 + + .../good/dts-extend-node-digit-node-order.dtso | 34 + + .../dt-style-selftest/good/dts-property-order.dts | 8 + + .../dt-style-selftest/good/dts-property-order.dtso | 50 ++ + .../dtc/dt-style-selftest/good/yaml-4space.yaml | 2 +- + .../dt-style-selftest/good/yaml-cont-align.yaml | 33 + + .../good/yaml-tricky-parsing.yaml | 2 +- + scripts/dtc/dt-style-selftest/run.sh | 2 +- + 162 files changed, 2357 insertions(+), 980 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/arm/mediatek/mediatek,g3dsys.txt + delete mode 100644 Documentation/devicetree/bindings/arm/omap/mpu.txt + create mode 100644 Documentation/devicetree/bindings/arm/ti/ti,omap-mpu.yaml + delete mode 100644 Documentation/devicetree/bindings/bus/omap-ocp2scp.txt + create mode 100644 Documentation/devicetree/bindings/bus/ti,omap-ocp2scp.yaml + delete mode 100644 Documentation/devicetree/bindings/hwmon/vexpress.txt + delete mode 100644 Documentation/devicetree/bindings/mailbox/mailbox.txt + delete mode 100644 Documentation/devicetree/bindings/regulator/vexpress.txt + create mode 100644 Documentation/devicetree/bindings/soc/hisilicon/hisilicon,hip05-cpld.yaml + create mode 100644 Documentation/devicetree/bindings/soc/mediatek/mediatek,mt2701-g3dsys.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-blank-lines.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-child-name-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-cont-align.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-digit-node-order.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-digit-node-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-extend-node-child-name-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-extend-node-digit-node-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-line-length.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-node-name.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-property-name.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-property-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-redundant-ws-strict.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-redundant-ws.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-redundant-ws.dtso + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-trailing-ws.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/dts-unused-label.dts + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-blank-lines.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-node-name.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-property-name.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-redundant-ws-strict.yaml + create mode 100644 scripts/dtc/dt-style-selftest/bad/yaml-redundant-ws.yaml + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-blank-lines.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-child-name-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-cont-align.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-digit-node-order.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-digit-node-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-extend-node-child-name-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-extend-node-digit-node-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-line-length.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-node-name.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-property-name.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-property-order.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-redundant-ws-strict.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-redundant-ws.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-redundant-ws.dtso.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-trailing-ws.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/dts-unused-label.dts.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-blank-lines.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-node-name.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-property-name.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-redundant-ws-strict.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/expected/yaml-redundant-ws.yaml.txt + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-child-name-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-digit-node-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-extend-node-child-name-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-extend-node-digit-node-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/dts-property-order.dtso + create mode 100644 scripts/dtc/dt-style-selftest/good/yaml-cont-align.yaml +Merging dt-krzk/for-next (dfcea1641506d Merge branches 'next/dt' and 'next/dt64' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-dt.git dt-krzk/for-next +Merge made by the 'ort' strategy. + arch/arm/boot/dts/actions/owl-s500.dtsi | 12 ++++++------ + arch/arm/boot/dts/arm/vexpress-v2p-ca5s.dts | 2 +- + arch/arm/boot/dts/aspeed/aspeed-bmc-opp-witherspoon.dts | 2 +- + arch/arm/boot/dts/aspeed/aspeed-bmc-supermicro-x11spi.dts | 2 +- + arch/arm64/boot/dts/actions/s700.dtsi | 6 +++--- + arch/arm64/boot/dts/actions/s900.dtsi | 6 +++--- + arch/arm64/boot/dts/cavium/thunder-88xx.dtsi | 6 +++--- + 7 files changed, 18 insertions(+), 18 deletions(-) +Merging mailbox/for-next (14af7a96afa39 mailbox: add Axiado AX3005 mailbox driver) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jassibrar/mailbox.git mailbox/for-next +Already up to date. +Merging spi/for-next (168b675c187d4 Merge spi/for-7.4 into spi-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/broonie/spi.git spi/for-next +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-class-spi-master | 6 +- + .../devicetree/bindings/spi/amlogic,a4-spisg.yaml | 30 +- + .../bindings/spi/amlogic,meson-gx-spicc.yaml | 13 +- + .../bindings/spi/microchip,pic32mzda-spi.yaml | 82 ++++ + .../bindings/spi/microchip,spi-pic32.txt | 34 -- + .../devicetree/bindings/spi/renesas,sh-msiof.yaml | 9 +- + .../bindings/spi/spi-peripheral-props.yaml | 8 + + Documentation/spi/multiple-data-lanes.rst | 10 +- + drivers/spi/Kconfig | 364 +++++++-------- + drivers/spi/spi-amlogic-spisg.c | 116 ++++- + drivers/spi/spi-ar934x.c | 13 +- + drivers/spi/spi-bcm2835.c | 4 +- + drivers/spi/spi-bcm63xx.c | 2 +- + drivers/spi/spi-dw-core.c | 15 +- + drivers/spi/spi-geni-qcom.c | 93 +++- + drivers/spi/spi-imx.c | 2 +- + drivers/spi/spi-ingenic.c | 5 +- + drivers/spi/spi-ma35d1-qspi.c | 27 +- + drivers/spi/spi-mem.c | 14 +- + drivers/spi/spi-mtk-nor.c | 1 + + drivers/spi/spi-mxic.c | 8 +- + drivers/spi/spi-omap2-mcspi.c | 4 +- + drivers/spi/spi-orion.c | 9 + + drivers/spi/spi-pic32.c | 2 +- + drivers/spi/spi-qpic-snand.c | 104 ++--- + drivers/spi/spi-realtek-rtl.c | 36 +- + drivers/spi/spi-rockchip-sfc.c | 21 +- + drivers/spi/spi-rspi.c | 49 +- + drivers/spi/spi-s3c64xx.c | 2 +- + drivers/spi/spi-sh-msiof.c | 6 +- + drivers/spi/spi-stm32.c | 2 +- + drivers/spi/spi-sun6i.c | 2 +- + drivers/spi/spi-sunplus-sp7021.c | 13 +- + drivers/spi/spi-tegra210-quad.c | 517 ++++++++++++++++++--- + drivers/spi/spi-virtio.c | 22 +- + drivers/spi/spi-xilinx.c | 2 +- + drivers/spi/spi.c | 15 +- + include/linux/spi/spi.h | 9 + + tools/spi/Makefile | 2 +- + tools/spi/spidev_test.c | 330 +++++++++---- + 40 files changed, 1396 insertions(+), 607 deletions(-) + create mode 100644 Documentation/devicetree/bindings/spi/microchip,pic32mzda-spi.yaml + delete mode 100644 Documentation/devicetree/bindings/spi/microchip,spi-pic32.txt +Merging tip/master (1aeb52f7869a6 Merge branch into tip/master: 'x86/tdx') + 2c6a75adb15f8 ("x86/um: Remove unused header") +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git tip/master +Auto-merging Documentation/ABI/testing/sysfs-devices-system-cpu +Auto-merging Documentation/devicetree/bindings/interrupt-controller/qcom,pdc.yaml +Auto-merging Documentation/scheduler/index.rst +CONFLICT (content): Merge conflict in Documentation/scheduler/index.rst +Auto-merging MAINTAINERS +Auto-merging arch/Kconfig +Auto-merging arch/arm64/Kconfig +Auto-merging arch/arm64/Kconfig.platforms +Auto-merging arch/arm64/configs/defconfig +CONFLICT (content): Merge conflict in arch/arm64/configs/defconfig +Auto-merging arch/arm64/include/asm/preempt.h +Auto-merging arch/loongarch/Kconfig +Auto-merging arch/powerpc/Kconfig +Auto-merging arch/riscv/Kconfig +Auto-merging arch/s390/Kconfig +Auto-merging arch/s390/kernel/hiperdispatch.c +Auto-merging arch/um/kernel/um_arch.c +Auto-merging arch/x86/Kconfig +Auto-merging arch/x86/crypto/aesni-intel_glue.c +Auto-merging arch/x86/include/asm/string_64.h +Auto-merging arch/x86/kvm/mmu/mmu.c +Auto-merging arch/x86/kvm/svm/sev.c +Auto-merging arch/x86/kvm/vmx/tdx.c +Auto-merging fs/aio.c +Auto-merging fs/exec.c +Auto-merging include/linux/sched.h +Auto-merging io_uring/rw.c +Auto-merging kernel/events/core.c +Auto-merging kernel/exit.c +Auto-merging kernel/sched/fair.c +Auto-merging net/core/pktgen.c +Resolved 'Documentation/scheduler/index.rst' using previous resolution. +Resolved 'arch/arm64/configs/defconfig' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 24bf019cbe7e4] Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/tip/tip.git +$ git diff -M --stat --summary HEAD^.. + Documentation/ABI/testing/sysfs-devices-system-cpu | 24 +- + Documentation/ABI/testing/sysfs-platform-ts5500 | 54 -- + Documentation/arch/arm64/silicon-errata.rst | 3 + + Documentation/arch/x86/tdx.rst | 21 + + Documentation/arch/x86/xstate.rst | 54 ++ + .../bindings/interrupt-controller/qcom,pdc.yaml | 1 + + Documentation/driver-api/index.rst | 1 + + Documentation/driver-api/steal-governor.rst | 151 ++++ + Documentation/filesystems/resctrl.rst | 19 +- + Documentation/scheduler/index.rst | 1 + + Documentation/scheduler/sched-paravirt.rst | 67 ++ + MAINTAINERS | 14 +- + arch/Kconfig | 38 - + arch/arm/kernel/perf_regs.c | 8 +- + arch/arm64/Kconfig | 10 +- + arch/arm64/Kconfig.platforms | 1 + + arch/arm64/boot/dts/qcom/purwa.dtsi | 5 + + arch/arm64/configs/defconfig | 2 - + arch/arm64/include/asm/preempt.h | 10 - + arch/arm64/kernel/paravirt.c | 4 +- + arch/arm64/kernel/perf_regs.c | 8 +- + arch/csky/kernel/perf_regs.c | 8 +- + arch/loongarch/Kconfig | 1 - + arch/loongarch/kernel/paravirt.c | 4 +- + arch/loongarch/kernel/perf_regs.c | 8 +- + arch/mips/kernel/perf_regs.c | 8 +- + arch/parisc/kernel/perf_regs.c | 8 +- + arch/powerpc/Kconfig | 1 - + arch/powerpc/perf/perf_regs.c | 2 +- + arch/powerpc/platforms/pseries/setup.c | 4 +- + arch/riscv/Kconfig | 1 - + arch/riscv/include/asm/smp.h | 12 + + arch/riscv/kernel/paravirt.c | 4 +- + arch/riscv/kernel/perf_regs.c | 8 +- + arch/riscv/kernel/sbi-ipi.c | 4 +- + arch/riscv/kernel/smp.c | 12 - + arch/s390/Kconfig | 1 - + arch/s390/include/asm/preempt.h | 11 - + arch/s390/kernel/hiperdispatch.c | 10 +- + arch/s390/kernel/perf_regs.c | 2 +- + arch/um/kernel/um_arch.c | 78 +- + arch/x86/Kconfig | 10 - + arch/x86/boot/compressed/error.c | 19 - + arch/x86/boot/compressed/error.h | 1 - + arch/x86/boot/compressed/mem.c | 42 - + arch/x86/boot/compressed/sev.h | 2 - + arch/x86/boot/compressed/tdx-shared.c | 2 + + arch/x86/boot/early_serial_console.c | 13 +- + arch/x86/boot/string.c | 13 +- + arch/x86/coco/sev/core.c | 32 +- + arch/x86/coco/tdx/tdx-shared.c | 31 + + arch/x86/coco/tdx/tdx.c | 41 +- + arch/x86/crypto/aegis128-aesni-glue.c | 3 +- + arch/x86/crypto/aesni-intel_glue.c | 7 +- + arch/x86/crypto/aria_aesni_avx2_glue.c | 11 +- + arch/x86/crypto/aria_aesni_avx_glue.c | 11 +- + arch/x86/crypto/aria_gfni_avx512_glue.c | 11 +- + arch/x86/crypto/camellia_aesni_avx2_glue.c | 11 +- + arch/x86/crypto/camellia_aesni_avx_glue.c | 11 +- + arch/x86/crypto/cast5_avx_glue.c | 7 +- + arch/x86/crypto/cast6_avx_glue.c | 7 +- + arch/x86/crypto/serpent_avx2_glue.c | 9 +- + arch/x86/crypto/serpent_avx_glue.c | 7 +- + arch/x86/crypto/sm4_aesni_avx2_glue.c | 11 +- + arch/x86/crypto/sm4_aesni_avx_glue.c | 11 +- + arch/x86/crypto/twofish_avx_glue.c | 6 +- + arch/x86/events/amd/uncore.c | 39 +- + arch/x86/events/core.c | 488 ++++++++++- + arch/x86/events/intel/core.c | 154 ++-- + arch/x86/events/intel/ds.c | 232 ++++-- + arch/x86/events/intel/lbr.c | 2 +- + arch/x86/events/perf_event.h | 217 ++++- + arch/x86/include/asm/cpufeatures.h | 5 +- + arch/x86/include/asm/cpuid/api.h | 2 +- + arch/x86/include/asm/fpu/regset.h | 6 +- + arch/x86/include/asm/fpu/sched.h | 6 +- + arch/x86/include/asm/fpu/types.h | 25 - + arch/x86/include/asm/fpu/xstate.h | 3 + + arch/x86/include/asm/kvm-x86-ops.h | 1 + + arch/x86/include/asm/kvm_host.h | 10 +- + arch/x86/include/asm/local.h | 4 +- + arch/x86/include/asm/math_emu.h | 15 - + arch/x86/include/asm/msr-index.h | 10 + + arch/x86/include/asm/perf_event.h | 46 +- + arch/x86/include/asm/preempt.h | 28 - + arch/x86/include/asm/processor.h | 2 +- + arch/x86/include/asm/sev.h | 4 + + arch/x86/include/asm/shared/string.h | 52 ++ + arch/x86/include/asm/shared/tdx.h | 6 + + arch/x86/include/asm/string.h | 21 +- + arch/x86/include/asm/string_64.h | 1 - + arch/x86/include/asm/tdx.h | 23 + + arch/x86/include/asm/tdx_global_metadata.h | 9 +- + arch/x86/include/asm/traps.h | 2 - + arch/x86/include/uapi/asm/perf_regs.h | 53 ++ + arch/x86/include/uapi/asm/sigcontext.h | 15 + + arch/x86/kernel/asm-offsets.c | 1 + + arch/x86/kernel/cpu/amd.c | 21 +- + arch/x86/kernel/cpu/bugs.c | 18 +- + arch/x86/kernel/cpu/common.c | 98 ++- + arch/x86/kernel/cpu/microcode/intel-ucode-defs.h | 91 +- + arch/x86/kernel/cpu/mtrr/amd.c | 9 +- + arch/x86/kernel/cpu/resctrl/ctrlmondata.c | 6 + + arch/x86/kernel/cpu/scattered.c | 2 + + arch/x86/kernel/cpu/sgx/main.c | 8 +- + arch/x86/kernel/cpu/vmware.c | 4 +- + arch/x86/kernel/crash.c | 6 +- + arch/x86/kernel/fpu/bugs.c | 4 - + arch/x86/kernel/fpu/core.c | 46 +- + arch/x86/kernel/fpu/init.c | 6 +- + arch/x86/kernel/fpu/regset.c | 6 - + arch/x86/kernel/fpu/signal.c | 136 +-- + arch/x86/kernel/fpu/xstate.c | 62 +- + arch/x86/kernel/kvm.c | 4 +- + arch/x86/kernel/perf_regs.c | 172 +++- + arch/x86/kernel/shstk.c | 2 +- + arch/x86/kvm/mmu/mmu.c | 4 + + arch/x86/kvm/svm/sev.c | 2 + + arch/x86/kvm/vmx/pmu_intel.c | 28 +- + arch/x86/kvm/vmx/tdx.c | 99 ++- + arch/x86/kvm/vmx/tdx.h | 2 + + arch/x86/kvm/vmx/vmx.c | 10 +- + arch/x86/kvm/vmx/vmx.h | 15 +- + arch/x86/platform/Makefile | 1 - + arch/x86/platform/olpc/olpc-xo15-sci.c | 4 +- + arch/x86/platform/pvh/enlighten.c | 3 +- + arch/x86/platform/ts5500/Makefile | 2 - + arch/x86/platform/ts5500/ts5500.c | 341 -------- + arch/x86/virt/svm/sev.c | 172 +++- + arch/x86/virt/vmx/tdx/seamcall_internal.h | 19 + + arch/x86/virt/vmx/tdx/tdx.c | 439 ++++++++-- + arch/x86/virt/vmx/tdx/tdx.h | 10 +- + arch/x86/virt/vmx/tdx/tdx_global_metadata.c | 23 +- + arch/x86/virt/vmx/tdx/tdxcall.S | 10 +- + arch/x86/xen/pmu.c | 5 +- + drivers/base/cpu.c | 12 + + drivers/clocksource/timer-clint.c | 4 +- + drivers/crypto/ccp/sev-dev.c | 2 + + drivers/firmware/efi/libstub/x86-stub.c | 39 + + drivers/irqchip/Kconfig | 6 +- + drivers/irqchip/irq-aclint-sswi.c | 4 +- + drivers/irqchip/irq-al-fic.c | 2 +- + drivers/irqchip/irq-gic-v3-its.c | 6 +- + drivers/irqchip/irq-gic-v3.c | 12 +- + drivers/irqchip/irq-gic-v5.c | 9 +- + drivers/irqchip/irq-gic.c | 8 +- + drivers/irqchip/irq-lan966x-oic.c | 1 - + drivers/irqchip/irq-mtk-cirq.c | 2 +- + drivers/irqchip/irq-pruss-intc.c | 38 +- + drivers/irqchip/irq-riscv-imsic-early.c | 4 +- + drivers/irqchip/irq-riscv-imsic-state.c | 13 +- + drivers/irqchip/irq-riscv-imsic-state.h | 1 - + drivers/irqchip/irq-sifive-plic.c | 2 +- + drivers/irqchip/irq-vic.c | 2 +- + drivers/irqchip/qcom-pdc.c | 3 + + drivers/resctrl/mpam_resctrl.c | 5 + + drivers/soc/fsl/qe/qe_ports_ic.c | 1 - + drivers/virt/Kconfig | 17 + + drivers/virt/Makefile | 1 + + drivers/virt/coco/tdx-guest/tdx-guest.c | 6 +- + drivers/virt/steal_governor.c | 296 +++++++ + drivers/xen/time.c | 4 +- + fs/aio.c | 2 +- + fs/exec.c | 13 +- + fs/proc/uptime.c | 6 +- + fs/resctrl/ctrlmondata.c | 6 +- + fs/resctrl/pseudo_lock.c | 9 +- + fs/resctrl/rdtgroup.c | 8 +- + include/asm-generic/preempt.h | 10 - + include/linux/bitmap.h | 14 + + include/linux/cpumask.h | 42 + + include/linux/hrtimer.h | 10 +- + include/linux/interrupt.h | 6 +- + include/linux/irq-entry-common.h | 17 +- + include/linux/kernel.h | 20 - + include/linux/kernel_stat.h | 11 + + include/linux/list.h | 6 +- + include/linux/perf_event.h | 23 + + include/linux/perf_regs.h | 36 +- + include/linux/posix-timers.h | 39 +- + include/linux/preempt.h | 20 +- + include/linux/resctrl.h | 19 + + include/linux/sched.h | 32 +- + include/linux/sched/cputime.h | 6 +- + include/linux/sched/task.h | 1 - + include/linux/wait.h | 2 +- + include/uapi/linux/perf_event.h | 49 +- + include/uapi/linux/sched.h | 2 +- + include/vdso/math64.h | 31 +- + io_uring/rw.c | 2 +- + kernel/Kconfig.kexec | 2 +- + kernel/Kconfig.preempt | 13 +- + kernel/cpu.c | 6 + + kernel/crash_core.c | 2 +- + kernel/entry/common.c | 17 +- + kernel/events/core.c | 177 +++- + kernel/exit.c | 13 +- + kernel/futex/requeue.c | 2 +- + kernel/futex/waitwake.c | 8 +- + kernel/irq/irqdomain.c | 1 + + kernel/locking/rtmutex.c | 2 +- + kernel/sched/core.c | 444 +++++----- + kernel/sched/cputime.c | 4 +- + kernel/sched/deadline.c | 9 +- + kernel/sched/debug.c | 21 +- + kernel/sched/ext/ext.c | 18 +- + kernel/sched/ext/ext.h | 7 + + kernel/sched/fair.c | 131 ++- + kernel/sched/idle.c | 5 +- + kernel/sched/rt.c | 7 +- + kernel/sched/sched.h | 80 +- + kernel/sched/stop_task.c | 5 +- + kernel/sched/wait.c | 22 +- + kernel/time/hrtimer.c | 23 +- + kernel/time/posix-cpu-timers.c | 103 ++- + kernel/time/posix-timers.c | 26 +- + kernel/time/posix-timers.h | 3 + + kernel/time/sleep_timeout.c | 4 +- + kernel/time/tick-sched.c | 30 +- + kernel/time/time_test.c | 16 + + kernel/time/timeconv.c | 6 +- + kernel/time/timekeeping.c | 4 +- + kernel/time/timer.c | 2 +- + kernel/time/timer_migration.c | 6 +- + kernel/time/vsyscall.c | 26 +- + lib/bitmap.c | 17 + + lib/crc/x86/crc-pclmul-template.h | 6 +- + lib/crypto/x86/blake2s.h | 4 +- + lib/crypto/x86/chacha.h | 3 +- + lib/crypto/x86/nh.h | 4 +- + lib/crypto/x86/poly1305.h | 7 +- + lib/crypto/x86/sha1.h | 4 +- + lib/crypto/x86/sha256.h | 4 +- + lib/crypto/x86/sha512.h | 3 +- + lib/crypto/x86/sm3.h | 3 +- + lib/raid/xor/Makefile | 2 +- + lib/raid/xor/x86/xor-avx512.c | 122 +++ + lib/raid/xor/x86/xor_arch.h | 31 +- + lib/vdso/gettimeofday.c | 2 + + net/core/pktgen.c | 4 +- + tools/objtool/Documentation/klp-test-design.txt | 286 +++++++ + tools/objtool/Documentation/klp-write-tests.txt | 266 ++++++ + tools/objtool/Makefile | 15 +- + .../tests/generic/fixtures/abs_and_addressable.c | 44 + + tools/objtool/tests/generic/fixtures/basic.c | 20 + + .../objtool/tests/generic/fixtures/changed_data.c | 16 + + .../objtool/tests/generic/fixtures/checksum_data.c | 116 +++ + .../objtool/tests/generic/fixtures/checksum_insn.c | 78 ++ + .../tests/generic/fixtures/checksum_position.c | 42 + + .../objtool/tests/generic/fixtures/checksum_skip.c | 47 ++ + .../objtool/tests/generic/fixtures/cold_function.c | 21 + + .../objtool/tests/generic/fixtures/cross_module.c | 25 + + .../tests/generic/fixtures/data_alignment.c | 29 + + .../tests/generic/fixtures/function_removal.c | 25 + + .../tests/generic/fixtures/init_reference.c | 17 + + tools/objtool/tests/generic/fixtures/jump_label.c | 76 ++ + tools/objtool/tests/generic/fixtures/klp_funcs.c | 31 + + .../tests/generic/fixtures/local_to_global.c | 34 + + tools/objtool/tests/generic/fixtures/new_data.c | 23 + + .../tests/generic/fixtures/new_export_ref.c | 35 + + .../objtool/tests/generic/fixtures/new_function.c | 21 + + tools/objtool/tests/generic/fixtures/no_modinfo.c | 11 + + .../tests/generic/fixtures/special_section.c | 24 + + .../generic/fixtures/special_section_shared.c | 31 + + tools/objtool/tests/generic/fixtures/static_call.c | 59 ++ + .../objtool/tests/generic/fixtures/static_local.c | 17 + + .../generic/fixtures/static_local_uncorrelated.c | 41 + + .../objtool/tests/generic/fixtures/switch_rodata.c | 31 + + .../tests/generic/fixtures/symid_discarded.c | 25 + + tools/objtool/tests/generic/fixtures/sympos_dup.c | 32 + + .../tests/generic/fixtures/sympos_vmlinux.c | 40 + + .../tests/generic/fixtures/thinlto_ambiguity.c | 57 ++ + .../objtool/tests/generic/fixtures/thinlto_local.c | 39 + + tools/objtool/tests/generic/fixtures/ubsan_noise.c | 49 ++ + .../tests/generic/test-abs-and-addressable.sh | 50 ++ + tools/objtool/tests/generic/test-basic.sh | 17 + + tools/objtool/tests/generic/test-changed-data.sh | 18 + + tools/objtool/tests/generic/test-checksum-data.sh | 61 ++ + tools/objtool/tests/generic/test-checksum-debug.sh | 49 ++ + tools/objtool/tests/generic/test-checksum-insn.sh | 49 ++ + .../tests/generic/test-checksum-position.sh | 51 ++ + tools/objtool/tests/generic/test-checksum-skip.sh | 81 ++ + tools/objtool/tests/generic/test-checksum-value.sh | 37 + + tools/objtool/tests/generic/test-cold-function.sh | 39 + + tools/objtool/tests/generic/test-data-alignment.sh | 40 + + .../generic/test-export-symbol-for-modules.sh | 39 + + .../objtool/tests/generic/test-function-removal.sh | 34 + + tools/objtool/tests/generic/test-init-reference.sh | 29 + + .../tests/generic/test-jump-label-exempt-keys.sh | 51 ++ + tools/objtool/tests/generic/test-jump-label-key.sh | 44 + + .../tests/generic/test-jump-label-module-key.sh | 22 + + .../generic/test-jump-label-module-static-key.sh | 45 + + .../tests/generic/test-jump-label-new-key.sh | 51 ++ + .../tests/generic/test-klp-funcs-content.sh | 45 + + .../tests/generic/test-local-to-global-flip.sh | 63 ++ + .../objtool/tests/generic/test-local-vs-export.sh | 32 + + .../objtool/tests/generic/test-missing-checksum.sh | 18 + + .../objtool/tests/generic/test-missing-modinfo.sh | 16 + + .../tests/generic/test-modname-normalize.sh | 26 + + tools/objtool/tests/generic/test-module-object.sh | 31 + + .../tests/generic/test-module-vmlinux-reloc.sh | 40 + + tools/objtool/tests/generic/test-new-data.sh | 27 + + tools/objtool/tests/generic/test-new-export-ref.sh | 46 ++ + tools/objtool/tests/generic/test-new-function.sh | 16 + + tools/objtool/tests/generic/test-post-link.sh | 42 + + .../tests/generic/test-special-section-shared.sh | 26 + + .../objtool/tests/generic/test-special-section.sh | 20 + + .../generic/test-static-call-annotate-stripped.sh | 42 + + .../tests/generic/test-static-call-module-key.sh | 36 + + .../objtool/tests/generic/test-static-call-new.sh | 45 + + .../generic/test-static-local-uncorrelated.sh | 41 + + tools/objtool/tests/generic/test-static-local.sh | 24 + + tools/objtool/tests/generic/test-switch-rodata.sh | 53 ++ + .../objtool/tests/generic/test-symid-discarded.sh | 44 + + tools/objtool/tests/generic/test-sympos-vmlinux.sh | 57 ++ + tools/objtool/tests/generic/test-sympos.sh | 51 ++ + .../tests/generic/test-symvers-parse-error.sh | 23 + + .../tests/generic/test-thinlto-ambiguity.sh | 77 ++ + tools/objtool/tests/generic/test-thinlto-local.sh | 48 ++ + tools/objtool/tests/generic/test-ubsan-noise.sh | 48 ++ + tools/objtool/tests/lib.sh | 911 +++++++++++++++++++++ + tools/objtool/tests/run-tests.sh | 239 ++++++ + tools/objtool/tests/x86/fixtures/alt_annotate.c | 57 ++ + tools/objtool/tests/x86/fixtures/checksum_alt.c | 66 ++ + .../objtool/tests/x86/fixtures/empty_alternative.c | 77 ++ + tools/objtool/tests/x86/fixtures/kcfi.c | 39 + + .../objtool/tests/x86/fixtures/special_sections.c | 77 ++ + .../tests/x86/fixtures/static_call_no_key.c | 32 + + tools/objtool/tests/x86/test-alt-annotation.sh | 38 + + tools/objtool/tests/x86/test-checksum-alt.sh | 45 + + tools/objtool/tests/x86/test-empty-alternative.sh | 31 + + tools/objtool/tests/x86/test-kcfi.sh | 39 + + .../tests/x86/test-manual-klp-static-call.sh | 40 + + tools/objtool/tests/x86/test-special-sections.sh | 42 + + tools/perf/trace/beauty/include/uapi/linux/sched.h | 2 +- + .../testing/selftests/timers/clocksource-switch.c | 23 +- + tools/testing/selftests/timers/posix_timers.c | 7 +- + tools/testing/selftests/timers/raw_skew.c | 12 + + tools/testing/selftests/x86/Makefile | 5 +- + .../selftests/x86/sigframe_fpu_portability.c | 235 ++++++ + tools/testing/selftests/x86/xstate.c | 12 - + tools/testing/selftests/x86/xstate.h | 20 + + 342 files changed, 10162 insertions(+), 2241 deletions(-) + delete mode 100644 Documentation/ABI/testing/sysfs-platform-ts5500 + create mode 100644 Documentation/driver-api/steal-governor.rst + create mode 100644 Documentation/scheduler/sched-paravirt.rst + delete mode 100644 arch/x86/include/asm/math_emu.h + create mode 100644 arch/x86/include/asm/shared/string.h + delete mode 100644 arch/x86/platform/ts5500/Makefile + delete mode 100644 arch/x86/platform/ts5500/ts5500.c + create mode 100644 drivers/virt/steal_governor.c + create mode 100644 lib/raid/xor/x86/xor-avx512.c + create mode 100644 tools/objtool/Documentation/klp-test-design.txt + create mode 100644 tools/objtool/Documentation/klp-write-tests.txt + create mode 100644 tools/objtool/tests/generic/fixtures/abs_and_addressable.c + create mode 100644 tools/objtool/tests/generic/fixtures/basic.c + create mode 100644 tools/objtool/tests/generic/fixtures/changed_data.c + create mode 100644 tools/objtool/tests/generic/fixtures/checksum_data.c + create mode 100644 tools/objtool/tests/generic/fixtures/checksum_insn.c + create mode 100644 tools/objtool/tests/generic/fixtures/checksum_position.c + create mode 100644 tools/objtool/tests/generic/fixtures/checksum_skip.c + create mode 100644 tools/objtool/tests/generic/fixtures/cold_function.c + create mode 100644 tools/objtool/tests/generic/fixtures/cross_module.c + create mode 100644 tools/objtool/tests/generic/fixtures/data_alignment.c + create mode 100644 tools/objtool/tests/generic/fixtures/function_removal.c + create mode 100644 tools/objtool/tests/generic/fixtures/init_reference.c + create mode 100644 tools/objtool/tests/generic/fixtures/jump_label.c + create mode 100644 tools/objtool/tests/generic/fixtures/klp_funcs.c + create mode 100644 tools/objtool/tests/generic/fixtures/local_to_global.c + create mode 100644 tools/objtool/tests/generic/fixtures/new_data.c + create mode 100644 tools/objtool/tests/generic/fixtures/new_export_ref.c + create mode 100644 tools/objtool/tests/generic/fixtures/new_function.c + create mode 100644 tools/objtool/tests/generic/fixtures/no_modinfo.c + create mode 100644 tools/objtool/tests/generic/fixtures/special_section.c + create mode 100644 tools/objtool/tests/generic/fixtures/special_section_shared.c + create mode 100644 tools/objtool/tests/generic/fixtures/static_call.c + create mode 100644 tools/objtool/tests/generic/fixtures/static_local.c + create mode 100644 tools/objtool/tests/generic/fixtures/static_local_uncorrelated.c + create mode 100644 tools/objtool/tests/generic/fixtures/switch_rodata.c + create mode 100644 tools/objtool/tests/generic/fixtures/symid_discarded.c + create mode 100644 tools/objtool/tests/generic/fixtures/sympos_dup.c + create mode 100644 tools/objtool/tests/generic/fixtures/sympos_vmlinux.c + create mode 100644 tools/objtool/tests/generic/fixtures/thinlto_ambiguity.c + create mode 100644 tools/objtool/tests/generic/fixtures/thinlto_local.c + create mode 100644 tools/objtool/tests/generic/fixtures/ubsan_noise.c + create mode 100755 tools/objtool/tests/generic/test-abs-and-addressable.sh + create mode 100755 tools/objtool/tests/generic/test-basic.sh + create mode 100755 tools/objtool/tests/generic/test-changed-data.sh + create mode 100755 tools/objtool/tests/generic/test-checksum-data.sh + create mode 100755 tools/objtool/tests/generic/test-checksum-debug.sh + create mode 100755 tools/objtool/tests/generic/test-checksum-insn.sh + create mode 100755 tools/objtool/tests/generic/test-checksum-position.sh + create mode 100755 tools/objtool/tests/generic/test-checksum-skip.sh + create mode 100755 tools/objtool/tests/generic/test-checksum-value.sh + create mode 100755 tools/objtool/tests/generic/test-cold-function.sh + create mode 100755 tools/objtool/tests/generic/test-data-alignment.sh + create mode 100755 tools/objtool/tests/generic/test-export-symbol-for-modules.sh + create mode 100755 tools/objtool/tests/generic/test-function-removal.sh + create mode 100755 tools/objtool/tests/generic/test-init-reference.sh + create mode 100755 tools/objtool/tests/generic/test-jump-label-exempt-keys.sh + create mode 100755 tools/objtool/tests/generic/test-jump-label-key.sh + create mode 100755 tools/objtool/tests/generic/test-jump-label-module-key.sh + create mode 100755 tools/objtool/tests/generic/test-jump-label-module-static-key.sh + create mode 100755 tools/objtool/tests/generic/test-jump-label-new-key.sh + create mode 100755 tools/objtool/tests/generic/test-klp-funcs-content.sh + create mode 100755 tools/objtool/tests/generic/test-local-to-global-flip.sh + create mode 100755 tools/objtool/tests/generic/test-local-vs-export.sh + create mode 100755 tools/objtool/tests/generic/test-missing-checksum.sh + create mode 100755 tools/objtool/tests/generic/test-missing-modinfo.sh + create mode 100755 tools/objtool/tests/generic/test-modname-normalize.sh + create mode 100755 tools/objtool/tests/generic/test-module-object.sh + create mode 100755 tools/objtool/tests/generic/test-module-vmlinux-reloc.sh + create mode 100755 tools/objtool/tests/generic/test-new-data.sh + create mode 100755 tools/objtool/tests/generic/test-new-export-ref.sh + create mode 100755 tools/objtool/tests/generic/test-new-function.sh + create mode 100755 tools/objtool/tests/generic/test-post-link.sh + create mode 100755 tools/objtool/tests/generic/test-special-section-shared.sh + create mode 100755 tools/objtool/tests/generic/test-special-section.sh + create mode 100755 tools/objtool/tests/generic/test-static-call-annotate-stripped.sh + create mode 100755 tools/objtool/tests/generic/test-static-call-module-key.sh + create mode 100755 tools/objtool/tests/generic/test-static-call-new.sh + create mode 100755 tools/objtool/tests/generic/test-static-local-uncorrelated.sh + create mode 100755 tools/objtool/tests/generic/test-static-local.sh + create mode 100755 tools/objtool/tests/generic/test-switch-rodata.sh + create mode 100755 tools/objtool/tests/generic/test-symid-discarded.sh + create mode 100755 tools/objtool/tests/generic/test-sympos-vmlinux.sh + create mode 100755 tools/objtool/tests/generic/test-sympos.sh + create mode 100755 tools/objtool/tests/generic/test-symvers-parse-error.sh + create mode 100755 tools/objtool/tests/generic/test-thinlto-ambiguity.sh + create mode 100755 tools/objtool/tests/generic/test-thinlto-local.sh + create mode 100755 tools/objtool/tests/generic/test-ubsan-noise.sh + create mode 100644 tools/objtool/tests/lib.sh + create mode 100755 tools/objtool/tests/run-tests.sh + create mode 100644 tools/objtool/tests/x86/fixtures/alt_annotate.c + create mode 100644 tools/objtool/tests/x86/fixtures/checksum_alt.c + create mode 100644 tools/objtool/tests/x86/fixtures/empty_alternative.c + create mode 100644 tools/objtool/tests/x86/fixtures/kcfi.c + create mode 100644 tools/objtool/tests/x86/fixtures/special_sections.c + create mode 100644 tools/objtool/tests/x86/fixtures/static_call_no_key.c + create mode 100755 tools/objtool/tests/x86/test-alt-annotation.sh + create mode 100755 tools/objtool/tests/x86/test-checksum-alt.sh + create mode 100755 tools/objtool/tests/x86/test-empty-alternative.sh + create mode 100755 tools/objtool/tests/x86/test-kcfi.sh + create mode 100755 tools/objtool/tests/x86/test-manual-klp-static-call.sh + create mode 100755 tools/objtool/tests/x86/test-special-sections.sh + create mode 100644 tools/testing/selftests/x86/sigframe_fpu_portability.c +Merging kexec/kexec-next (9d0b028715b00 Merge branch 'kexec-7.4' into kexec-next) +$ git merge -m Merge branch 'kexec-next' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git kexec/kexec-next +Auto-merging include/linux/mm.h +Auto-merging mm/memory-failure.c +Merge made by the 'ort' strategy. + arch/riscv/kernel/kexec_elf.c | 2 +- + include/linux/kexec.h | 15 --------------- + include/linux/mm.h | 14 ++++++++++++++ + kernel/kexec_core.c | 10 ++++++++++ + kernel/kexec_file.c | 22 ++++++++++++++++++++-- + mm/memory-failure.c | 40 ++++++++++++++++++++++++++++++++++++++++ + 6 files changed, 85 insertions(+), 18 deletions(-) +Merging liveupdate/next (5221d141653a8 Merge branch 'kho-bootmem' into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git liveupdate/next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging MAINTAINERS +Auto-merging drivers/pci/pci.h +Auto-merging drivers/pci/probe.c +Auto-merging drivers/pci/quirks.c +Auto-merging include/linux/memblock.h +Auto-merging include/linux/pci.h +Auto-merging mm/Kconfig +Auto-merging mm/memblock.c +CONFLICT (content): Merge conflict in mm/memblock.c +Auto-merging mm/mm_init.c +CONFLICT (content): Merge conflict in mm/mm_init.c +Resolved 'mm/memblock.c' using previous resolution. +Resolved 'mm/mm_init.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 39f8a23bd04b6] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/liveupdate/linux.git +$ git diff -M --stat --summary HEAD^.. + Documentation/PCI/index.rst | 1 + + Documentation/PCI/liveupdate.rst | 35 + + Documentation/admin-guide/kernel-parameters.txt | 22 +- + Documentation/admin-guide/mm/kho.rst | 18 +- + Documentation/core-api/kho/index.rst | 36 +- + Documentation/core-api/liveupdate.rst | 5 + + MAINTAINERS | 15 + + arch/x86/boot/compressed/kaslr.c | 12 +- + arch/x86/include/uapi/asm/setup_data.h | 4 +- + arch/x86/kernel/e820.c | 11 +- + arch/x86/kernel/kexec-bzimage64.c | 6 +- + arch/x86/kernel/setup.c | 2 +- + arch/x86/realmode/init.c | 2 +- + drivers/firmware/efi/efi-init.c | 6 +- + drivers/of/fdt.c | 8 +- + drivers/of/kexec.c | 16 +- + drivers/pci/Kconfig | 15 + + drivers/pci/Makefile | 1 + + drivers/pci/liveupdate.c | 1015 +++++++++++++++++++++++ + drivers/pci/liveupdate.h | 62 ++ + drivers/pci/pci-driver.c | 9 +- + drivers/pci/pci.c | 79 +- + drivers/pci/pci.h | 5 + + drivers/pci/probe.c | 16 +- + drivers/pci/quirks.c | 58 +- + include/asm-generic/kexec_handover.h | 2 +- + include/linux/kexec.h | 2 +- + include/linux/kexec_handover.h | 16 +- + include/linux/kho/abi/pci.h | 66 ++ + include/linux/liveupdate.h | 22 + + include/linux/memblock.h | 30 +- + include/linux/pci.h | 7 + + include/linux/pci_liveupdate.h | 72 ++ + kernel/kexec_file.c | 4 +- + kernel/liveupdate/Kconfig | 1 - + kernel/liveupdate/kexec_handover.c | 347 ++++---- + kernel/liveupdate/kexec_handover_debugfs.c | 24 +- + kernel/liveupdate/kexec_handover_internal.h | 4 +- + kernel/liveupdate/luo_file.c | 84 ++ + kernel/liveupdate/luo_internal.h | 17 + + mm/Kconfig | 4 - + mm/memblock.c | 56 +- + mm/memfd_luo.c | 3 +- + mm/mm_init.c | 4 +- + tools/testing/memblock/internal.h | 2 +- + 45 files changed, 1871 insertions(+), 355 deletions(-) + create mode 100644 Documentation/PCI/liveupdate.rst + create mode 100644 drivers/pci/liveupdate.c + create mode 100644 drivers/pci/liveupdate.h + create mode 100644 include/linux/kho/abi/pci.h + create mode 100644 include/linux/pci_liveupdate.h +Merging clockevents/timers/drivers/next (1b8b356b4b06e clocksource/drivers/armada: Unwind timer clock on init failure) +$ git merge -m Merge branch 'timers/drivers/next' of https://git.kernel.org/pub/scm/linux/kernel/git/daniel.lezcano/linux.git clockevents/timers/drivers/next +Auto-merging drivers/clocksource/timer-ti-dm.c +Auto-merging drivers/pwm/pwm-samsung.c +Merge made by the 'ort' strategy. +Merging edac/edac-for-next (898e6a2c5ce0e Merge ras/edac-drivers into for-next) +$ git merge -m Merge branch 'edac-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ras/ras.git edac/edac-for-next +Auto-merging MAINTAINERS +Auto-merging arch/x86/include/asm/msr-index.h +Auto-merging drivers/base/cacheinfo.c +Merge made by the 'ort' strategy. + MAINTAINERS | 26 +++- + arch/x86/include/asm/mce.h | 5 + + arch/x86/include/asm/msr-index.h | 2 + + drivers/base/cacheinfo.c | 17 +++ + drivers/edac/Kconfig | 22 +++ + drivers/edac/Makefile | 5 +- + drivers/edac/altera_edac.c | 67 ++++----- + drivers/edac/bluefield_edac.c | 4 +- + drivers/edac/dummy_edac.c | 191 +++++++++++++++++++++++++ + drivers/edac/i10nm_base.c | 2 + + drivers/edac/ie31200_edac.c | 4 +- + drivers/edac/intel-bff.c | 295 +++++++++++++++++++++++++++++++++++++++ + drivers/edac/loongson_edac.c | 4 +- + drivers/edac/skx_base.c | 2 +- + drivers/edac/versalnet_edac.c | 1 + + include/linux/cacheinfo.h | 12 +- + 16 files changed, 603 insertions(+), 56 deletions(-) + create mode 100644 drivers/edac/dummy_edac.c + create mode 100644 drivers/edac/intel-bff.c +Merging ftrace/for-next (6dc993d4520fd tools/bootconfig: remove unused PAGE_SIZE macro) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/trace/linux-trace.git ftrace/for-next +Auto-merging lib/bootconfig.c +Auto-merging tools/bootconfig/main.c +Merge made by the 'ort' strategy. + include/linux/bootconfig.h | 7 +------ + lib/bootconfig.c | 22 +++++++++------------- + tools/bootconfig/main.c | 2 -- + 3 files changed, 10 insertions(+), 21 deletions(-) +$ git am -3 ../patches/0001-ftrace-Fix-semantic-conflict-with-mm-tree.patch +Applying: ftrace: Fix semantic conflict with mm tree +Using index info to reconstruct a base tree... +M kernel/trace/trace_printk.c +Falling back to patching base and 3-way merge... +Auto-merging kernel/trace/trace_printk.c +No changes -- Patch already applied. +Merging rcu/next (aebf6c4777d53 Merge branches 'misc.2026.09.18a' and 'torture.2026.09.18a' into HEAD) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/rcu/linux rcu/next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Merge made by the 'ort' strategy. + Documentation/RCU/stallwarn.rst | 22 +- + Documentation/admin-guide/kernel-parameters.txt | 23 +- + include/linux/rcuref.h | 2 +- + include/linux/srcu.h | 83 +++- + include/linux/srcutiny.h | 18 +- + include/linux/srcutree.h | 19 +- + kernel/rcu/Kconfig | 6 + + kernel/rcu/rcu.h | 14 + + kernel/rcu/rcutorture.c | 208 ++++++++- + kernel/rcu/srcutiny.c | 182 +++++++- + kernel/rcu/srcutree.c | 464 +++++++++++++++++++-- + kernel/rcu/tasks.h | 15 +- + kernel/rcu/tiny.c | 127 +++++- + kernel/rcu/tree.c | 164 +++++++- + kernel/rcu/tree.h | 6 + + kernel/rcu/tree_exp.h | 2 +- + kernel/rcu/tree_nocb.h | 3 +- + kernel/rcu/tree_stall.h | 35 +- + .../testing/selftests/bpf/prog_tests/rcu_reentry.c | 93 +++++ + tools/testing/selftests/bpf/progs/rcu_reentry.c | 51 +++ + .../testing/selftests/rcutorture/bin/kvm-remote.sh | 22 +- + tools/testing/selftests/rcutorture/bin/torture.sh | 24 ++ + 22 files changed, 1430 insertions(+), 153 deletions(-) + create mode 100644 tools/testing/selftests/bpf/prog_tests/rcu_reentry.c + create mode 100644 tools/testing/selftests/bpf/progs/rcu_reentry.c +Merging paulmck/non-rcu/next (c87605b21fdd4 Merge branches 'csd-lock.2026.09.03a', 'hazptr.2026.09.18a', 'nmi.2026.09.17a' and 'usb-mtu3.2026.09.29a' into HEAD) +$ git merge -m Merge branch 'non-rcu/next' of https://git.kernel.org/pub/scm/linux/kernel/git/paulmck/linux-rcu.git paulmck/non-rcu/next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging init/main.c +Auto-merging kernel/Makefile +Auto-merging kernel/sched/core.c +Auto-merging lib/Kconfig.debug +Auto-merging tools/testing/selftests/rcutorture/bin/torture.sh +Merge made by the 'ort' strategy. + Documentation/admin-guide/kernel-parameters.txt | 96 +++ + arch/x86/kernel/nmi.c | 21 +- + drivers/usb/mtu3/mtu3_trace.h | 4 +- + include/linux/hazptr.h | 325 +++++++ + include/linux/torture.h | 6 +- + init/main.c | 2 + + kernel/Makefile | 2 +- + kernel/hazptr.c | 284 +++++++ + kernel/rcu/Kconfig.debug | 26 + + kernel/rcu/Makefile | 1 + + kernel/rcu/hazptrtorture.c | 946 +++++++++++++++++++++ + kernel/rcu/refscale.c | 43 + + kernel/rcu/update.c | 3 +- + kernel/sched/core.c | 2 + + kernel/smp.c | 77 +- + kernel/torture.c | 38 +- + lib/Kconfig.debug | 12 + + lib/Makefile | 1 + + lib/test_csd_lock.c | 173 ++++ + tools/testing/selftests/rcutorture/bin/kvm.sh | 10 +- + tools/testing/selftests/rcutorture/bin/torture.sh | 26 +- + .../selftests/rcutorture/configs/hazptr/CFLIST | 2 + + .../selftests/rcutorture/configs/hazptr/CFcommon | 2 + + .../selftests/rcutorture/configs/hazptr/NOPREEMPT | 19 + + .../rcutorture/configs/hazptr/NOPREEMPT.boot | 1 + + .../selftests/rcutorture/configs/hazptr/PREEMPT | 16 + + .../rcutorture/configs/hazptr/ver_functions.sh | 40 + + 27 files changed, 2140 insertions(+), 38 deletions(-) + create mode 100644 include/linux/hazptr.h + create mode 100644 kernel/hazptr.c + create mode 100644 kernel/rcu/hazptrtorture.c + create mode 100644 lib/test_csd_lock.c + create mode 100644 tools/testing/selftests/rcutorture/configs/hazptr/CFLIST + create mode 100644 tools/testing/selftests/rcutorture/configs/hazptr/CFcommon + create mode 100644 tools/testing/selftests/rcutorture/configs/hazptr/NOPREEMPT + create mode 100644 tools/testing/selftests/rcutorture/configs/hazptr/NOPREEMPT.boot + create mode 100644 tools/testing/selftests/rcutorture/configs/hazptr/PREEMPT + create mode 100644 tools/testing/selftests/rcutorture/configs/hazptr/ver_functions.sh +Merging kvm/next (d4b7fb647204f Merge branch 'kvm-xen-longmode' into HEAD) +$ git merge -m Merge branch 'next' of git://git.kernel.org/pub/scm/virt/kvm/kvm.git kvm/next +Auto-merging arch/x86/include/asm/kvm_host.h +Auto-merging include/linux/kvm_host.h +Auto-merging virt/kvm/kvm_main.c +Merge made by the 'ort' strategy. + arch/x86/include/asm/kvm_host.h | 3 +- + arch/x86/kvm/xen.c | 212 ++++++++++++++++++++++++---------------- + arch/x86/kvm/xen.h | 5 + + include/linux/kvm_host.h | 2 + + virt/kvm/kvm_main.c | 10 ++ + virt/kvm/pfncache.c | 18 ++-- + 6 files changed, 157 insertions(+), 93 deletions(-) +$ git am -3 ../patches/kvm-x86-static-cpu-has +Applying: Signed-off-by: Mark Brown +Using index info to reconstruct a base tree... +M arch/x86/kvm/msrs.c +Falling back to patching base and 3-way merge... +Auto-merging arch/x86/kvm/msrs.c +No changes -- Patch already applied. +Merging kvm-arm/next (8c00199d322b9 Merge branch kvm-arm64/immutable-idregs-7.4 into kvmarm-master/next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/kvmarm/kvmarm.git kvm-arm/next +Auto-merging arch/arm64/include/asm/sysreg.h +Auto-merging arch/arm64/kernel/pi/Makefile +Auto-merging arch/arm64/kvm/arm.c +Auto-merging arch/arm64/kvm/hyp/include/nvhe/pkvm.h +Auto-merging arch/arm64/kvm/hyp/nvhe/hyp-main.c +Auto-merging arch/arm64/kvm/hyp/nvhe/pkvm.c +Auto-merging arch/arm64/kvm/mmu.c +Auto-merging arch/arm64/kvm/vgic/vgic-init.c +Auto-merging arch/arm64/kvm/vgic/vgic-its.c +Auto-merging arch/arm64/kvm/vgic/vgic.c +Auto-merging arch/s390/kvm/s390/s390.c +Auto-merging arch/x86/kvm/mmu/mmu.c +Auto-merging arch/x86/kvm/vmx/tdx.c +Auto-merging drivers/irqchip/irq-gic-v5.c +Auto-merging include/linux/kvm_host.h +Auto-merging tools/testing/selftests/kvm/Makefile.kvm +Auto-merging virt/kvm/kvm_main.c +Merge made by the 'ort' strategy. + Documentation/virt/kvm/api.rst | 40 +- + Documentation/virt/kvm/devices/arm-vgic-v5.rst | 271 ++- + Documentation/virt/kvm/devices/vcpu.rst | 2 +- + arch/arm64/include/asm/esr.h | 29 + + arch/arm64/include/asm/kvm_asm.h | 3 + + arch/arm64/include/asm/kvm_emulate.h | 60 +- + arch/arm64/include/asm/kvm_hcall.h | 247 +++ + arch/arm64/include/asm/kvm_host.h | 81 +- + arch/arm64/include/asm/kvm_hyp.h | 3 + + arch/arm64/include/asm/kvm_mmu.h | 16 +- + arch/arm64/include/asm/kvm_nested.h | 7 + + arch/arm64/include/asm/kvm_pgtable.h | 7 +- + arch/arm64/include/asm/kvm_pkvm.h | 2 +- + arch/arm64/include/asm/stacktrace/nvhe.h | 10 +- + arch/arm64/include/asm/sysreg.h | 13 +- + arch/arm64/include/uapi/asm/kvm.h | 15 + + arch/arm64/kernel/pi/Makefile | 1 + + arch/arm64/kernel/vmlinux.lds.S | 2 +- + arch/arm64/kvm/Kconfig | 1 + + arch/arm64/kvm/Makefile | 3 +- + arch/arm64/kvm/arm.c | 13 +- + arch/arm64/kvm/hyp/entry.S | 2 +- + arch/arm64/kvm/hyp/include/nvhe/pkvm.h | 7 +- + arch/arm64/kvm/hyp/include/nvhe/trace.h | 5 +- + arch/arm64/kvm/hyp/nvhe/Makefile | 1 + + arch/arm64/kvm/hyp/nvhe/events.c | 2 + + arch/arm64/kvm/hyp/nvhe/ffa.c | 220 ++- + arch/arm64/kvm/hyp/nvhe/hyp-main.c | 485 +++-- + arch/arm64/kvm/hyp/nvhe/hyp.lds.S | 1 - + arch/arm64/kvm/hyp/nvhe/mem_protect.c | 10 +- + arch/arm64/kvm/hyp/nvhe/mm.c | 2 +- + arch/arm64/kvm/hyp/nvhe/pkvm.c | 12 +- + arch/arm64/kvm/hyp/nvhe/stacktrace.c | 3 +- + arch/arm64/kvm/hyp/nvhe/trace.c | 6 +- + arch/arm64/kvm/hyp/pgtable.c | 5 +- + arch/arm64/kvm/hyp/vgic-v5-sr.c | 45 + + arch/arm64/kvm/hyp_trace.c | 2 +- + arch/arm64/kvm/hypercalls.c | 4 + + arch/arm64/kvm/mmu.c | 420 ++++- + arch/arm64/kvm/nested.c | 151 +- + arch/arm64/kvm/pauth.c | 2 +- + arch/arm64/kvm/ptdump.c | 79 +- + arch/arm64/kvm/stacktrace.c | 3 - + arch/arm64/kvm/sys_regs.c | 63 +- + arch/arm64/kvm/sys_regs.h | 2 +- + arch/arm64/kvm/vgic-sys-reg-v5.c | 519 ++++++ + arch/arm64/kvm/vgic/vgic-init.c | 160 +- + arch/arm64/kvm/vgic/vgic-irqfd.c | 18 +- + arch/arm64/kvm/vgic/vgic-irs-v5.c | 1217 +++++++++++++ + arch/arm64/kvm/vgic/vgic-its.c | 6 +- + arch/arm64/kvm/vgic/vgic-kvm-device.c | 272 ++- + arch/arm64/kvm/vgic/vgic-mmio.c | 6 + + arch/arm64/kvm/vgic/vgic-mmio.h | 2 + + arch/arm64/kvm/vgic/vgic-v5-tables.c | 1872 ++++++++++++++++++++ + arch/arm64/kvm/vgic/vgic-v5-tables.h | 140 ++ + arch/arm64/kvm/vgic/vgic-v5.c | 1152 +++++++++++- + arch/arm64/kvm/vgic/vgic.c | 39 +- + arch/arm64/kvm/vgic/vgic.h | 21 + + arch/s390/kvm/s390/s390.c | 11 +- + arch/x86/kvm/mmu/mmu.c | 11 +- + arch/x86/kvm/vmx/tdx.c | 2 +- + drivers/irqchip/irq-gic-v5-irs.c | 19 +- + drivers/irqchip/irq-gic-v5.c | 111 +- + include/kvm/arm_vgic.h | 213 ++- + include/linux/irqchip/arm-gic-v5.h | 254 ++- + include/linux/irqchip/arm-vgic-info.h | 5 + + include/linux/kvm_host.h | 1 + + include/linux/trace_remote_event.h | 2 + + tools/arch/arm64/include/uapi/asm/kvm.h | 15 + + tools/testing/selftests/kvm/Makefile.kvm | 2 + + .../testing/selftests/kvm/arm64/debug-exceptions.c | 2 +- + .../testing/selftests/kvm/arm64/external_aborts.c | 3 +- + tools/testing/selftests/kvm/arm64/hypercalls.c | 31 +- + .../selftests/kvm/arm64/nv_pre_fault_memory_test.c | 158 ++ + tools/testing/selftests/kvm/arm64/sea_to_user.c | 75 +- + tools/testing/selftests/kvm/arm64/set_id_regs.c | 106 +- + tools/testing/selftests/kvm/arm64/vgic_v5.c | 1849 ++++++++++++++++++- + tools/testing/selftests/kvm/include/arm64/gic_v5.h | 105 ++ + tools/testing/selftests/kvm/include/test_util.h | 1 + + tools/testing/selftests/kvm/lib/test_util.c | 15 + + .../testing/selftests/kvm/pre_fault_memory_test.c | 150 +- + virt/kvm/kvm_main.c | 10 +- + 82 files changed, 10179 insertions(+), 754 deletions(-) + create mode 100644 arch/arm64/include/asm/kvm_hcall.h + create mode 100644 arch/arm64/kvm/vgic-sys-reg-v5.c + create mode 100644 arch/arm64/kvm/vgic/vgic-irs-v5.c + create mode 100644 arch/arm64/kvm/vgic/vgic-v5-tables.c + create mode 100644 arch/arm64/kvm/vgic/vgic-v5-tables.h + create mode 100644 tools/testing/selftests/kvm/arm64/nv_pre_fault_memory_test.c +Merging kvms390/next (044ae0767d8cc KVM: s390: Kick PV cpus at the right time for service irqs) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/kvms390/linux.git kvms390/next +Auto-merging arch/s390/kernel/uv.c +Auto-merging arch/s390/kvm/s390/interrupt.c +Auto-merging arch/s390/kvm/s390/s390.c +Auto-merging arch/s390/kvm/s390/s390.h +Auto-merging tools/testing/selftests/kvm/Makefile.kvm +Merge made by the 'ort' strategy. + Documentation/virt/kvm/devices/vm.rst | 18 ++++ + arch/s390/kernel/uv.c | 41 +++---- + arch/s390/kvm/s390/intercept.c | 32 +++--- + arch/s390/kvm/s390/interrupt.c | 130 ++++++++++++++++++----- + arch/s390/kvm/s390/priv.c | 19 ++-- + arch/s390/kvm/s390/s390.c | 12 +-- + arch/s390/kvm/s390/s390.h | 1 + + tools/testing/selftests/kvm/Makefile.kvm | 1 + + tools/testing/selftests/kvm/s390/irq_injection.c | 74 +++++++++++++ + 9 files changed, 257 insertions(+), 71 deletions(-) + create mode 100644 tools/testing/selftests/kvm/s390/irq_injection.c +Merging kvm-ppc/topic/ppc-kvm (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'topic/ppc-kvm' of https://git.kernel.org/pub/scm/linux/kernel/git/powerpc/linux.git kvm-ppc/topic/ppc-kvm +Already up to date. +Merging kvm-riscv/riscv_kvm_next (41e81f7e3ef96 RISC-V: KVM: Fix HSM hart status error propagation) +$ git merge -m Merge branch 'riscv_kvm_next' of https://github.com/kvm-riscv/linux.git kvm-riscv/riscv_kvm_next +Already up to date. +Merging kvm-x86/next (b378201ccd528 Merge branch 'vmx') + 9b2f8146fcef9 ("KVM: selftests: Extend nested x2APIC test to validate disabling x2APIC virt") + a4bab1c12fd52 ("KVM: selftests: Extend nested x2APIC test to validate using eVMCS for vmcs12") +$ git merge -m Merge branch 'next' of https://github.com/kvm-x86/linux.git kvm-x86/next +Auto-merging Documentation/admin-guide/kernel-parameters.txt +Auto-merging Documentation/virt/kvm/api.rst +Auto-merging arch/arm64/kvm/mmu.c +Auto-merging arch/arm64/kvm/nested.c +Auto-merging arch/x86/include/asm/cpufeatures.h +Auto-merging arch/x86/include/asm/kvm-x86-ops.h +Auto-merging arch/x86/include/asm/kvm_host.h +Auto-merging arch/x86/kernel/cpu/scattered.c +Auto-merging arch/x86/kvm/mmu/mmu.c +Auto-merging arch/x86/kvm/svm/sev.c +Auto-merging arch/x86/kvm/vmx/tdx.c +Auto-merging arch/x86/kvm/vmx/vmx.c +Auto-merging arch/x86/kvm/vmx/vmx.h +Auto-merging include/linux/kvm_host.h +Auto-merging tools/testing/selftests/kvm/Makefile.kvm +Auto-merging tools/testing/selftests/kvm/arm64/hypercalls.c +Auto-merging tools/testing/selftests/kvm/include/test_util.h +Auto-merging tools/testing/selftests/kvm/lib/test_util.c +Auto-merging tools/testing/selftests/kvm/x86/nested_x2apic_test.c +CONFLICT (content): Merge conflict in tools/testing/selftests/kvm/x86/nested_x2apic_test.c +Auto-merging virt/kvm/guest_memfd.c +Auto-merging virt/kvm/kvm_main.c +Recorded preimage for 'tools/testing/selftests/kvm/x86/nested_x2apic_test.c' +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +Recorded resolution for 'tools/testing/selftests/kvm/x86/nested_x2apic_test.c'. +[master 95683d8e36322] Merge branch 'next' of https://github.com/kvm-x86/linux.git +$ git diff -M --stat --summary HEAD^.. + Documentation/admin-guide/kernel-parameters.txt | 26 + + Documentation/virt/kvm/api.rst | 116 +++- + .../virt/kvm/x86/amd-memory-encryption.rst | 22 +- + Documentation/virt/kvm/x86/hypercalls.rst | 2 +- + Documentation/virt/kvm/x86/intel-tdx.rst | 4 + + arch/arm64/kvm/mmu.c | 4 +- + arch/arm64/kvm/nested.c | 4 +- + arch/x86/include/asm/cpufeatures.h | 1 + + arch/x86/include/asm/kvm-x86-ops.h | 3 +- + arch/x86/include/asm/kvm_host.h | 32 +- + arch/x86/include/asm/svm.h | 18 +- + arch/x86/include/asm/vmx.h | 8 + + arch/x86/include/uapi/asm/svm.h | 2 + + arch/x86/kernel/cpu/scattered.c | 1 + + arch/x86/kvm/Kconfig | 15 +- + arch/x86/kvm/cpuid.c | 17 +- + arch/x86/kvm/cpuid.h | 1 - + arch/x86/kvm/hyperv.c | 9 +- + arch/x86/kvm/kvm_emulate.h | 4 +- + arch/x86/kvm/lapic.c | 4 +- + arch/x86/kvm/lapic.h | 2 +- + arch/x86/kvm/mmu/mmu.c | 183 ++++-- + arch/x86/kvm/mmu/paging_tmpl.h | 21 +- + arch/x86/kvm/msrs.c | 3 +- + arch/x86/kvm/pmu.c | 14 +- + arch/x86/kvm/regs.c | 8 +- + arch/x86/kvm/reverse_cpuid.h | 2 +- + arch/x86/kvm/svm/avic.c | 22 +- + arch/x86/kvm/svm/nested.c | 20 + + arch/x86/kvm/svm/sev.c | 112 ++-- + arch/x86/kvm/svm/svm.c | 138 ++++- + arch/x86/kvm/svm/svm.h | 2 + + arch/x86/kvm/vmx/common.h | 90 ++- + arch/x86/kvm/vmx/main.c | 153 ++++- + arch/x86/kvm/vmx/nested.c | 60 +- + arch/x86/kvm/vmx/posted_intr.c | 24 +- + arch/x86/kvm/vmx/posted_intr.h | 8 +- + arch/x86/kvm/vmx/tdx.c | 141 +++-- + arch/x86/kvm/vmx/vmx.c | 302 +++------- + arch/x86/kvm/vmx/vmx.h | 46 +- + arch/x86/kvm/vmx/x86_ops.h | 6 +- + arch/x86/kvm/x86.c | 577 +++++++++--------- + arch/x86/kvm/x86.h | 17 +- + include/linux/kvm_host.h | 89 +-- + include/trace/events/kvm.h | 6 +- + include/uapi/linux/kvm.h | 16 + + tools/testing/selftests/kvm/Makefile.kvm | 2 +- + tools/testing/selftests/kvm/arm64/hypercalls.c | 2 +- + .../testing/selftests/kvm/arm64/page_fault_test.c | 18 +- + .../testing/selftests/kvm/arm64/vgic_lpi_stress.c | 21 +- + tools/testing/selftests/kvm/demand_paging_test.c | 3 +- + tools/testing/selftests/kvm/guest_memfd_test.c | 2 +- + tools/testing/selftests/kvm/include/kvm_util.h | 245 +++++++- + tools/testing/selftests/kvm/include/numaif.h | 43 ++ + tools/testing/selftests/kvm/include/test_util.h | 36 +- + tools/testing/selftests/kvm/include/x86/evmcs.h | 78 +-- + .../testing/selftests/kvm/include/x86/processor.h | 13 +- + tools/testing/selftests/kvm/include/x86/smm.h | 2 +- + tools/testing/selftests/kvm/include/x86/vmx.h | 186 +++--- + tools/testing/selftests/kvm/lib/arm64/processor.c | 4 +- + tools/testing/selftests/kvm/lib/elf.c | 48 +- + tools/testing/selftests/kvm/lib/io.c | 157 ----- + tools/testing/selftests/kvm/lib/kvm_util.c | 350 +++++++---- + .../selftests/kvm/lib/loongarch/processor.c | 5 +- + tools/testing/selftests/kvm/lib/riscv/processor.c | 4 +- + tools/testing/selftests/kvm/lib/s390/processor.c | 7 +- + tools/testing/selftests/kvm/lib/test_util.c | 7 - + tools/testing/selftests/kvm/lib/userfaultfd_util.c | 6 +- + tools/testing/selftests/kvm/lib/x86/memstress.c | 8 +- + tools/testing/selftests/kvm/lib/x86/processor.c | 23 +- + tools/testing/selftests/kvm/lib/x86/vmx.c | 159 +++-- + tools/testing/selftests/kvm/memslot_perf_test.c | 5 +- + tools/testing/selftests/kvm/s390/cmma_test.c | 19 +- + tools/testing/selftests/kvm/s390/irq_routing.c | 2 +- + .../testing/selftests/kvm/set_memory_region_test.c | 8 +- + tools/testing/selftests/kvm/steal_time.c | 20 +- + tools/testing/selftests/kvm/x86/amx_test.c | 2 +- + tools/testing/selftests/kvm/x86/aperfmperf_test.c | 10 +- + tools/testing/selftests/kvm/x86/cpuid_test.c | 2 +- + .../selftests/kvm/x86/evmcs_smm_controls_test.c | 6 +- + .../testing/selftests/kvm/x86/feature_msrs_test.c | 2 +- + .../testing/selftests/kvm/x86/fix_hypercall_test.c | 2 +- + .../kvm/x86/guest_memfd_conversions_test.c | 512 ++++++++++++++++ + tools/testing/selftests/kvm/x86/hyperv_clock.c | 2 +- + tools/testing/selftests/kvm/x86/hyperv_evmcs.c | 87 +-- + tools/testing/selftests/kvm/x86/hyperv_svm_test.c | 2 +- + tools/testing/selftests/kvm/x86/kvm_buslock_test.c | 8 +- + .../selftests/kvm/x86/nested_close_kvm_test.c | 6 +- + .../selftests/kvm/x86/nested_dirty_log_test.c | 8 +- + .../selftests/kvm/x86/nested_emulation_test.c | 23 +- + .../selftests/kvm/x86/nested_exceptions_test.c | 33 +- + .../selftests/kvm/x86/nested_invalid_cr3_test.c | 14 +- + .../selftests/kvm/x86/nested_tdp_fault_test.c | 16 +- + .../selftests/kvm/x86/nested_tsc_adjust_test.c | 10 +- + .../selftests/kvm/x86/nested_tsc_scaling_test.c | 12 +- + .../testing/selftests/kvm/x86/nested_x2apic_test.c | 2 +- + .../testing/selftests/kvm/x86/nx_huge_pages_test.c | 3 +- + .../kvm/x86/private_mem_conversions_test.c | 66 ++- + .../selftests/kvm/x86/private_mem_kvm_exits_test.c | 36 +- + .../kvm/x86/save_restore_pf_stress_test.c | 14 +- + tools/testing/selftests/kvm/x86/set_boot_cpu_id.c | 2 +- + tools/testing/selftests/kvm/x86/set_sregs_test.c | 11 + + tools/testing/selftests/kvm/x86/sev_dbg_test.c | 6 +- + tools/testing/selftests/kvm/x86/sev_smoke_test.c | 3 +- + .../kvm/x86/smaller_maxphyaddr_emulation_test.c | 9 +- + tools/testing/selftests/kvm/x86/smm_test.c | 4 +- + tools/testing/selftests/kvm/x86/state_test.c | 74 +-- + .../selftests/kvm/x86/triple_fault_event_test.c | 8 +- + tools/testing/selftests/kvm/x86/tsc_msrs_test.c | 2 +- + .../selftests/kvm/x86/vmx_apic_access_test.c | 22 +- + .../selftests/kvm/x86/vmx_apicv_updates_test.c | 16 +- + .../x86/vmx_exception_with_invalid_guest_state.c | 2 +- + .../kvm/x86/vmx_invalid_nested_guest_state.c | 12 +- + .../selftests/kvm/x86/vmx_nested_la57_state_test.c | 12 +- + .../selftests/kvm/x86/vmx_preemption_timer_test.c | 28 +- + tools/testing/selftests/kvm/x86/xapic_ipi_test.c | 78 +-- + virt/kvm/Kconfig | 3 - + virt/kvm/guest_memfd.c | 649 +++++++++++++++++---- + virt/kvm/guest_memfd.h | 19 +- + virt/kvm/kvm_main.c | 159 +++-- + 120 files changed, 3827 insertions(+), 2008 deletions(-) + delete mode 100644 tools/testing/selftests/kvm/lib/io.c + create mode 100644 tools/testing/selftests/kvm/x86/guest_memfd_conversions_test.c +$ git am -3 ../patches/0001-KVM-selftests-Fix-up-semantic-changes.patch +Applying: KVM: selftests: Fix up semantic changes +Using index info to reconstruct a base tree... +M tools/testing/selftests/kvm/lib/kvm_util.c +Falling back to patching base and 3-way merge... +Auto-merging tools/testing/selftests/kvm/lib/kvm_util.c +No changes -- Patch already applied. +$ git am -3 ../patches/0001-KVM-selftests-Fix-up-phy_pages_alloc-API-rework-inte.patch +Applying: KVM: selftests: Fix up phy_pages_alloc() API rework interaction +$ git reset HEAD^ +Unstaged changes after reset: +M tools/testing/selftests/kvm/arm64/vgic_its_save.c +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master 546bb0c1a1012] Merge branch 'next' of https://github.com/kvm-x86/linux.git + Date: Wed Sep 30 13:35:02 2026 +0100 +Merging xen-tip/linux-next (93f51579e7df2 Linux 7.3-rc4) +$ git merge -m Merge branch 'linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/xen/tip.git xen-tip/linux-next +Already up to date. +Merging percpu/for-next (8f0b4cce4481f Linux 6.19-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/dennis/percpu.git percpu/for-next +Already up to date. +Merging workqueues/for-next (fee1265a72572 Merge branch 'for-7.3-fixes' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tj/wq.git workqueues/for-next +Auto-merging kernel/workqueue.c +Merge made by the 'ort' strategy. + include/trace/events/workqueue.h | 146 ++++++++++ + kernel/workqueue.c | 603 ++++++++++++++++++++++++++------------- + 2 files changed, 547 insertions(+), 202 deletions(-) +Merging sched-ext/for-next (91a186b9e599f Merge branch 'for-7.4' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tj/sched_ext.git sched-ext/for-next +Auto-merging init/Kconfig +Auto-merging kernel/sched/ext/cid.c +Auto-merging kernel/sched/ext/ext.c +Auto-merging kernel/sched/ext/sub.c +Auto-merging kernel/sched/sched.h +Auto-merging tools/sched_ext/include/scx/common.bpf.h +Merge made by the 'ort' strategy. + Documentation/scheduler/sched-ext.rst | 13 + + include/linux/sched/ext.h | 38 +- + init/Kconfig | 2 - + kernel/sched/ext/cid.c | 67 +- + kernel/sched/ext/cid.h | 11 +- + kernel/sched/ext/ext.c | 896 ++++++++++++++++---- + kernel/sched/ext/ext.h | 32 +- + kernel/sched/ext/idle.c | 14 +- + kernel/sched/ext/inlines.h | 8 +- + kernel/sched/ext/internal.h | 105 ++- + kernel/sched/ext/sub.c | 110 +-- + kernel/sched/ext/sub.h | 38 +- + kernel/sched/ext/types.h | 28 +- + kernel/sched/sched.h | 12 +- + kernel/sched/syscalls.c | 2 + + tools/sched_ext/Makefile | 2 +- + tools/sched_ext/include/scx/common.bpf.h | 3 +- + tools/sched_ext/include/scx/compat.bpf.h | 14 + + tools/sched_ext/include/scx/compat.h | 17 +- + tools/sched_ext/include/scx/enum_defs.autogen.h | 7 + + tools/sched_ext/include/scx/enums.autogen.bpf.h | 12 + + tools/sched_ext/include/scx/enums.autogen.h | 4 + + tools/sched_ext/include/scx/enums_abi.autogen.h | 6 +- + tools/sched_ext/scx_flatcg.bpf.c | 2 +- + tools/sched_ext/scx_qmap.bpf.c | 79 +- + tools/sched_ext/scx_qmap.c | 13 +- + tools/sched_ext/scx_qmap.h | 1 + + tools/testing/selftests/sched_ext/.gitignore | 4 + + tools/testing/selftests/sched_ext/Makefile | 7 +- + .../testing/selftests/sched_ext/allowed_cpus.bpf.c | 24 +- + tools/testing/selftests/sched_ext/allowed_cpus.c | 123 ++- + tools/testing/selftests/sched_ext/config | 2 + + tools/testing/selftests/sched_ext/create_dsq.bpf.c | 35 + + tools/testing/selftests/sched_ext/create_dsq.c | 4 +- + .../testing/selftests/sched_ext/enq_blocked.bpf.c | 116 +++ + tools/testing/selftests/sched_ext/enq_blocked.c | 917 +++++++++++++++++++++ + tools/testing/selftests/sched_ext/enq_blocked.h | 28 + + tools/testing/selftests/sched_ext/hotplug.c | 34 +- + tools/testing/selftests/sched_ext/kick.bpf.c | 164 ++++ + tools/testing/selftests/sched_ext/kick.c | 519 ++++++++++++ + tools/testing/selftests/sched_ext/kick_test.h | 29 + + tools/testing/selftests/sched_ext/nohz_tick.bpf.c | 57 +- + tools/testing/selftests/sched_ext/nohz_tick.c | 207 ++++- + tools/testing/selftests/sched_ext/nohz_tick_test.h | 13 + + tools/testing/selftests/sched_ext/rt_stall.c | 80 +- + tools/testing/selftests/sched_ext/runner.c | 2 +- + .../selftests/sched_ext/test_modules/Makefile | 13 + + .../sched_ext/test_modules/scx_enq_blocked_test.c | 195 +++++ + tools/testing/selftests/sched_ext/util.c | 143 ++++ + tools/testing/selftests/sched_ext/util.h | 12 + + 50 files changed, 3836 insertions(+), 428 deletions(-) + create mode 100644 tools/testing/selftests/sched_ext/enq_blocked.bpf.c + create mode 100644 tools/testing/selftests/sched_ext/enq_blocked.c + create mode 100644 tools/testing/selftests/sched_ext/enq_blocked.h + create mode 100644 tools/testing/selftests/sched_ext/kick.bpf.c + create mode 100644 tools/testing/selftests/sched_ext/kick.c + create mode 100644 tools/testing/selftests/sched_ext/kick_test.h + create mode 100644 tools/testing/selftests/sched_ext/nohz_tick_test.h + create mode 100644 tools/testing/selftests/sched_ext/test_modules/Makefile + create mode 100644 tools/testing/selftests/sched_ext/test_modules/scx_enq_blocked_test.c +Merging drivers-x86/for-next (fe5030c8cc715 Merge branch 'platform-drivers-x86-intel-pmt' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/pdx86/platform-drivers-x86.git drivers-x86/for-next +Auto-merging MAINTAINERS +Auto-merging drivers/platform/x86/asus-laptop.c +Auto-merging drivers/platform/x86/hp/hp-wmi.c +Merge made by the 'ort' strategy. + Documentation/ABI/testing/debugfs-tpmi | 8 +- + .../admin-guide/laptops/thinkpad-acpi.rst | 26 +- + Documentation/wmi/devices/acer-wmi-battery.rst | 70 ++++ + Documentation/wmi/devices/bitland-mifs-wmi.rst | 64 ++- + MAINTAINERS | 3 +- + drivers/platform/arm64/acer-aspire1-ec.c | 2 +- + drivers/platform/arm64/huawei-gaokun-ec.c | 2 +- + drivers/platform/arm64/lenovo-thinkpad-t14s.c | 10 +- + drivers/platform/arm64/lenovo-yoga-c630.c | 2 +- + drivers/platform/mellanox/mlxbf-tmfifo.c | 1 - + drivers/platform/mellanox/mlxreg-hotplug.c | 4 +- + .../platform/surface/surface_aggregator_registry.c | 5 +- + drivers/platform/x86/Kconfig | 12 + + drivers/platform/x86/Makefile | 1 + + drivers/platform/x86/acer-wmi-battery.c | 450 +++++++++++++++++++++ + drivers/platform/x86/acer-wmi.c | 56 ++- + drivers/platform/x86/amd/pmc/pmc-quirks.c | 9 + + drivers/platform/x86/amd/pmc/pmc.c | 15 +- + drivers/platform/x86/amd/pmf/core.c | 1 + + drivers/platform/x86/amd/pmf/sps.c | 10 +- + drivers/platform/x86/amd/pmf/util.c | 1 + + drivers/platform/x86/asus-armoury.c | 29 +- + drivers/platform/x86/asus-armoury.h | 41 ++ + drivers/platform/x86/asus-laptop.c | 2 +- + drivers/platform/x86/asus-nb-wmi.c | 15 +- + drivers/platform/x86/asus-tf103c-dock.c | 2 +- + drivers/platform/x86/asus-wmi.c | 97 +++-- + drivers/platform/x86/asus-wmi.h | 1 - + drivers/platform/x86/bitland-mifs-wmi.c | 146 ++++--- + drivers/platform/x86/dell/alienware-wmi-wmax.c | 8 + + drivers/platform/x86/dell/dell-dw5826e-reset.c | 30 +- + drivers/platform/x86/dell/dell-wmi-aio.c | 200 ++++----- + drivers/platform/x86/hp/hp-bioscfg/bioscfg.c | 2 +- + drivers/platform/x86/hp/hp-wmi.c | 27 +- + drivers/platform/x86/hp/hp_accel.c | 2 +- + drivers/platform/x86/huawei-wmi.c | 1 + + drivers/platform/x86/intel/bxtwc_tmu.c | 5 +- + drivers/platform/x86/intel/bytcrc_pwrsrc.c | 2 +- + drivers/platform/x86/intel/crystal_cove_charger.c | 2 +- + drivers/platform/x86/intel/int0002_vgpio.c | 4 +- + drivers/platform/x86/intel/pmc/core.c | 1 + + drivers/platform/x86/intel/pmt/class.c | 35 +- + drivers/platform/x86/intel/pmt/class.h | 5 +- + drivers/platform/x86/intel/pmt/crashlog.c | 235 +++++++---- + drivers/platform/x86/intel/pmt/discovery.c | 2 +- + drivers/platform/x86/intel/pmt/telemetry.c | 3 + + drivers/platform/x86/intel/punit_ipc.c | 4 +- + .../x86/intel/speed_select_if/isst_if_common.c | 57 ++- + .../x86/intel/uncore-frequency/uncore-frequency.c | 1 + + drivers/platform/x86/intel/vsec.c | 19 +- + drivers/platform/x86/intel/vsec_tpmi.c | 2 +- + drivers/platform/x86/lenovo/ideapad-laptop.c | 4 + + drivers/platform/x86/lenovo/think-lmi.c | 2 +- + drivers/platform/x86/lenovo/thinkpad_acpi.c | 273 ++++++++----- + drivers/platform/x86/msi-laptop.c | 2 +- + drivers/platform/x86/panasonic-laptop.c | 14 +- + drivers/platform/x86/samsung-laptop.c | 2 +- + drivers/platform/x86/topstar-laptop.c | 6 +- + drivers/platform/x86/uniwill/uniwill-acpi.c | 2 +- + include/linux/intel_vsec.h | 14 +- + include/linux/platform_data/x86/asus-wmi.h | 4 +- + include/linux/string_choices.h | 6 + + include/uapi/linux/amd-pmf.h | 3 + + 63 files changed, 1525 insertions(+), 539 deletions(-) + create mode 100644 Documentation/wmi/devices/acer-wmi-battery.rst + create mode 100644 drivers/platform/x86/acer-wmi-battery.c +Merging chrome-platform/for-next (5859f97c6f404 platform/chrome: cros_ec_rpmsg: Fix repeated word 'from' in comment) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chrome-platform/linux.git chrome-platform/for-next +Merge made by the 'ort' strategy. + drivers/platform/chrome/cros_ec_ishtp.c | 2 +- + drivers/platform/chrome/cros_ec_proto.c | 43 +++++-- + drivers/platform/chrome/cros_ec_proto_test.c | 158 ++++++++++++++++++++++++- + drivers/platform/chrome/cros_ec_rpmsg.c | 2 +- + drivers/platform/chrome/cros_usbpd_notify.c | 4 +- + include/linux/platform_data/cros_ec_commands.h | 72 +++++++++++ + 6 files changed, 262 insertions(+), 19 deletions(-) +Merging chrome-platform-firmware/for-firmware-next (0e30b98a54599 firmware: coreboot: Use named initializers for acpi_device_id) +$ git merge -m Merge branch 'for-firmware-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chrome-platform/linux.git chrome-platform-firmware/for-firmware-next +Auto-merging MAINTAINERS +Auto-merging drivers/firmware/Kconfig +Auto-merging drivers/firmware/Makefile +Merge made by the 'ort' strategy. + MAINTAINERS | 20 +++--- + drivers/firmware/Kconfig | 2 +- + drivers/firmware/Makefile | 2 +- + drivers/firmware/{google => coreboot}/Kconfig | 74 ++++++++++++++++------ + drivers/firmware/{google => coreboot}/Makefile | 10 +-- + drivers/firmware/{google => coreboot}/cbmem.c | 0 + .../firmware/{google => coreboot}/coreboot_table.c | 4 +- + .../firmware/{google => coreboot}/coreboot_table.h | 0 + .../{google => coreboot}/framebuffer-coreboot.c | 0 + drivers/firmware/{google => coreboot}/gsmi.c | 0 + .../{google => coreboot}/memconsole-coreboot.c | 0 + .../{google => coreboot}/memconsole-x86-legacy.c | 0 + drivers/firmware/{google => coreboot}/memconsole.c | 0 + drivers/firmware/{google => coreboot}/memconsole.h | 6 +- + drivers/firmware/{google => coreboot}/vpd.c | 0 + drivers/firmware/{google => coreboot}/vpd_decode.c | 0 + drivers/firmware/{google => coreboot}/vpd_decode.h | 0 + 17 files changed, 78 insertions(+), 40 deletions(-) + rename drivers/firmware/{google => coreboot}/Kconfig (64%) + rename drivers/firmware/{google => coreboot}/Makefile (51%) + rename drivers/firmware/{google => coreboot}/cbmem.c (100%) + rename drivers/firmware/{google => coreboot}/coreboot_table.c (99%) + rename drivers/firmware/{google => coreboot}/coreboot_table.h (100%) + rename drivers/firmware/{google => coreboot}/framebuffer-coreboot.c (100%) + rename drivers/firmware/{google => coreboot}/gsmi.c (100%) + rename drivers/firmware/{google => coreboot}/memconsole-coreboot.c (100%) + rename drivers/firmware/{google => coreboot}/memconsole-x86-legacy.c (100%) + rename drivers/firmware/{google => coreboot}/memconsole.c (100%) + rename drivers/firmware/{google => coreboot}/memconsole.h (82%) + rename drivers/firmware/{google => coreboot}/vpd.c (100%) + rename drivers/firmware/{google => coreboot}/vpd_decode.c (100%) + rename drivers/firmware/{google => coreboot}/vpd_decode.h (100%) +Merging hsi/for-next (e81250ec6b692 hsi: omap_ssi_core: fix missing DMA mask setup for SSI controller device) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sre/linux-hsi.git hsi/for-next +Already up to date. +Merging leds-lj/for-leds-next (05b4738b0078f leds: trigger: Add led_trigger_notify_hw_control_changed() interface) +$ git merge -m Merge branch 'for-leds-next' of https://git.kernel.org/pub/scm/linux/kernel/git/lee/leds.git leds-lj/for-leds-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-class-led | 28 ++- + .../ABI/testing/sysfs-class-led-trigger-netdev | 3 + + .../devicetree/bindings/leds/adi,ltc3208.yaml | 183 ++++++++++++++++ + .../devicetree/bindings/leds/leds-cpcap.txt | 29 --- + .../bindings/leds/motorola,cpcap-leds.yaml | 42 ++++ + Documentation/leds/leds-class.rst | 72 +++++++ + Documentation/leds/leds-st1202.rst | 5 + + MAINTAINERS | 8 + + drivers/leds/Kconfig | 12 ++ + drivers/leds/Makefile | 1 + + drivers/leds/flash/leds-aat1290.c | 5 +- + drivers/leds/flash/leds-ktd2692.c | 10 +- + drivers/leds/led-class-flash.c | 21 +- + drivers/leds/led-class.c | 46 ++-- + drivers/leds/led-core.c | 5 +- + drivers/leds/led-triggers.c | 177 ++++++++++++++- + drivers/leds/leds-cros_ec.c | 6 + + drivers/leds/leds-is31fl32xx.c | 20 +- + drivers/leds/leds-lm3692x.c | 4 +- + drivers/leds/leds-lp8860.c | 5 +- + drivers/leds/leds-ltc3208.c | 234 ++++++++++++++++++++ + drivers/leds/leds-max77705.c | 3 +- + drivers/leds/leds-pca9532.c | 9 +- + drivers/leds/leds-pca963x.c | 5 +- + drivers/leds/leds-ss4200.c | 2 +- + drivers/leds/leds-st1202.c | 240 ++++++++++++++++----- + drivers/leds/leds-syscon.c | 9 +- + drivers/leds/leds-turris-omnia.c | 7 + + drivers/leds/leds.h | 8 +- + drivers/leds/rgb/leds-qcom-lpg.c | 38 ++-- + drivers/leds/trigger/Kconfig | 10 + + drivers/leds/trigger/ledtrig-netdev.c | 13 +- + include/linux/leds.h | 24 +++ + 33 files changed, 1089 insertions(+), 195 deletions(-) + create mode 100644 Documentation/devicetree/bindings/leds/adi,ltc3208.yaml + delete mode 100644 Documentation/devicetree/bindings/leds/leds-cpcap.txt + create mode 100644 Documentation/devicetree/bindings/leds/motorola,cpcap-leds.yaml + create mode 100644 drivers/leds/leds-ltc3208.c +Merging ipmi/for-next (89a312991dc6e Merge tag 'cifs-fixes-7.3-rc2' of https://git.manguebit.org/linux) +$ git merge -m Merge branch 'for-next' of https://github.com/cminyard/linux-ipmi.git ipmi/for-next +Already up to date. +Merging driver-core/driver-core-next (f1850e443b0e4 docs: admin-guide: Handle TAINT_FORCED_BIND when parsing /proc/sys/kernel/tainted) +$ git merge -m Merge branch 'driver-core-next' of https://git.kernel.org/pub/scm/linux/kernel/git/driver-core/driver-core.git driver-core/driver-core-next +Auto-merging drivers/base/bus.c +Auto-merging include/linux/panic.h +Auto-merging kernel/module/main.c +Auto-merging kernel/panic.c +Auto-merging lib/kobject_uevent.c +Auto-merging rust/kernel/auxiliary.rs +Merge made by the 'ort' strategy. + Documentation/admin-guide/tainted-kernels.rst | 54 ++++++++++++++------------- + arch/powerpc/perf/hv-24x7.c | 2 +- + arch/x86/kernel/cpu/mce/core.c | 6 +-- + drivers/base/bus.c | 3 ++ + drivers/base/core.c | 30 +++++++-------- + drivers/base/node.c | 17 ++++----- + drivers/base/platform.c | 19 +++++----- + drivers/base/property.c | 2 +- + drivers/base/transport_class.c | 2 +- + drivers/perf/alibaba_uncore_drw_pmu.c | 2 +- + drivers/perf/arm-cci.c | 2 +- + drivers/perf/arm-ccn.c | 2 +- + drivers/perf/arm_cspmu/arm_cspmu.h | 2 +- + drivers/perf/arm_dsu_pmu.c | 2 +- + drivers/perf/arm_spe_pmu.c | 6 +-- + drivers/perf/cxl_pmu.c | 10 ++--- + drivers/perf/fsl_imx8_ddr_perf.c | 2 +- + drivers/perf/fujitsu_uncore_pmu.c | 2 +- + drivers/perf/hisilicon/hisi_pcie_pmu.c | 2 +- + drivers/perf/hisilicon/hisi_uncore_pmu.h | 6 +-- + drivers/perf/hisilicon/hns3_pmu.c | 6 +-- + drivers/perf/nvidia_t410_c2c_pmu.c | 2 +- + drivers/perf/nvidia_t410_cmem_latency_pmu.c | 12 +++--- + drivers/perf/qcom_l3_pmu.c | 8 ++-- + drivers/perf/starfive_starlink_pmu.c | 2 +- + drivers/perf/xgene_pmu.c | 2 +- + include/linux/device.h | 14 +++---- + include/linux/kernfs.h | 2 +- + include/linux/module.h | 10 +++++ + include/linux/panic.h | 3 +- + include/trace/events/module.h | 3 +- + kernel/module/main.c | 17 +++++++-- + kernel/panic.c | 10 +++-- + lib/kobject_uevent.c | 2 +- + rust/helpers/dma.c | 5 +++ + rust/kernel/auxiliary.rs | 8 ++++ + rust/kernel/scatterlist.rs | 12 +++++- + tools/debugging/kernel-chktaint | 8 ++++ + 38 files changed, 180 insertions(+), 119 deletions(-) +Merging usb/usb-next (d58dffe9ee2c8 USB: core: amend usb_get_from_anchor() kernel-doc) +$ git merge -m Merge branch 'usb-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/usb.git usb/usb-next +Auto-merging Documentation/devicetree/bindings/usb/qcom,snps-dwc3.yaml +Auto-merging arch/arm/boot/dts/st/stm32mp131.dtsi +Auto-merging arch/arm/boot/dts/st/stm32mp151.dtsi +Auto-merging arch/arm64/boot/dts/st/stm32mp231.dtsi +Auto-merging arch/arm64/boot/dts/st/stm32mp251.dtsi +Auto-merging drivers/net/usb/r8152.c +Auto-merging drivers/usb/core/hub.c +Auto-merging drivers/usb/dwc2/hcd.c +Auto-merging drivers/usb/gadget/udc/lpc32xx_udc.c +Auto-merging drivers/usb/typec/anx7411.c +Auto-merging drivers/usb/typec/tcpm/tcpm.c +Auto-merging drivers/usb/typec/tipd/core.c +Merge made by the 'ort' strategy. + .../devicetree/bindings/dma/ti/ti,cppi41.yaml | 109 + + .../devicetree/bindings/phy/ti,am335x-usb-phy.yaml | 49 + + .../devicetree/bindings/usb/am33xx-usb.txt | 200 - + .../devicetree/bindings/usb/da8xx-usb.txt | 81 - + .../devicetree/bindings/usb/generic-ehci.yaml | 24 + + .../devicetree/bindings/usb/generic-ohci.yaml | 24 + + .../devicetree/bindings/usb/gpio-sbu-mux.yaml | 1 + + .../devicetree/bindings/usb/ohci-da8xx.txt | 23 - + .../devicetree/bindings/usb/qcom,snps-dwc3.yaml | 19 + + .../bindings/usb/ti,am335x-usb-ctrl-module.yaml | 38 + + .../devicetree/bindings/usb/ti,am33xx-usb.yaml | 103 + + .../devicetree/bindings/usb/ti,da830-musb.yaml | 113 + + .../devicetree/bindings/usb/ti,da830-ohci.yaml | 61 + + .../devicetree/bindings/usb/ti,hd3ss3220.yaml | 11 + + .../devicetree/bindings/usb/ti,musb-am33xx.yaml | 113 + + Documentation/usb/usbmon.rst | 2 +- + arch/arm/boot/dts/st/stm32mp131.dtsi | 4 +- + arch/arm/boot/dts/st/stm32mp151.dtsi | 4 +- + arch/arm64/boot/dts/st/stm32mp231.dtsi | 64 + + arch/arm64/boot/dts/st/stm32mp251.dtsi | 63 + + drivers/net/usb/r8152.c | 2 +- + drivers/usb/cdns3/core.c | 4 +- + drivers/usb/cdns3/drd.c | 4 +- + drivers/usb/chipidea/ci_hdrc_imx.c | 8 +- + drivers/usb/chipidea/otg_fsm.c | 2 +- + drivers/usb/chipidea/udc.c | 2 +- + drivers/usb/common/ulpi.c | 4 +- + drivers/usb/common/usb-conn-gpio.c | 8 +- + drivers/usb/core/driver.c | 10 +- + drivers/usb/core/hub.c | 2 +- + drivers/usb/core/urb.c | 4 +- + drivers/usb/core/usb.c | 2 +- + drivers/usb/dwc2/core.c | 2 +- + drivers/usb/dwc2/core.h | 2 +- + drivers/usb/dwc2/debugfs.c | 6 +- + drivers/usb/dwc2/gadget.c | 98 +- + drivers/usb/dwc2/hcd.c | 4 +- + drivers/usb/dwc2/hcd_ddma.c | 2 +- + drivers/usb/dwc3/debugfs.c | 6 +- + drivers/usb/dwc3/dwc3-google.c | 4 +- + drivers/usb/dwc3/dwc3-imx.c | 2 +- + drivers/usb/dwc3/dwc3-imx8mp.c | 4 +- + drivers/usb/dwc3/dwc3-keystone.c | 5 +- + drivers/usb/dwc3/dwc3-omap.c | 5 +- + drivers/usb/fotg210/Kconfig | 5 + + drivers/usb/fotg210/fotg210-core.c | 273 +- + drivers/usb/fotg210/fotg210-hcd.c | 5666 +------------------- + drivers/usb/fotg210/fotg210-hcd.h | 689 --- + drivers/usb/fotg210/fotg210-udc.c | 114 +- + drivers/usb/fotg210/fotg210-udc.h | 2 +- + drivers/usb/fotg210/fotg210.h | 26 +- + drivers/usb/gadget/composite.c | 2 +- + drivers/usb/gadget/function/f_mass_storage.c | 7 +- + drivers/usb/gadget/udc/aspeed-vhub/core.c | 4 +- + drivers/usb/gadget/udc/aspeed-vhub/dev.c | 2 - + drivers/usb/gadget/udc/aspeed_udc.c | 5 +- + drivers/usb/gadget/udc/at91_udc.c | 1 - + drivers/usb/gadget/udc/atmel_usba_udc.c | 5 +- + drivers/usb/gadget/udc/bcm63xx_udc.c | 1 - + drivers/usb/gadget/udc/fsl_udc_core.c | 3 - + drivers/usb/gadget/udc/lpc32xx_udc.c | 1 - + drivers/usb/gadget/udc/r8a66597-udc.c | 4 +- + drivers/usb/gadget/udc/renesas_usbf.c | 8 +- + drivers/usb/gadget/udc/snps_udc_plat.c | 4 +- + drivers/usb/gadget/udc/tegra-xudc.c | 5 +- + drivers/usb/host/Kconfig | 13 + + drivers/usb/host/ehci-dbg.c | 3 +- + drivers/usb/host/ehci-hcd.c | 31 +- + drivers/usb/host/ehci-hub.c | 104 +- + drivers/usb/host/ehci-ppc-of.c | 47 +- + drivers/usb/host/ehci-timer.c | 3 +- + drivers/usb/host/ehci.h | 55 +- + drivers/usb/host/ohci-dbg.c | 2 +- + drivers/usb/host/ohci-hub.c | 2 +- + drivers/usb/host/ohci-ppc-of.c | 44 +- + drivers/usb/host/pci-quirks.c | 1 + + drivers/usb/host/xhci-tegra.c | 8 +- + drivers/usb/image/microtek.c | 1 + + drivers/usb/misc/apple-mfi-fastcharge.c | 2 +- + drivers/usb/misc/brcmstb-usb-pinmap.c | 9 +- + drivers/usb/misc/onboard_usb_dev.c | 2 +- + drivers/usb/misc/qcom_eud.c | 2 +- + drivers/usb/mon/mon_main.c | 3 + + drivers/usb/mtu3/mtu3_core.c | 4 +- + drivers/usb/musb/da8xx.c | 8 +- + drivers/usb/musb/musb_dsps.c | 25 + + drivers/usb/phy/phy-ab8500-usb.c | 12 +- + drivers/usb/phy/phy-generic.c | 3 +- + drivers/usb/phy/phy-gpio-vbus-usb.c | 5 +- + drivers/usb/renesas_usbhs/mod.c | 4 +- + drivers/usb/storage/alauda.c | 95 +- + drivers/usb/storage/ene_ub6250.c | 2 + + drivers/usb/storage/sddr09.c | 9 + + drivers/usb/storage/usb.c | 6 +- + drivers/usb/typec/anx7411.c | 4 +- + drivers/usb/typec/bus.c | 4 +- + drivers/usb/typec/hd3ss3220.c | 36 +- + drivers/usb/typec/mux/gpio-sbu-mux.c | 13 +- + drivers/usb/typec/mux/it5205.c | 2 +- + drivers/usb/typec/tcpm/tcpci_maxim_core.c | 3 +- + drivers/usb/typec/tcpm/tcpci_mt6360.c | 1 - + drivers/usb/typec/tcpm/tcpci_mt6370.c | 2 +- + drivers/usb/typec/tcpm/tcpm.c | 57 +- + drivers/usb/typec/tipd/core.c | 3 + + drivers/usb/typec/ucsi/ucsi.c | 52 +- + drivers/usb/typec/ucsi/ucsi.h | 3 +- + drivers/usb/usbip/stub_main.c | 2 +- + include/linux/ulpi/driver.h | 4 +- + include/linux/usb.h | 6 +- + include/linux/usb/typec_altmode.h | 4 +- + 110 files changed, 1791 insertions(+), 7035 deletions(-) + create mode 100644 Documentation/devicetree/bindings/dma/ti/ti,cppi41.yaml + create mode 100644 Documentation/devicetree/bindings/phy/ti,am335x-usb-phy.yaml + delete mode 100644 Documentation/devicetree/bindings/usb/am33xx-usb.txt + delete mode 100644 Documentation/devicetree/bindings/usb/da8xx-usb.txt + delete mode 100644 Documentation/devicetree/bindings/usb/ohci-da8xx.txt + create mode 100644 Documentation/devicetree/bindings/usb/ti,am335x-usb-ctrl-module.yaml + create mode 100644 Documentation/devicetree/bindings/usb/ti,am33xx-usb.yaml + create mode 100644 Documentation/devicetree/bindings/usb/ti,da830-musb.yaml + create mode 100644 Documentation/devicetree/bindings/usb/ti,da830-ohci.yaml + create mode 100644 Documentation/devicetree/bindings/usb/ti,musb-am33xx.yaml + delete mode 100644 drivers/usb/fotg210/fotg210-hcd.h +Merging thunderbolt/next (a93a8e3200002 thunderbolt: stream: Make read return framing error to the userspace) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/westeri/thunderbolt.git thunderbolt/next +Auto-merging drivers/thunderbolt/nhi.c +Auto-merging drivers/thunderbolt/nhi.h +Auto-merging drivers/thunderbolt/pci.c +Auto-merging drivers/thunderbolt/quirks.c +Auto-merging drivers/thunderbolt/stream.c +Auto-merging drivers/thunderbolt/switch.c +Auto-merging drivers/thunderbolt/tb.c +Auto-merging include/linux/thunderbolt.h +Merge made by the 'ort' strategy. + drivers/net/thunderbolt/main.c | 6 +- + drivers/thunderbolt/debugfs.c | 126 ++++++++++++++---- + drivers/thunderbolt/dma_test.c | 10 +- + drivers/thunderbolt/eeprom.c | 23 +++- + drivers/thunderbolt/nhi.c | 222 ++++++++++++++++++++++--------- + drivers/thunderbolt/nhi.h | 2 + + drivers/thunderbolt/path.c | 2 +- + drivers/thunderbolt/pci.c | 113 ++++++++++++++++ + drivers/thunderbolt/quirks.c | 18 +++ + drivers/thunderbolt/sb_regs.h | 2 + + drivers/thunderbolt/stream.c | 287 +++++++++++++++++++++++++++++++---------- + drivers/thunderbolt/switch.c | 4 + + drivers/thunderbolt/tb.c | 78 ----------- + drivers/thunderbolt/tb.h | 5 +- + drivers/thunderbolt/usb4.c | 10 +- + include/linux/thunderbolt.h | 47 ++++++- + 16 files changed, 703 insertions(+), 252 deletions(-) +Merging usb-serial/usb-next (6583f9741341b USB: serial: wwan: replace __get_free_page() with kmalloc()) +$ git merge -m Merge branch 'usb-next' of https://git.kernel.org/pub/scm/linux/kernel/git/johan/usb-serial.git usb-serial/usb-next +Merge made by the 'ort' strategy. + drivers/usb/serial/usb_wwan.c | 6 +++--- + 1 file changed, 3 insertions(+), 3 deletions(-) +Merging tty/tty-next (bf961847813d9 serial: sifive: fix off-by-one in console port bounds check) +$ git merge -m Merge branch 'tty-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git tty/tty-next +Auto-merging MAINTAINERS +Auto-merging drivers/tty/serial/8250/8250_port.c +Auto-merging drivers/tty/serial/kgdboc.c +Auto-merging drivers/tty/serial/qcom_geni_serial.c +CONFLICT (content): Merge conflict in drivers/tty/serial/qcom_geni_serial.c +Auto-merging drivers/tty/serial/serial_core.c +Auto-merging drivers/tty/tty_port.c +Auto-merging include/linux/soc/qcom/geni-se.h +CONFLICT (content): Merge conflict in include/linux/soc/qcom/geni-se.h +Resolved 'drivers/tty/serial/qcom_geni_serial.c' using previous resolution. +Resolved 'include/linux/soc/qcom/geni-se.h' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 473edeccacc76] Merge branch 'tty-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/tty.git +$ git diff -M --stat --summary HEAD^.. + Documentation/admin-guide/devices.txt | 13 +- + Documentation/devicetree/bindings/serial/8250.yaml | 32 +- + .../devicetree/bindings/serial/cdns,uart.yaml | 1 + + .../devicetree/bindings/serial/renesas,hscif.yaml | 4 +- + .../devicetree/bindings/serial/renesas,scif.yaml | 4 +- + .../bindings/serial/snps-dw-apb-uart.yaml | 1 + + Documentation/driver-api/serial/driver.rst | 2 +- + MAINTAINERS | 2 +- + drivers/acpi/acpi_apd.c | 1 + + drivers/tty/Kconfig | 10 - + drivers/tty/Makefile | 1 - + drivers/tty/hvc/hvcs.c | 6 +- + drivers/tty/moxa.c | 2136 -------------------- + drivers/tty/pty.c | 5 +- + drivers/tty/serdev/serdev-ttyport.c | 2 + + drivers/tty/serial/8250/8250.h | 17 +- + drivers/tty/serial/8250/8250_airoha.c | 187 ++ + drivers/tty/serial/8250/8250_core.c | 25 +- + drivers/tty/serial/8250/8250_dw.c | 19 +- + drivers/tty/serial/8250/8250_dwlib.c | 5 + + drivers/tty/serial/8250/8250_dwlib.h | 5 + + drivers/tty/serial/8250/8250_hub6.c | 11 +- + drivers/tty/serial/8250/8250_mid.c | 15 +- + drivers/tty/serial/8250/8250_mxpcie.c | 30 +- + drivers/tty/serial/8250/8250_pci.c | 2 +- + drivers/tty/serial/8250/8250_platform.c | 9 +- + drivers/tty/serial/8250/8250_pnp.c | 4 +- + drivers/tty/serial/8250/8250_port.c | 31 +- + drivers/tty/serial/8250/8250_uniphier.c | 28 +- + drivers/tty/serial/8250/Kconfig | 15 +- + drivers/tty/serial/8250/Makefile | 3 +- + drivers/tty/serial/Kconfig | 1 + + drivers/tty/serial/arc_uart.c | 2 +- + drivers/tty/serial/atmel_serial.c | 8 +- + drivers/tty/serial/fsl_lpuart.c | 18 +- + drivers/tty/serial/kgdboc.c | 2 +- + drivers/tty/serial/qcom_geni_serial.c | 43 +- + drivers/tty/serial/serial_core.c | 51 +- + drivers/tty/serial/serial_txx9.c | 62 +- + drivers/tty/serial/sh-sci.c | 8 +- + drivers/tty/serial/sifive.c | 2 +- + drivers/tty/sysrq.c | 5 +- + drivers/tty/tty_port.c | 8 +- + drivers/tty/vt/keyboard.c | 4 +- + include/linux/serial_core.h | 2 - + include/linux/soc/qcom/geni-se.h | 12 +- + 46 files changed, 460 insertions(+), 2394 deletions(-) + delete mode 100644 drivers/tty/moxa.c + create mode 100644 drivers/tty/serial/8250/8250_airoha.c +Merging char-misc/char-misc-next (315860c4f9129 binder: Fix up Kconfig dependancy due to removal of .c code) +$ git merge -m Merge branch 'char-misc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git char-misc/char-misc-next +CONFLICT (modify/delete): drivers/android/binder.c deleted in char-misc/char-misc-next and modified in HEAD. Version HEAD of drivers/android/binder.c left in tree. +Auto-merging drivers/android/binder/node.rs +Auto-merging drivers/android/binder/node/wrapper.rs +Auto-merging drivers/android/binder/page_range.rs +Auto-merging drivers/android/binder/rust_binderfs.c +Auto-merging drivers/android/binder/thread.rs +CONFLICT (modify/delete): drivers/android/binder_alloc.c deleted in char-misc/char-misc-next and modified in HEAD. Version HEAD of drivers/android/binder_alloc.c left in tree. +CONFLICT (modify/delete): drivers/android/binderfs.c deleted in char-misc/char-misc-next and modified in HEAD. Version HEAD of drivers/android/binderfs.c left in tree. +Automatic merge failed; fix conflicts and then commit the result. +$ git rm -f drivers/android/binder.c +drivers/android/binderfs.c +drivers/android/binder_alloc.c +fatal: pathspec 'drivers/android/binder.c +drivers/android/binderfs.c +drivers/android/binder_alloc.c' did not match any files +$ git commit --no-edit -v -a +[master 70b76384f1cc2] Merge branch 'char-misc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/char-misc.git +$ git diff -M --stat --summary HEAD^.. + drivers/android/Kconfig | 41 +- + drivers/android/Makefile | 3 - + drivers/android/binder.c | 7187 -------------------- + drivers/android/binder/Makefile | 4 +- + drivers/android/binder/allocation.rs | 10 +- + drivers/android/binder/defs.rs | 3 +- + drivers/android/binder/freeze.rs | 11 +- + drivers/android/binder/node.rs | 10 +- + drivers/android/binder/node/wrapper.rs | 2 +- + drivers/android/binder/page_range.rs | 6 +- + drivers/android/binder/process.rs | 21 +- + drivers/android/binder/range_alloc/array.rs | 2 +- + drivers/android/binder/range_alloc/mod.rs | 2 +- + drivers/android/binder/range_alloc/tree.rs | 2 +- + drivers/android/binder/rust_binder_main.rs | 14 +- + drivers/android/binder/rust_binderfs.c | 2 +- + drivers/android/binder/thread.rs | 58 +- + drivers/android/binder/transaction.rs | 3 +- + drivers/android/binder_alloc.c | 1398 ---- + drivers/android/binder_alloc.h | 189 - + drivers/android/binder_internal.h | 597 -- + drivers/android/binder_netlink.c | 32 - + drivers/android/binder_netlink.h | 21 - + drivers/android/binder_trace.h | 448 -- + drivers/android/binderfs.c | 785 --- + drivers/android/dbitmap.h | 169 - + drivers/android/tests/.kunitconfig | 7 - + drivers/android/tests/Makefile | 6 - + drivers/android/tests/binder_alloc_kunit.c | 572 -- + include/uapi/linux/android/binder.h | 1 + + .../selftests/filesystems/binderfs/binderfs_test.c | 2 +- + .../testing/selftests/filesystems/binderfs/config | 3 +- + 32 files changed, 98 insertions(+), 11513 deletions(-) + delete mode 100644 drivers/android/binder.c + delete mode 100644 drivers/android/binder_alloc.c + delete mode 100644 drivers/android/binder_alloc.h + delete mode 100644 drivers/android/binder_internal.h + delete mode 100644 drivers/android/binder_netlink.c + delete mode 100644 drivers/android/binder_netlink.h + delete mode 100644 drivers/android/binder_trace.h + delete mode 100644 drivers/android/binderfs.c + delete mode 100644 drivers/android/dbitmap.h + delete mode 100644 drivers/android/tests/.kunitconfig + delete mode 100644 drivers/android/tests/Makefile + delete mode 100644 drivers/android/tests/binder_alloc_kunit.c +Merging coresight/next (9e3604d7369cf coresight: etm4x: remove redundant fields in etmv4_save_state) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/coresight/linux.git coresight/next +Already up to date. +Merging fastrpc/for-next (ef071c4906eb4 Merge branches 'fastrpc-fixes' and 'fastrpc-for-7.4' into fastrpc-for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/srini/fastrpc.git fastrpc/for-next +Auto-merging drivers/misc/fastrpc.c +Merge made by the 'ort' strategy. + drivers/misc/fastrpc.c | 200 ++++++++++++++++++++++++++----------------------- + 1 file changed, 107 insertions(+), 93 deletions(-) +Merging fpga/for-next (093da48782df3 fpga: altera-cvp: Retry teardown and reset CVP state on failure) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fpga/linux-fpga.git fpga/for-next +Merge made by the 'ort' strategy. + drivers/fpga/altera-cvp.c | 43 ++++++++++++++++++++++++++++++++++------- + drivers/fpga/dfl.h | 1 - + drivers/fpga/fpga-mgr.c | 2 +- + drivers/fpga/stratix10-soc.c | 2 +- + drivers/fpga/xilinx-selectmap.c | 39 ++++++++++++++++++++++++++++--------- + 5 files changed, 68 insertions(+), 19 deletions(-) +Merging icc/icc-next (3a6d690152557 Merge branch 'icc-fixes' into icc-next) +$ git merge -m Merge branch 'icc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/djakov/icc.git icc/icc-next +Merge made by the 'ort' strategy. + .../bindings/interconnect/qcom,kuno-rpmh.yaml | 118 +++ + .../bindings/interconnect/qcom,osm-l3.yaml | 1 + + drivers/interconnect/core.c | 32 + + drivers/interconnect/debugfs-client.c | 81 +- + drivers/interconnect/mediatek/Makefile | 2 +- + drivers/interconnect/qcom/Kconfig | 57 ++ + drivers/interconnect/qcom/Makefile | 2 + + drivers/interconnect/qcom/bcm-voter.c | 59 +- + drivers/interconnect/qcom/bcm-voter.h | 1 + + drivers/interconnect/qcom/eliza.c | 18 +- + drivers/interconnect/qcom/icc-common.h | 9 + + drivers/interconnect/qcom/icc-rpm.c | 105 ++- + drivers/interconnect/qcom/icc-rpm.h | 6 +- + drivers/interconnect/qcom/icc-rpmh.c | 71 +- + drivers/interconnect/qcom/kuno.c | 988 +++++++++++++++++++++ + drivers/interconnect/qcom/milos.c | 66 +- + drivers/interconnect/qcom/msm8976.c | 3 +- + drivers/interconnect/qcom/msm8996.c | 1 + + drivers/interconnect/qcom/sm6350.c | 56 +- + drivers/interconnect/qcom/sm8650.c | 70 +- + drivers/interconnect/qcom/smd-rpm.c | 8 +- + drivers/interconnect/samsung/exynos.c | 1 + + include/dt-bindings/interconnect/qcom,kuno.h | 89 ++ + 23 files changed, 1672 insertions(+), 172 deletions(-) + create mode 100644 Documentation/devicetree/bindings/interconnect/qcom,kuno-rpmh.yaml + create mode 100644 drivers/interconnect/qcom/kuno.c + create mode 100644 include/dt-bindings/interconnect/qcom,kuno.h +Merging iio/togreg (a3b3580713f3a iio: dac: ad5758: Fix the offset calculation) +$ git merge -m Merge branch 'togreg' of https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git iio/togreg +Auto-merging Documentation/devicetree/bindings/spi/spi-peripheral-props.yaml +Auto-merging Documentation/devicetree/bindings/trivial-devices.yaml +Auto-merging MAINTAINERS +Auto-merging drivers/iio/accel/kionix-kx022a.c +Auto-merging drivers/iio/accel/kxcjk-1013.c +Auto-merging drivers/iio/accel/sca3000.c +Auto-merging drivers/iio/adc/ad7173.c +Auto-merging drivers/iio/adc/ade9000.c +CONFLICT (content): Merge conflict in drivers/iio/adc/ade9000.c +Auto-merging drivers/iio/adc/adi-axi-adc.c +Auto-merging drivers/iio/adc/pac1934.c +Auto-merging drivers/iio/adc/stm32-adc.c +Auto-merging drivers/iio/buffer/industrialio-buffer-dmaengine.c +Auto-merging drivers/iio/frequency/adf4377.c +Auto-merging drivers/iio/imu/adis16400.c +Auto-merging drivers/iio/light/lm3533-als.c +Auto-merging drivers/iio/pressure/rohm-bm1390.c +Auto-merging drivers/iio/proximity/aw96103.c +Resolved 'drivers/iio/adc/ade9000.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 0025a46a07d9b] Merge branch 'togreg' of https://git.kernel.org/pub/scm/linux/kernel/git/jic23/iio.git +$ git diff -M --stat --summary HEAD^.. + Documentation/ABI/testing/debugfs-iio-ad9910 | 23 + + Documentation/ABI/testing/debugfs-iio-backend | 16 +- + Documentation/ABI/testing/sysfs-bus-iio | 115 + + Documentation/ABI/testing/sysfs-bus-iio-adc | 9 + + .../ABI/testing/sysfs-bus-iio-frequency-ad9910 | 31 + + .../bindings/iio/accel/adi,adis16201.yaml | 6 +- + .../devicetree/bindings/iio/accel/adi,adxl367.yaml | 13 +- + .../devicetree/bindings/iio/adc/adi,ad4080.yaml | 38 +- + .../devicetree/bindings/iio/adc/adi,ad4130.yaml | 3 +- + .../devicetree/bindings/iio/adc/adi,ad4851.yaml | 8 +- + .../devicetree/bindings/iio/adc/adi,ad7173.yaml | 2 +- + .../devicetree/bindings/iio/adc/adi,ad7768-1.yaml | 6 +- + .../devicetree/bindings/iio/adc/adi,ad7768.yaml | 270 +++ + .../devicetree/bindings/iio/adc/adi,ad7779.yaml | 30 +- + .../devicetree/bindings/iio/adc/adi,ade9000.yaml | 58 +- + .../devicetree/bindings/iio/adc/adi,max40080.yaml | 65 + + .../bindings/iio/adc/axiado,ax3000-saradc.yaml | 63 + + .../devicetree/bindings/iio/adc/lltc,ltc2497.yaml | 2 +- + .../bindings/iio/adc/maxim,max34408.yaml | 28 +- + .../bindings/iio/adc/mediatek,mt2701-auxadc.yaml | 1 + + .../bindings/iio/adc/nxp,lpc1850-adc.yaml | 2 +- + .../bindings/iio/adc/renesas,r9a09g077-adc.yaml | 22 +- + .../bindings/iio/adc/renesas,rzg2l-adc.yaml | 17 +- + .../devicetree/bindings/iio/adc/ti,ads1015.yaml | 6 + + .../devicetree/bindings/iio/adc/ti,ads1100.yaml | 10 +- + .../devicetree/bindings/iio/adc/ti,ads112c04.yaml | 149 ++ + .../devicetree/bindings/iio/adc/ti,ads1298.yaml | 18 +- + .../devicetree/bindings/iio/addac/adi,ad74115.yaml | 3 +- + .../bindings/iio/addac/adi,ad74413r.yaml | 3 +- + .../devicetree/bindings/iio/dac/adi,ad3530r.yaml | 15 +- + .../devicetree/bindings/iio/dac/adi,ad5529r.yaml | 254 +++ + .../devicetree/bindings/iio/dac/adi,ad5710r.yaml | 145 ++ + .../bindings/iio/dac/microchip,mcp47feb02.yaml | 219 +- + .../bindings/iio/frequency/adi,ad9910.yaml | 209 ++ + .../bindings/iio/imu/invensense,icm42600.yaml | 24 +- + .../devicetree/bindings/iio/light/adux1020.yaml | 8 +- + .../bindings/iio/light/amstaos,tsl2591.yaml | 2 +- + .../bindings/iio/light/capella,cm32181.yaml | 44 + + .../devicetree/bindings/iio/light/isl29018.yaml | 6 +- + .../bindings/iio/light/liteon,ltr501.yaml | 32 +- + .../devicetree/bindings/iio/light/noa1305.yaml | 4 +- + .../devicetree/bindings/iio/light/stk33xx.yaml | 19 +- + .../devicetree/bindings/iio/light/tsl2583.yaml | 4 +- + .../devicetree/bindings/iio/light/tsl2772.yaml | 14 +- + .../bindings/iio/light/vishay,veml6030.yaml | 21 +- + .../bindings/iio/pressure/infineon,dps310.yaml | 10 +- + .../iio/proximity/pulsedlight,lidar-lite-v2.yaml | 71 + + .../bindings/iio/proximity/vishay,vcnl3020.yaml | 6 +- + .../bindings/iio/temperature/adi,ltc2983.yaml | 1 + + .../bindings/spi/spi-peripheral-props.yaml | 7 + + .../devicetree/bindings/trivial-devices.yaml | 6 +- + Documentation/iio/ad7768.rst | 275 +++ + Documentation/iio/ad9910.rst | 792 +++++++ + Documentation/iio/ade9000.rst | 26 +- + Documentation/iio/adis16475.rst | 2 +- + Documentation/iio/adis16480.rst | 2 +- + Documentation/iio/adis16550.rst | 2 +- + Documentation/iio/adxl313.rst | 2 +- + Documentation/iio/adxl345.rst | 2 +- + Documentation/iio/adxl380.rst | 10 +- + Documentation/iio/index.rst | 2 + + MAINTAINERS | 74 +- + drivers/iio/accel/Kconfig | 7 +- + drivers/iio/accel/adis16201.c | 133 +- + drivers/iio/accel/adxl313_spi.c | 2 + + drivers/iio/accel/adxl367.c | 59 +- + drivers/iio/accel/adxl372_spi.c | 2 + + drivers/iio/accel/adxl380.c | 7 + + drivers/iio/accel/adxl380_spi.c | 2 + + drivers/iio/accel/bma220_spi.c | 3 +- + drivers/iio/accel/bma400_core.c | 2 - + drivers/iio/accel/kionix-kx022a.c | 22 +- + drivers/iio/accel/kionix-kx022a.h | 2 +- + drivers/iio/accel/kxcjk-1013.c | 1 + + drivers/iio/accel/mma8452.c | 145 +- + drivers/iio/accel/mma9551.c | 1 + + drivers/iio/accel/mma9553.c | 1 + + drivers/iio/accel/sca3000.c | 2 + + drivers/iio/adc/Kconfig | 86 +- + drivers/iio/adc/Makefile | 4 + + drivers/iio/adc/ad4080.c | 34 +- + drivers/iio/adc/ad4134.c | 1 - + drivers/iio/adc/ad7173.c | 3 +- + drivers/iio/adc/ad7192.c | 3 + + drivers/iio/adc/ad7606_spi.c | 6 +- + drivers/iio/adc/ad7768-1.c | 3 + + drivers/iio/adc/ad7768.c | 1921 ++++++++++++++++ + drivers/iio/adc/ade9000.c | 180 +- + drivers/iio/adc/adi-axi-adc.c | 21 + + drivers/iio/adc/at91-sama5d2_adc.c | 10 +- + drivers/iio/adc/axiado_saradc.c | 277 +++ + drivers/iio/adc/bcm_iproc_adc.c | 101 +- + drivers/iio/adc/ltc2497-core.c | 297 ++- + drivers/iio/adc/ltc2497.c | 42 + + drivers/iio/adc/ltc2497.h | 31 +- + drivers/iio/adc/max11205.c | 2 + + drivers/iio/adc/max14001.c | 1 + + drivers/iio/adc/max40080.c | 574 +++++ + drivers/iio/adc/mcp3911.c | 2 + + drivers/iio/adc/meson_saradc.c | 2 +- + drivers/iio/adc/nxp-sar-adc.c | 2 +- + drivers/iio/adc/pac1934.c | 9 +- + drivers/iio/adc/rockchip_saradc.c | 16 + + drivers/iio/adc/rzg2l_adc.c | 29 +- + drivers/iio/adc/rzt2h_adc.c | 499 ++++- + drivers/iio/adc/sophgo-cv1800b-adc.c | 2 + + drivers/iio/adc/stm32-adc.c | 14 +- + drivers/iio/adc/stm32-dfsdm-adc.c | 4 +- + drivers/iio/adc/ti-adc128s052.c | 8 +- + drivers/iio/adc/ti-ads1015.c | 30 + + drivers/iio/adc/ti-ads1018.c | 6 +- + drivers/iio/adc/ti-ads1100.c | 188 +- + drivers/iio/adc/ti-ads112c04.c | 524 +++++ + drivers/iio/adc/ti-ads112c14.c | 1274 ++++++++++- + drivers/iio/adc/ti-ads131m02.c | 2 + + drivers/iio/adc/ti-ads7950.c | 4 +- + drivers/iio/adc/ti_am335x_adc.c | 4 +- + drivers/iio/amplifiers/ad8366.c | 2 + + drivers/iio/buffer/industrialio-buffer-dmaengine.c | 8 +- + drivers/iio/chemical/ens160_core.c | 7 +- + drivers/iio/chemical/sps30.c | 11 +- + .../iio/common/cros_ec_sensors/cros_ec_sensors.c | 2 +- + .../common/cros_ec_sensors/cros_ec_sensors_core.c | 5 +- + drivers/iio/dac/Kconfig | 50 +- + drivers/iio/dac/Makefile | 5 +- + drivers/iio/dac/ad3530r.c | 321 ++- + drivers/iio/dac/ad3552r.c | 2 +- + drivers/iio/dac/ad5529r.c | 546 +++++ + drivers/iio/dac/ad5755.c | 3 + + drivers/iio/dac/ad5758.c | 8 +- + drivers/iio/dac/ad8460.c | 2 +- + .../iio/dac/{mcp47feb02.c => mcp47feb02-core.c} | 393 +--- + drivers/iio/dac/mcp47feb02-i2c.c | 145 ++ + drivers/iio/dac/mcp47feb02-spi.c | 145 ++ + drivers/iio/dac/mcp47feb02.h | 41 + + drivers/iio/dac/mcp4821.c | 7 +- + drivers/iio/dummy/iio_simple_dummy.c | 2 +- + drivers/iio/frequency/Kconfig | 21 + + drivers/iio/frequency/Makefile | 1 + + drivers/iio/frequency/ad9910.c | 2347 ++++++++++++++++++++ + drivers/iio/frequency/adf4350.c | 2 +- + drivers/iio/frequency/adf4377.c | 3 + + drivers/iio/gyro/adxrs290.c | 4 +- + drivers/iio/gyro/bmg160_core.c | 4 +- + drivers/iio/gyro/itg3200_buffer.c | 3 +- + drivers/iio/humidity/Kconfig | 8 +- + drivers/iio/humidity/am2315.c | 34 +- + drivers/iio/humidity/ens210.c | 4 +- + drivers/iio/humidity/hts221_core.c | 168 +- + drivers/iio/humidity/hts221_i2c.c | 11 +- + drivers/iio/humidity/hts221_spi.c | 11 +- + drivers/iio/iio_core.h | 7 + + drivers/iio/imu/adis16400.c | 8 +- + drivers/iio/imu/inv_icm42600/inv_icm42600.h | 5 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_accel.c | 33 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_buffer.c | 114 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_core.c | 26 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_gyro.c | 24 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_i2c.c | 16 +- + drivers/iio/imu/inv_icm42600/inv_icm42600_spi.c | 16 +- + drivers/iio/imu/inv_mpu6050/inv_mpu_core.c | 6 +- + drivers/iio/imu/st_lsm6dsx/st_lsm6dsx_shub.c | 4 +- + drivers/iio/industrialio-backend.c | 33 + + drivers/iio/industrialio-core.c | 255 ++- + drivers/iio/industrialio-gts-helper.c | 2 +- + drivers/iio/light/Kconfig | 14 + + drivers/iio/light/Makefile | 1 + + drivers/iio/light/apds9306.c | 2 +- + drivers/iio/light/apds9999.c | 12 +- + drivers/iio/light/cros_ec_light_prox.c | 2 +- + drivers/iio/light/iqs621-als.c | 24 +- + drivers/iio/light/isl29018.c | 7 +- + drivers/iio/light/isl29028.c | 33 +- + drivers/iio/light/lm3533-als.c | 11 +- + drivers/iio/light/ltr390.c | 6 - + drivers/iio/light/ltr501.c | 113 +- + drivers/iio/light/stk3310.c | 203 +- + drivers/iio/light/tsl2583.c | 15 +- + drivers/iio/light/tsl2772.c | 16 +- + drivers/iio/light/vcnl4000.c | 3 +- + drivers/iio/light/veml3328.c | 53 +- + drivers/iio/light/veml6031x00.c | 1387 ++++++++++++ + drivers/iio/magnetometer/ak8975.c | 234 +- + drivers/iio/position/iqs624-pos.c | 24 +- + drivers/iio/potentiometer/x9250.c | 7 +- + drivers/iio/pressure/Kconfig | 2 + + drivers/iio/pressure/Makefile | 2 + + drivers/iio/pressure/bmp280-spi.c | 2 + + drivers/iio/pressure/cros_ec_baro.c | 2 +- + drivers/iio/pressure/dps310.c | 313 ++- + drivers/iio/pressure/rohm-bm1390.c | 6 +- + drivers/iio/proximity/aw96103.c | 2 +- + drivers/iio/temperature/iqs620at-temp.c | 9 +- + drivers/iio/temperature/ltc2983.c | 19 +- + drivers/iio/temperature/mlx90632.c | 10 +- + drivers/iio/temperature/mlx90635.c | 11 +- + drivers/iio/temperature/tmp117.c | 18 +- + drivers/iio/temperature/tsys01.c | 17 +- + drivers/iio/temperature/tsys02d.c | 12 +- + drivers/iio/test/Kconfig | 13 + + drivers/iio/test/Makefile | 1 + + drivers/iio/test/iio-test-channel-prefix.c | 307 +++ + drivers/iio/test/iio-test-rescale.c | 4 +- + .../iio/Documentation/sysfs-bus-iio-adc-ad7280a | 2 +- + drivers/staging/iio/Kconfig | 1 - + drivers/staging/iio/Makefile | 1 - + drivers/staging/iio/accel/Kconfig | 19 - + drivers/staging/iio/accel/Makefile | 6 - + drivers/staging/iio/accel/adis16203.c | 315 --- + drivers/staging/iio/adc/ad7816.c | 33 +- + drivers/staging/iio/frequency/ad9832.c | 6 + + drivers/staging/iio/frequency/ad9834.c | 6 + + include/linux/iio/adc/qcom-adc5-gen3-common.h | 2 +- + include/linux/iio/backend.h | 6 + + include/linux/iio/iio-gts-helper.h | 11 +- + include/linux/iio/iio-opaque.h | 2 +- + include/linux/iio/iio.h | 25 +- + include/linux/notifier.h | 7 + + include/uapi/linux/iio/types.h | 1 + + kernel/notifier.c | 105 + + tools/iio/iio_event_monitor.c | 2 + + 221 files changed, 16048 insertions(+), 2228 deletions(-) + create mode 100644 Documentation/ABI/testing/debugfs-iio-ad9910 + create mode 100644 Documentation/ABI/testing/sysfs-bus-iio-adc + create mode 100644 Documentation/ABI/testing/sysfs-bus-iio-frequency-ad9910 + create mode 100644 Documentation/devicetree/bindings/iio/adc/adi,ad7768.yaml + create mode 100644 Documentation/devicetree/bindings/iio/adc/adi,max40080.yaml + create mode 100644 Documentation/devicetree/bindings/iio/adc/axiado,ax3000-saradc.yaml + create mode 100644 Documentation/devicetree/bindings/iio/adc/ti,ads112c04.yaml + create mode 100644 Documentation/devicetree/bindings/iio/dac/adi,ad5529r.yaml + create mode 100644 Documentation/devicetree/bindings/iio/dac/adi,ad5710r.yaml + create mode 100644 Documentation/devicetree/bindings/iio/frequency/adi,ad9910.yaml + create mode 100644 Documentation/devicetree/bindings/iio/light/capella,cm32181.yaml + create mode 100644 Documentation/devicetree/bindings/iio/proximity/pulsedlight,lidar-lite-v2.yaml + create mode 100644 Documentation/iio/ad7768.rst + create mode 100644 Documentation/iio/ad9910.rst + create mode 100644 drivers/iio/adc/ad7768.c + create mode 100644 drivers/iio/adc/axiado_saradc.c + create mode 100644 drivers/iio/adc/max40080.c + create mode 100644 drivers/iio/adc/ti-ads112c04.c + create mode 100644 drivers/iio/dac/ad5529r.c + rename drivers/iio/dac/{mcp47feb02.c => mcp47feb02-core.c} (69%) + create mode 100644 drivers/iio/dac/mcp47feb02-i2c.c + create mode 100644 drivers/iio/dac/mcp47feb02-spi.c + create mode 100644 drivers/iio/dac/mcp47feb02.h + create mode 100644 drivers/iio/frequency/ad9910.c + create mode 100644 drivers/iio/light/veml6031x00.c + create mode 100644 drivers/iio/test/iio-test-channel-prefix.c + delete mode 100644 drivers/staging/iio/accel/Kconfig + delete mode 100644 drivers/staging/iio/accel/Makefile + delete mode 100644 drivers/staging/iio/accel/adis16203.c +Merging nfc/for-next (fd73f4a665989 Linux 7.3-rc3) +$ git merge -m Merge branch 'for-next' of https://codeberg.org/linux-nfc/linux.git nfc/for-next +Already up to date. +Merging phy-next/next (c7f2322431cb6 dt-bindings: phy: ti,tcan104x-can: Fix property constrains) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/phy/linux-phy.git phy-next/next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../bindings/phy/mediatek,mt8195-dp-phy.yaml | 77 ++ + .../devicetree/bindings/phy/mediatek,tphy.yaml | 1 + + .../bindings/phy/qcom,m31-eusb2-phy.yaml | 1 + + .../bindings/phy/qcom,sc8280xp-qmp-pcie-phy.yaml | 2 + + .../phy/qcom,sc8280xp-qmp-usb43dp-phy.yaml | 59 +- + .../bindings/phy/qcom,x1e80100-csi2-phy.yaml | 209 ++++++ + .../devicetree/bindings/phy/ti,tcan104x-can.yaml | 9 +- + MAINTAINERS | 10 + + drivers/phy/apple/atc.c | 2 - + drivers/phy/broadcom/phy-brcm-usb.c | 29 +- + drivers/phy/cadence/phy-cadence-sierra.c | 11 +- + drivers/phy/hisilicon/phy-hi3670-pcie.c | 2 + + drivers/phy/mediatek/phy-mtk-dp.c | 826 ++++++++++++++++++--- + drivers/phy/phy-core.c | 166 ++++- + drivers/phy/qualcomm/Kconfig | 15 + + drivers/phy/qualcomm/Makefile | 5 + + drivers/phy/qualcomm/phy-qcom-mipi-csi2-3ph-dphy.c | 385 ++++++++++ + drivers/phy/qualcomm/phy-qcom-mipi-csi2-core.c | 463 ++++++++++++ + drivers/phy/qualcomm/phy-qcom-mipi-csi2.h | 97 +++ + drivers/phy/qualcomm/phy-qcom-qmp-combo.c | 402 ++++++++-- + drivers/phy/qualcomm/phy-qcom-qmp-pcie.c | 73 ++ + drivers/phy/qualcomm/phy-qcom-qmp-pcs-v8_50.h | 4 +- + drivers/phy/samsung/phy-exynos5-usbdrd.c | 78 +- + drivers/phy/socionext/phy-uniphier-usb3hs.c | 3 +- + drivers/phy/starfive/phy-jh7110-dphy-tx.c | 16 +- + include/dt-bindings/phy/phy-qcom-qmp.h | 1 + + include/linux/phy/phy-thunderbolt.h | 14 + + include/linux/phy/phy.h | 15 + + 28 files changed, 2715 insertions(+), 260 deletions(-) + create mode 100644 Documentation/devicetree/bindings/phy/mediatek,mt8195-dp-phy.yaml + create mode 100644 Documentation/devicetree/bindings/phy/qcom,x1e80100-csi2-phy.yaml + create mode 100644 drivers/phy/qualcomm/phy-qcom-mipi-csi2-3ph-dphy.c + create mode 100644 drivers/phy/qualcomm/phy-qcom-mipi-csi2-core.c + create mode 100644 drivers/phy/qualcomm/phy-qcom-mipi-csi2.h + create mode 100644 include/linux/phy/phy-thunderbolt.h +Merging soundwire/next (90b63b309fd6c soundwire: Intel: stop sdw clock in system suspend) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/soundwire.git soundwire/next +Auto-merging drivers/soundwire/dmi-quirks.c +Merge made by the 'ort' strategy. + drivers/soundwire/bus.c | 54 +++++++++++++++++++++++------------- + drivers/soundwire/bus_type.c | 16 +++++++---- + drivers/soundwire/dmi-quirks.c | 14 ++++++++++ + drivers/soundwire/intel.h | 6 ++-- + drivers/soundwire/intel_auxdevice.c | 11 +++++--- + drivers/soundwire/intel_bus_common.c | 8 +++--- + drivers/soundwire/qcom.c | 38 ++++++++++++++++++------- + drivers/soundwire/slave.c | 28 +++++++++++++------ + drivers/soundwire/stream.c | 8 +++--- + include/linux/soundwire/sdw.h | 4 +++ + include/linux/soundwire/sdw_intel.h | 2 +- + 11 files changed, 129 insertions(+), 60 deletions(-) +Merging extcon/extcon-next (8d3ae59288f1e Linux 7.2) +$ git merge -m Merge branch 'extcon-next' of https://git.kernel.org/pub/scm/linux/kernel/git/chanwoo/extcon.git extcon/extcon-next +Already up to date. +Merging gnss/gnss-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'gnss-next' of https://git.kernel.org/pub/scm/linux/kernel/git/johan/gnss.git gnss/gnss-next +Already up to date. +Merging vfio/next (b30b52c2fb82e vfio/pci: Restore 8-byte ioeventfd support) +$ git merge -m Merge branch 'next' of https://github.com/awilliam/linux-vfio.git vfio/next +Auto-merging Documentation/driver-api/index.rst +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + Documentation/driver-api/index.rst | 1 + + Documentation/driver-api/vfio-selftests.rst | 252 +++ + MAINTAINERS | 1 + + drivers/vfio/cdx/main.c | 17 +- + drivers/vfio/device_cdev.c | 12 + + drivers/vfio/fsl-mc/vfio_fsl_mc.c | 15 +- + drivers/vfio/mdev/mdev_core.c | 2 +- + drivers/vfio/pci/vfio_pci_rdwr.c | 3 - + drivers/vfio/platform/vfio_platform_common.c | 15 +- + drivers/vfio/vfio_iommu_type1.c | 12 +- + drivers/vfio/vfio_main.c | 7 - + include/linux/mlx5/cq.h | 10 - + include/linux/mlx5/device.h | 231 +-- + include/linux/mlx5/mlx5_ifc.h | 178 ++ + include/linux/mlx5/mlx5_ifc_macros.h | 185 ++ + tools/arch/arm64/include/asm/barrier.h | 4 + + tools/arch/x86/include/asm/barrier.h | 5 + + tools/include/asm-generic/io.h | 28 + + tools/include/asm/barrier.h | 8 + + tools/include/linux/stddef.h | 10 + + .../selftests/kvm/include/arm64/processor.h | 4 +- + tools/testing/selftests/kvm/irq_test.c | 2 - + tools/testing/selftests/vfio/.gitignore | 1 + + tools/testing/selftests/vfio/lib/drivers/dsa/dsa.c | 1 + + tools/testing/selftests/vfio/lib/drivers/igb/igb.c | 1 + + .../testing/selftests/vfio/lib/drivers/ioat/ioat.c | 1 + + .../testing/selftests/vfio/lib/drivers/mlx5/mlx5.c | 1928 ++++++++++++++++++++ + .../selftests/vfio/lib/drivers/mlx5/mlx5_hw.h | 114 ++ + .../selftests/vfio/lib/drivers/mlx5/mlx5_ifc.h | 1 + + .../vfio/lib/drivers/mlx5/mlx5_ifc_fpga.h | 1 + + .../vfio/lib/drivers/mlx5/mlx5_ifc_macros.h | 1 + + .../vfio/lib/drivers/nv_falcon/nv_falcon.c | 1 + + .../vfio/lib/include/libvfio/vfio_pci_device.h | 11 + + .../vfio/lib/include/libvfio/vfio_pci_driver.h | 6 + + tools/testing/selftests/vfio/lib/iova_allocator.c | 7 +- + tools/testing/selftests/vfio/lib/libvfio.mk | 1 + + tools/testing/selftests/vfio/lib/sysfs.c | 1 + + tools/testing/selftests/vfio/lib/vfio_pci_driver.c | 6 + + tools/testing/selftests/vfio/settings | 5 + + .../testing/selftests/vfio/vfio_pci_driver_test.c | 5 +- + .../selftests/vfio/vfio_pci_sriov_uapi_test.c | 35 + + 41 files changed, 2850 insertions(+), 279 deletions(-) + create mode 100644 Documentation/driver-api/vfio-selftests.rst + create mode 100644 include/linux/mlx5/mlx5_ifc_macros.h + create mode 100644 tools/include/linux/stddef.h + create mode 100644 tools/testing/selftests/vfio/lib/drivers/mlx5/mlx5.c + create mode 100644 tools/testing/selftests/vfio/lib/drivers/mlx5/mlx5_hw.h + create mode 120000 tools/testing/selftests/vfio/lib/drivers/mlx5/mlx5_ifc.h + create mode 120000 tools/testing/selftests/vfio/lib/drivers/mlx5/mlx5_ifc_fpga.h + create mode 120000 tools/testing/selftests/vfio/lib/drivers/mlx5/mlx5_ifc_macros.h + create mode 100644 tools/testing/selftests/vfio/settings +Merging w1/for-next (813a5b9a3c9f6 w1: fix spelling mistakes in comments across the subsystem) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krzk/linux-w1.git w1/for-next +Merge made by the 'ort' strategy. + drivers/w1/masters/ds2482.c | 2 +- + drivers/w1/masters/ds2490.c | 4 ++-- + drivers/w1/masters/omap_hdq.c | 2 +- + drivers/w1/slaves/w1_ds28e17.c | 2 +- + drivers/w1/slaves/w1_therm.c | 2 +- + drivers/w1/w1.c | 2 +- + drivers/w1/w1_family.c | 2 +- + drivers/w1/w1_netlink.h | 2 +- + 8 files changed, 9 insertions(+), 9 deletions(-) +Merging spmi/spmi-next (8cdeaa50eae8d Linux 7.2-rc2) +$ git merge -m Merge branch 'spmi-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sboyd/spmi.git spmi/spmi-next +Already up to date. +Merging staging/staging-next (8444548bd905f staging: rtl8723bs: remove unused SSIZE_PTR definition) +$ git merge -m Merge branch 'staging-next' of https://git.kernel.org/pub/scm/linux/kernel/git/gregkh/staging.git staging/staging-next +Auto-merging drivers/staging/rtl8723bs/os_dep/ioctl_cfg80211.c +Merge made by the 'ort' strategy. + drivers/staging/axis-fifo/axis-fifo.c | 6 +- + drivers/staging/fbtft/fb_bd663474.c | 4 +- + drivers/staging/greybus/loopback.c | 5 +- + drivers/staging/most/video/video.c | 1 - + drivers/staging/octeon/ethernet-tx.c | 2 +- + drivers/staging/rtl8723bs/core/rtw_ap.c | 2 +- + drivers/staging/rtl8723bs/core/rtw_cmd.c | 25 +- + drivers/staging/rtl8723bs/core/rtw_efuse.c | 147 +-- + drivers/staging/rtl8723bs/core/rtw_mlme.c | 1006 ++++++++++---------- + drivers/staging/rtl8723bs/core/rtw_mlme_ext.c | 111 +-- + drivers/staging/rtl8723bs/core/rtw_recv.c | 49 +- + drivers/staging/rtl8723bs/core/rtw_security.c | 4 +- + drivers/staging/rtl8723bs/core/rtw_wlan_util.c | 6 +- + drivers/staging/rtl8723bs/core/rtw_xmit.c | 48 +- + drivers/staging/rtl8723bs/hal/HalPhyRf.c | 5 +- + drivers/staging/rtl8723bs/hal/HalPhyRf_8723B.c | 75 -- + drivers/staging/rtl8723bs/hal/hal_com.c | 6 +- + drivers/staging/rtl8723bs/hal/hal_intf.c | 9 +- + drivers/staging/rtl8723bs/hal/odm.h | 9 - + drivers/staging/rtl8723bs/hal/odm_DIG.h | 1 - + drivers/staging/rtl8723bs/hal/odm_DynamicTxPower.h | 1 - + drivers/staging/rtl8723bs/hal/odm_precomp.h | 1 - + drivers/staging/rtl8723bs/hal/rtl8723b_hal_init.c | 2 - + drivers/staging/rtl8723bs/hal/rtl8723b_rf6052.c | 2 +- + drivers/staging/rtl8723bs/hal/rtl8723bs_recv.c | 3 +- + drivers/staging/rtl8723bs/hal/rtl8723bs_xmit.c | 3 - + drivers/staging/rtl8723bs/hal/sdio_halinit.c | 24 - + drivers/staging/rtl8723bs/include/basic_types.h | 4 - + drivers/staging/rtl8723bs/include/cmd_osdep.h | 6 +- + drivers/staging/rtl8723bs/include/hal_com.h | 2 +- + drivers/staging/rtl8723bs/include/hal_data.h | 16 - + drivers/staging/rtl8723bs/include/hal_intf.h | 2 +- + drivers/staging/rtl8723bs/include/hal_pg.h | 2 +- + drivers/staging/rtl8723bs/include/hal_phy.h | 16 - + drivers/staging/rtl8723bs/include/osdep_service.h | 4 +- + .../rtl8723bs/include/osdep_service_linux.h | 2 +- + drivers/staging/rtl8723bs/include/rtl8723b_hal.h | 2 +- + drivers/staging/rtl8723bs/include/rtw_cmd.h | 58 +- + drivers/staging/rtl8723bs/include/rtw_efuse.h | 5 +- + drivers/staging/rtl8723bs/include/rtw_io.h | 14 +- + drivers/staging/rtl8723bs/include/rtw_ioctl_set.h | 2 - + drivers/staging/rtl8723bs/include/rtw_mlme.h | 78 +- + drivers/staging/rtl8723bs/include/rtw_mlme_ext.h | 38 +- + drivers/staging/rtl8723bs/include/rtw_pwrctrl.h | 20 +- + drivers/staging/rtl8723bs/include/rtw_recv.h | 69 +- + drivers/staging/rtl8723bs/include/rtw_xmit.h | 8 +- + drivers/staging/rtl8723bs/include/sdio_ops.h | 21 +- + drivers/staging/rtl8723bs/include/sta_info.h | 84 +- + drivers/staging/rtl8723bs/include/wlan_bssdef.h | 8 +- + drivers/staging/rtl8723bs/include/xmit_osdep.h | 14 +- + drivers/staging/rtl8723bs/os_dep/ioctl_cfg80211.c | 86 +- + drivers/staging/rtl8723bs/os_dep/sdio_intf.c | 4 +- + drivers/staging/sm750fb/sm750.c | 12 +- + drivers/staging/sm750fb/sm750_accel.h | 4 +- + drivers/staging/vme_user/vme_tsi148.c | 3 +- + 55 files changed, 929 insertions(+), 1212 deletions(-) +Merging counter-next/counter-next (edac5cf356994 MAINTAINERS: Mark ftm-quaddec driver as orphaned) +$ git merge -m Merge branch 'counter-next' of https://git.kernel.org/pub/scm/linux/kernel/git/wbg/counter.git counter-next/counter-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 6 ++---- + 1 file changed, 2 insertions(+), 4 deletions(-) +Merging mux/for-next (ac7bde3c53166 mux: Add driver for Renesas RZ/V2H VBENCTL VBUS_SEL mux) +$ git merge -m Merge branch 'for-next' of https://gitlab.com/peda-linux/mux.git mux/for-next +Merge made by the 'ort' strategy. + drivers/mux/Kconfig | 13 ++++++++ + drivers/mux/Makefile | 2 ++ + drivers/mux/rzv2h-vbenctl.c | 81 +++++++++++++++++++++++++++++++++++++++++++++ + 3 files changed, 96 insertions(+) + create mode 100644 drivers/mux/rzv2h-vbenctl.c +Merging dmaengine/next (0a8dda0a15d39 dmaengine: bestcomm: use devm_platform_get_and_ioremap_resource() to simplify code) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git dmaengine/next +CONFLICT (modify/delete): Documentation/devicetree/bindings/usb/am33xx-usb.txt deleted in HEAD and modified in dmaengine/next. Version dmaengine/next of Documentation/devicetree/bindings/usb/am33xx-usb.txt left in tree. +CONFLICT (modify/delete): Documentation/devicetree/bindings/usb/da8xx-usb.txt deleted in HEAD and modified in dmaengine/next. Version dmaengine/next of Documentation/devicetree/bindings/usb/da8xx-usb.txt left in tree. +Auto-merging MAINTAINERS +Auto-merging drivers/dma/dmaengine.c +Auto-merging drivers/dma/mmp_pdma.c +Auto-merging drivers/dma/pxa_dma.c +Auto-merging drivers/dma/sprd-dma.c +Auto-merging drivers/dma/sun6i-dma.c +Auto-merging drivers/dma/switchtec_dma.c +Auto-merging drivers/dma/xilinx/xilinx_dma.c +Auto-merging include/linux/dmaengine.h +Automatic merge failed; fix conflicts and then commit the result. +$ git rm -f Documentation/devicetree/bindings/usb/am33xx-usb.txt +rm 'Documentation/devicetree/bindings/usb/am33xx-usb.txt' +$ git commit --no-edit -v -a +[master e818ae9edd692] Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/vkoul/dmaengine.git +$ git diff -M --stat --summary HEAD^.. + .../devicetree/bindings/dma/apple,admac.yaml | 10 +- + .../devicetree/bindings/dma/snps,dw-axi-dmac.yaml | 22 ++ + .../bindings/dma/ti,dra7-dma-crossbar.yaml | 94 ++++++ + .../devicetree/bindings/dma/ti-dma-crossbar.txt | 68 ---- + Documentation/driver-api/dmaengine/provider.rst | 2 +- + MAINTAINERS | 2 +- + drivers/dma/altera-msgdma.c | 10 +- + drivers/dma/amba-pl08x.c | 4 +- + drivers/dma/apple-admac.c | 24 ++ + drivers/dma/arm-dma350.c | 2 +- + drivers/dma/at_hdmac.c | 139 ++++---- + drivers/dma/at_xdmac.c | 159 +++++----- + drivers/dma/bestcomm/ata.c | 2 - + drivers/dma/bestcomm/bestcomm.c | 46 +-- + drivers/dma/bestcomm/gen_bd.c | 23 +- + drivers/dma/dma-jz4780.c | 10 +- + drivers/dma/dmaengine.c | 34 +- + drivers/dma/dw-axi-dmac/dw-axi-dmac-platform.c | 151 +++++---- + drivers/dma/dw-axi-dmac/dw-axi-dmac.h | 57 ++-- + drivers/dma/dw-edma/dw-edma-core.c | 349 ++++++++++++++------- + drivers/dma/dw-edma/dw-edma-core.h | 67 +++- + drivers/dma/dw-edma/dw-edma-pcie.c | 18 ++ + drivers/dma/dw-edma/dw-edma-v0-core.c | 59 ++-- + drivers/dma/dw-edma/dw-hdma-v0-core.c | 82 +++-- + drivers/dma/dw-edma/dw-hdma-v0-debugfs.c | 17 +- + drivers/dma/dw-edma/dw-hdma-v0-regs.h | 10 - + drivers/dma/dw/core.c | 47 ++- + drivers/dma/ep93xx_dma.c | 37 +-- + drivers/dma/fsl-edma-main.c | 5 +- + drivers/dma/fsl_raid.c | 38 ++- + drivers/dma/fsldma.c | 22 +- + drivers/dma/idma64.c | 10 +- + drivers/dma/img-mdc-dma.c | 2 +- + drivers/dma/loongson/loongson1-apb-dma.c | 23 +- + drivers/dma/loongson/loongson2-apb-cmc-dma.c | 19 +- + drivers/dma/loongson/loongson2-apb-dma.c | 9 +- + drivers/dma/mediatek/mtk-hsdma.c | 26 +- + drivers/dma/mmp_pdma.c | 3 +- + drivers/dma/moxart-dma.c | 23 +- + drivers/dma/mv_xor.c | 8 +- + drivers/dma/nbpfaxi.c | 2 +- + drivers/dma/owl-dma.c | 21 +- + drivers/dma/pch_dma.c | 41 ++- + drivers/dma/pl330.c | 4 +- + drivers/dma/pxa_dma.c | 52 +-- + drivers/dma/qcom/gpi.c | 9 +- + drivers/dma/sprd-dma.c | 45 +-- + drivers/dma/st_fdma.c | 2 +- + drivers/dma/ste_dma40.c | 15 +- + drivers/dma/stm32/stm32-dma.c | 69 ++-- + drivers/dma/stm32/stm32-dma3.c | 99 +++--- + drivers/dma/stm32/stm32-mdma.c | 96 +++--- + drivers/dma/sun4i-dma.c | 17 +- + drivers/dma/sun6i-dma.c | 62 ++-- + drivers/dma/switchtec_dma.c | 14 +- + drivers/dma/tegra186-gpc-dma.c | 4 +- + drivers/dma/tegra20-apb-dma.c | 2 +- + drivers/dma/ti/k3-udma.c | 6 +- + drivers/dma/timb_dma.c | 63 ++-- + drivers/dma/txx9dmac.c | 89 +++--- + drivers/dma/virt-dma.h | 16 + + drivers/dma/xilinx/xilinx_dma.c | 85 ++++- + drivers/dma/xilinx/zynqmp_dma.c | 74 +++-- + drivers/pci/controller/dwc/pcie-designware.c | 1 + + include/linux/dma/edma.h | 3 +- + include/linux/dma/imx-dma.h | 5 - + include/linux/dmaengine.h | 23 +- + include/linux/fsl/bestcomm/gen_bd.h | 8 - + include/trace/events/tegra_apb_dma.h | 6 +- + 69 files changed, 1481 insertions(+), 1185 deletions(-) + create mode 100644 Documentation/devicetree/bindings/dma/ti,dra7-dma-crossbar.yaml + delete mode 100644 Documentation/devicetree/bindings/dma/ti-dma-crossbar.txt +Merging cgroup/for-next (8deb0752fa761 Merge branch 'for-7.4' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/tj/cgroup.git cgroup/for-next +Auto-merging Documentation/admin-guide/cgroup-v2.rst +Auto-merging block/blk-cgroup.c +Auto-merging kernel/cgroup/cgroup-internal.h +Auto-merging kernel/cgroup/cgroup-v1.c +Auto-merging kernel/cgroup/dmem.c +Merge made by the 'ort' strategy. + Documentation/admin-guide/cgroup-v2.rst | 62 +++++++++--- + block/blk-cgroup.c | 11 ++- + kernel/cgroup/cgroup-internal.h | 2 +- + kernel/cgroup/cgroup-v1.c | 26 +++-- + kernel/cgroup/cgroup.c | 121 ++++++++++++++--------- + kernel/cgroup/cpuset.c | 77 +++------------ + kernel/cgroup/debug.c | 18 ++-- + kernel/cgroup/dmem.c | 4 +- + kernel/cgroup/freezer.c | 5 +- + kernel/cgroup/namespace.c | 2 - + tools/testing/selftests/cgroup/lib/cgroup_util.c | 45 +++++++-- + tools/testing/selftests/cgroup/settings | 1 + + 12 files changed, 210 insertions(+), 164 deletions(-) + create mode 100644 tools/testing/selftests/cgroup/settings +Merging scsi/for-next (6147f16c23efb Merge branch 'misc' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jejb/scsi.git scsi/for-next +Auto-merging drivers/ata/libata-scsi.c +CONFLICT (content): Merge conflict in drivers/ata/libata-scsi.c +Auto-merging drivers/scsi/scsi_scan.c +Auto-merging drivers/ufs/host/ufs-qcom.c +Auto-merging drivers/usb/gadget/function/f_mass_storage.c +Resolved 'drivers/ata/libata-scsi.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 11e052aca8eb2] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jejb/scsi.git +$ git diff -M --stat --summary HEAD^.. + Documentation/scsi/scsi_mid_low_api.rst | 8 +- + drivers/ata/libata-eh.c | 17 +- + drivers/ata/libata-sata.c | 20 +- + drivers/ata/libata-scsi.c | 344 ++-- + drivers/ata/libata.h | 5 +- + drivers/cdrom/cdrom.c | 10 +- + drivers/message/fusion/mptfc.c | 8 +- + drivers/s390/scsi/zfcp_erp.c | 30 +- + drivers/s390/scsi/zfcp_ext.h | 2 +- + drivers/s390/scsi/zfcp_fsf.c | 9 +- + drivers/s390/scsi/zfcp_scsi.c | 6 +- + drivers/s390/scsi/zfcp_sysfs.c | 4 +- + drivers/scsi/3w-9xxx.c | 20 +- + drivers/scsi/3w-sas.c | 28 +- + drivers/scsi/3w-xxxx.c | 27 +- + drivers/scsi/53c700.c | 16 +- + drivers/scsi/BusLogic.c | 20 +- + drivers/scsi/a100u2w.c | 4 +- + drivers/scsi/a2091.c | 4 +- + drivers/scsi/a3000.c | 4 +- + drivers/scsi/aacraid/commsup.c | 8 +- + drivers/scsi/advansys.c | 8 +- + drivers/scsi/aha1542.c | 20 +- + drivers/scsi/aha1740.c | 12 +- + drivers/scsi/arm/eesox.c | 8 +- + drivers/scsi/arm/fas216.c | 20 +- + drivers/scsi/atp870u.c | 12 +- + drivers/scsi/bfa/bfad_bsg.c | 4 +- + drivers/scsi/ch.c | 55 +- + drivers/scsi/constants.c | 61 +- + drivers/scsi/csiostor/csio_attr.c | 4 +- + drivers/scsi/csiostor/csio_scsi.c | 4 +- + drivers/scsi/dc395x.c | 8 +- + drivers/scsi/device_handler/scsi_dh_alua.c | 55 +- + drivers/scsi/device_handler/scsi_dh_emc.c | 28 +- + drivers/scsi/device_handler/scsi_dh_hp_sw.c | 29 +- + drivers/scsi/device_handler/scsi_dh_rdac.c | 55 +- + drivers/scsi/esp_scsi.c | 32 +- + drivers/scsi/fcoe/fcoe.c | 4 +- + drivers/scsi/fdomain.c | 16 +- + drivers/scsi/gvp11.c | 4 +- + drivers/scsi/hosts.c | 17 +- + drivers/scsi/hpsa.c | 69 +- + drivers/scsi/hpsa_cmd.h | 23 - + drivers/scsi/hptiop.c | 8 +- + drivers/scsi/ibmvscsi/ibmvfc-core.c | 262 +-- + drivers/scsi/ibmvscsi/ibmvfc-nvme.c | 8 +- + drivers/scsi/ibmvscsi/ibmvscsi.c | 86 +- + drivers/scsi/ibmvscsi_tgt/ibmvscsi_tgt.c | 5 +- + drivers/scsi/imm.c | 4 +- + drivers/scsi/initio.c | 8 +- + drivers/scsi/ipr.c | 324 ++-- + drivers/scsi/ips.c | 26 +- + drivers/scsi/leapraid/leapraid_func.h | 7 - + drivers/scsi/leapraid/leapraid_os.c | 15 +- + drivers/scsi/libfc/fc_fcp.c | 8 +- + drivers/scsi/libiscsi.c | 3 +- + drivers/scsi/libsas/sas_scsi_host.c | 10 +- + drivers/scsi/lpfc/lpfc_els.c | 30 +- + drivers/scsi/lpfc/lpfc_hbadisc.c | 44 +- + drivers/scsi/lpfc/lpfc_init.c | 16 +- + drivers/scsi/lpfc/lpfc_scsi.c | 26 +- + drivers/scsi/lpfc/lpfc_sli.c | 4 +- + drivers/scsi/mac53c94.c | 8 +- + drivers/scsi/megaraid.c | 8 +- + drivers/scsi/megaraid/mega_common.h | 2 - + drivers/scsi/megaraid/megaraid_mbox.c | 12 +- + drivers/scsi/megaraid/megaraid_sas_base.c | 18 +- + drivers/scsi/mesh.c | 24 +- + drivers/scsi/mpi3mr/mpi/mpi30_cnfg.h | 77 +- + drivers/scsi/mpi3mr/mpi/mpi30_image.h | 7 +- + drivers/scsi/mpi3mr/mpi/mpi30_ioc.h | 15 +- + drivers/scsi/mpi3mr/mpi/mpi30_transport.h | 2 +- + drivers/scsi/mpi3mr/mpi3mr.h | 13 +- + drivers/scsi/mpi3mr/mpi3mr_app.c | 87 +- + drivers/scsi/mpi3mr/mpi3mr_fw.c | 223 ++- + drivers/scsi/mpi3mr/mpi3mr_os.c | 320 ++-- + drivers/scsi/mpi3mr/mpi3mr_transport.c | 106 +- + drivers/scsi/mpt3sas/mpt3sas_scsih.c | 69 +- + drivers/scsi/mvumi.c | 28 +- + drivers/scsi/myrb.c | 67 +- + drivers/scsi/myrs.c | 23 +- + drivers/scsi/nsp32.c | 8 +- + drivers/scsi/pcmcia/sym53c500_cs.c | 8 +- + drivers/scsi/pm8001/pm8001_defs.h | 98 ++ + drivers/scsi/pm8001/pm8001_hwi.c | 2 +- + drivers/scsi/pm8001/pm8001_hwi.h | 4 +- + drivers/scsi/pm8001/pm80xx_hwi.c | 2 +- + drivers/scsi/pm8001/pm80xx_hwi.h | 100 +- + drivers/scsi/pmcraid.c | 84 +- + drivers/scsi/ps3rom.c | 5 +- + drivers/scsi/qla1280.c | 44 +- + drivers/scsi/qla2xxx/qla_attr.c | 4 +- + drivers/scsi/qla2xxx/qla_init.c | 4 +- + drivers/scsi/qla2xxx/qla_isr.c | 9 +- + drivers/scsi/qla2xxx/qla_target.c | 99 +- + drivers/scsi/qla2xxx/qla_target.h | 3 - + drivers/scsi/qlogicfas408.c | 8 +- + drivers/scsi/qlogicpti.c | 20 +- + drivers/scsi/scsi.c | 15 +- + drivers/scsi/scsi_common.c | 24 +- + drivers/scsi/scsi_debug.c | 514 ++++-- + drivers/scsi/scsi_debugfs.c | 2 +- + drivers/scsi/scsi_devinfo.c | 5 +- + drivers/scsi/scsi_error.c | 164 +- + drivers/scsi/scsi_ioctl.c | 3 +- + drivers/scsi/scsi_lib.c | 179 +- + drivers/scsi/scsi_lib_test.c | 95 +- + drivers/scsi/scsi_logging.c | 22 +- + drivers/scsi/scsi_proc.c | 9 +- + drivers/scsi/scsi_scan.c | 38 +- + drivers/scsi/scsi_sysfs.c | 28 +- + drivers/scsi/scsi_transport_fc.c | 136 +- + drivers/scsi/scsi_transport_spi.c | 12 +- + drivers/scsi/sd.c | 126 +- + drivers/scsi/sd_zbc.c | 2 +- + drivers/scsi/sense_codes.h | 2352 +++++++++++++++++--------- + drivers/scsi/ses.c | 23 +- + drivers/scsi/sgiwd93.c | 4 +- + drivers/scsi/smartpqi/smartpqi_init.c | 31 +- + drivers/scsi/snic/snic_disc.c | 16 +- + drivers/scsi/sr.c | 3 +- + drivers/scsi/sr_ioctl.c | 29 +- + drivers/scsi/st.c | 35 +- + drivers/scsi/stex.c | 51 +- + drivers/scsi/storvsc_drv.c | 15 +- + drivers/scsi/sym53c8xx_2/sym_glue.c | 56 +- + drivers/scsi/sym53c8xx_2/sym_hipd.h | 4 +- + drivers/scsi/wd33c93.c | 4 +- + drivers/scsi/wd719x.c | 24 +- + drivers/scsi/xen-scsifront.c | 40 +- + drivers/scsi/zorro7xx.c | 31 +- + drivers/target/target_core_file.c | 4 +- + drivers/target/target_core_pscsi.c | 16 +- + drivers/target/target_core_spc.c | 11 +- + drivers/target/target_core_transport.c | 11 +- + drivers/target/target_core_ua.c | 28 +- + drivers/target/target_core_ua.h | 7 +- + drivers/ufs/Kconfig | 1 + + drivers/ufs/core/ufs-debugfs.c | 4 +- + drivers/ufs/core/ufs-rpmb.c | 29 +- + drivers/ufs/core/ufs-sysfs.c | 12 +- + drivers/ufs/core/ufshcd.c | 189 ++- + drivers/ufs/host/ufs-mediatek.c | 4 +- + drivers/ufs/host/ufs-qcom.c | 47 +- + drivers/ufs/host/ufs-qcom.h | 5 +- + drivers/ufs/host/ufs-sprd.c | 4 +- + drivers/usb/gadget/function/f_mass_storage.c | 34 +- + drivers/usb/gadget/function/storage_common.h | 69 +- + drivers/usb/storage/debug.c | 11 +- + drivers/usb/storage/debug.h | 4 +- + drivers/usb/storage/transport.c | 7 +- + drivers/usb/storage/uas.c | 12 +- + drivers/usb/storage/usb.h | 4 +- + include/scsi/scsi_cmnd.h | 3 +- + include/scsi/scsi_common.h | 16 +- + include/scsi/scsi_dbg.h | 6 +- + include/scsi/scsi_device.h | 26 +- + include/scsi/scsi_host.h | 9 +- + include/scsi/scsi_proto.h | 71 +- + include/scsi/scsi_sense.h | 950 +++++++++++ + include/trace/events/scsi.h | 4 +- + 162 files changed, 5925 insertions(+), 3412 deletions(-) + create mode 100644 include/scsi/scsi_sense.h +Merging scsi-mkp/for-next (f09d2c7485b32 scsi: target: core: Use assign_bit() where applicable) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git scsi-mkp/for-next +Auto-merging drivers/ata/libata-scsi.c +Auto-merging drivers/scsi/mpi3mr/mpi3mr_transport.c +CONFLICT (content): Merge conflict in drivers/scsi/mpi3mr/mpi3mr_transport.c +Resolved 'drivers/scsi/mpi3mr/mpi3mr_transport.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 16ea30b5942d3] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git +$ git diff -M --stat --summary HEAD^.. + drivers/scsi/mpi3mr/mpi3mr_transport.c | 2 +- + drivers/target/target_core_user.c | 5 +- + include/scsi/scsi_sense.h | 236 ++++++++++++++++----------------- + 3 files changed, 120 insertions(+), 123 deletions(-) +$ git am -3 ../patches/device-id-zorro +Applying: foofof +Using index info to reconstruct a base tree... +M include/linux/device-id/zorro.h +Falling back to patching base and 3-way merge... +No changes -- Patch already applied. +$ git am -3 ../patches/0001-libata-Fix-up-semantic-conflict-in-ata_scsi_set_sens.patch +Applying: libata: Fix up semantic conflict in ata_scsi_set_sense() +$ git reset HEAD^ +Unstaged changes after reset: +M drivers/ata/libata-scsi.c +$ git add -A . +$ git commit -v -a --amend +warning: notes ref refs/notes/commits is invalid +[master efb88db687078] Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mkp/scsi.git + Date: Wed Sep 30 13:37:18 2026 +0100 +Merging vhost/linux-next (8f2c2fb94a013 vhost/vsock: add VHOST_RESET_OWNER ioctl) +$ git merge -m Merge branch 'linux-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mst/vhost.git vhost/linux-next +Auto-merging arch/um/drivers/virtio_uml.c +Auto-merging drivers/platform/mellanox/mlxbf-tmfifo.c +Merge made by the 'ort' strategy. + arch/um/drivers/virtio_uml.c | 10 ++++ + drivers/platform/mellanox/mlxbf-tmfifo.c | 15 ++++++ + drivers/remoteproc/remoteproc_core.c | 12 +++++ + drivers/remoteproc/remoteproc_virtio.c | 39 +++++++++++--- + drivers/s390/virtio/virtio_ccw.c | 6 +-- + drivers/vhost/net.c | 4 +- + drivers/vhost/scsi.c | 2 +- + drivers/vhost/test.c | 4 +- + drivers/vhost/vdpa.c | 4 +- + drivers/vhost/vhost.c | 26 ++++++++-- + drivers/vhost/vhost.h | 5 +- + drivers/vhost/vsock.c | 89 +++++++++++++++++++++++++------- + drivers/virtio/virtio.c | 2 + + drivers/virtio/virtio_pci_legacy.c | 2 - + drivers/virtio/virtio_pci_modern.c | 15 +++--- + drivers/virtio/virtio_vdpa.c | 34 ++++++++++-- + include/linux/remoteproc.h | 3 ++ + 17 files changed, 213 insertions(+), 59 deletions(-) +Merging rpmsg/for-next (7fbd9a5319c33 Merge branches 'rpmsg-next', 'rproc-fixes' and 'rproc-next' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/remoteproc/linux.git rpmsg/for-next +Auto-merging MAINTAINERS +Auto-merging arch/arm64/boot/dts/qcom/sdx75.dtsi +Auto-merging drivers/remoteproc/remoteproc_core.c +Auto-merging drivers/soc/qcom/Kconfig +Merge made by the 'ort' strategy. + .../devicetree/bindings/remoteproc/mtk,scp.yaml | 22 +- + .../bindings/remoteproc/qcom,pas-common.yaml | 9 + + .../bindings/remoteproc/qcom,sm8550-pas.yaml | 31 +- + .../bindings/remoteproc/ti,am3352-wkup-m3.yaml | 1 - + Documentation/staging/rpmsg.rst | 17 + + MAINTAINERS | 8 + + arch/arm64/boot/dts/qcom/sdx75.dtsi | 3 +- + drivers/remoteproc/Kconfig | 11 +- + drivers/remoteproc/imx_dsp_rproc.c | 7 +- + drivers/remoteproc/imx_rproc.c | 51 +- + drivers/remoteproc/omap_remoteproc.c | 12 +- + drivers/remoteproc/qcom_q6v5_adsp.c | 2 +- + drivers/remoteproc/qcom_q6v5_pas.c | 166 +++++- + drivers/remoteproc/remoteproc_core.c | 3 + + drivers/remoteproc/remoteproc_debugfs.c | 6 +- + drivers/remoteproc/remoteproc_elf_loader.c | 23 +- + drivers/remoteproc/remoteproc_sysfs.c | 7 +- + drivers/remoteproc/stm32_rproc.c | 2 +- + drivers/remoteproc/ti_k3_common.c | 10 +- + drivers/remoteproc/ti_k3_common.h | 4 + + drivers/remoteproc/ti_k3_dsp_remoteproc.c | 7 +- + drivers/remoteproc/ti_k3_m4_remoteproc.c | 4 +- + drivers/remoteproc/ti_k3_r5_remoteproc.c | 9 +- + drivers/remoteproc/xlnx_r5_remoteproc.c | 13 +- + drivers/rpmsg/virtio_rpmsg_bus.c | 156 ++++-- + drivers/soc/qcom/Kconfig | 12 + + drivers/soc/qcom/Makefile | 1 + + drivers/soc/qcom/qmi_tmd.c | 595 +++++++++++++++++++++ + include/dt-bindings/thermal/qcom,pas.h | 20 + + include/linux/omap-mailbox.h | 5 +- + include/linux/rpmsg/virtio_rpmsg.h | 41 ++ + include/linux/soc/qcom/qmi.h | 1 + + include/linux/soc/qcom/qmi_tmd.h | 36 ++ + include/uapi/linux/rpmsg.h | 15 +- + samples/rpmsg/rpmsg_client_sample.c | 20 +- + 35 files changed, 1173 insertions(+), 157 deletions(-) + create mode 100644 drivers/soc/qcom/qmi_tmd.c + create mode 100644 include/dt-bindings/thermal/qcom,pas.h + create mode 100644 include/linux/rpmsg/virtio_rpmsg.h + create mode 100644 include/linux/soc/qcom/qmi_tmd.h +Merging gpio-brgl/gpio/for-next (c4e74a7058b57 gpio: pxa: stop reading gpio_chip::base in direction callbacks) +$ git merge -m Merge branch 'gpio/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git gpio-brgl/gpio/for-next +Auto-merging MAINTAINERS +CONFLICT (content): Merge conflict in MAINTAINERS +Auto-merging drivers/gpio/Kconfig +Auto-merging drivers/gpio/Makefile +Auto-merging drivers/gpio/gpio-mvebu.c +Auto-merging drivers/gpio/gpiolib-acpi-quirks.c +CONFLICT (content): Merge conflict in drivers/gpio/gpiolib-acpi-quirks.c +Auto-merging drivers/gpio/gpiolib-cdev.c +Auto-merging drivers/gpio/gpiolib.c +Resolved 'MAINTAINERS' using previous resolution. +Resolved 'drivers/gpio/gpiolib-acpi-quirks.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master ba4d7f800fca0] Merge branch 'gpio/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git +$ git diff -M --stat --summary HEAD^.. + .../bindings/gpio/axiado,ax3005-sgpio.yaml | 112 ++++ + .../devicetree/bindings/gpio/fsl,qoriq-gpio.yaml | 7 + + .../devicetree/bindings/gpio/gpio-zynq.yaml | 6 + + .../bindings/gpio/nvidia,tegra186-gpio.yaml | 9 + + .../bindings/gpio/realtek,otto-gpio.yaml | 2 + + .../bindings/gpio/renesas,rcar-gpio.yaml | 5 + + Documentation/driver-api/gpio/board.rst | 2 +- + Documentation/driver-api/gpio/driver.rst | 7 + + MAINTAINERS | 9 + + drivers/gpio/Kconfig | 33 + + drivers/gpio/Makefile | 2 + + drivers/gpio/gpio-ad7768.c | 121 ++++ + drivers/gpio/gpio-adp5585.c | 20 +- + drivers/gpio/gpio-altera.c | 5 +- + drivers/gpio/gpio-axiado-sgpio.c | 736 +++++++++++++++++++++ + drivers/gpio/gpio-dln2.c | 6 +- + drivers/gpio/gpio-eic-sprd.c | 17 +- + drivers/gpio/gpio-it87.c | 2 +- + drivers/gpio/gpio-mlxbf2.c | 4 +- + drivers/gpio/gpio-mlxbf3.c | 4 +- + drivers/gpio/gpio-mmio.c | 65 +- + drivers/gpio/gpio-mvebu.c | 19 +- + drivers/gpio/gpio-omap.c | 11 +- + drivers/gpio/gpio-pcf857x.c | 16 +- + drivers/gpio/gpio-pxa.c | 4 +- + drivers/gpio/gpio-rcar.c | 86 ++- + drivers/gpio/gpio-realtek-otto.c | 21 +- + drivers/gpio/gpio-regmap.c | 98 ++- + drivers/gpio/gpio-rtd1625.c | 26 +- + drivers/gpio/gpio-siox.c | 4 +- + drivers/gpio/gpio-sloppy-logic-analyzer.c | 4 +- + drivers/gpio/gpio-tegra186.c | 74 +++ + drivers/gpio/gpiolib-acpi-quirks.c | 13 + + drivers/gpio/gpiolib-cdev.c | 2 + + drivers/gpio/gpiolib-kunit.c | 14 +- + drivers/gpio/gpiolib.c | 21 + + drivers/pinctrl/freescale/pinctrl-imx.c | 54 +- + drivers/pinctrl/freescale/pinctrl-imx.h | 4 + + drivers/pinctrl/freescale/pinctrl-vf610.c | 2 + + include/linux/gpio/driver.h | 9 + + include/linux/gpio/regmap.h | 2 + + include/linux/pinctrl/consumer.h | 4 +- + tools/gpio/gpio-hammer.c | 2 +- + 43 files changed, 1532 insertions(+), 132 deletions(-) + create mode 100644 Documentation/devicetree/bindings/gpio/axiado,ax3005-sgpio.yaml + create mode 100644 drivers/gpio/gpio-ad7768.c + create mode 100644 drivers/gpio/gpio-axiado-sgpio.c +Merging gpio-intel/for-next (0fc424b6a8ef4 gpio: wcove: use regmap_assign_bits() for conditional set/clear) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-gpio-intel.git gpio-intel/for-next +Merge made by the 'ort' strategy. + drivers/gpio/gpio-wcove.c | 5 +---- + 1 file changed, 1 insertion(+), 4 deletions(-) +Merging pinctrl/for-next (c34fbaaac34ba Merge branch 'devel' into for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/linusw/linux-pinctrl.git pinctrl/for-next +Merge made by the 'ort' strategy. + .../pinctrl/allwinner,sun55i-a523-pinctrl.yaml | 6 +- + .../bindings/pinctrl/awinic,aw9523-pinctrl.yaml | 50 +- + .../bindings/pinctrl/cix,sky1-pinctrl.yaml | 2 +- + .../bindings/pinctrl/econet,en7528-pinctrl.yaml | 190 +++ + .../bindings/pinctrl/fsl,imx35-pinctrl.yaml | 4 +- + .../bindings/pinctrl/fsl,imx7d-pinctrl.yaml | 12 +- + .../bindings/pinctrl/fsl,imx8m-pinctrl.yaml | 4 +- + .../bindings/pinctrl/fsl,imx8ulp-pinctrl.yaml | 4 +- + .../bindings/pinctrl/fsl,imx9-pinctrl.yaml | 4 +- + .../devicetree/bindings/pinctrl/fsl,imxrt1050.yaml | 4 +- + .../devicetree/bindings/pinctrl/fsl,imxrt1170.yaml | 4 +- + .../devicetree/bindings/pinctrl/intel,lgm-io.yaml | 14 +- + .../bindings/pinctrl/mediatek,mt8183-pinctrl.yaml | 4 +- + .../bindings/pinctrl/mediatek,mt8195-pinctrl.yaml | 2 +- + .../devicetree/bindings/pinctrl/pinctrl-palmas.txt | 105 -- + .../bindings/pinctrl/qcom,hawi-tlmm.yaml | 4 +- + .../bindings/pinctrl/qcom,kaanapali-tlmm.yaml | 4 +- + .../bindings/pinctrl/qcom,maili-tlmm.yaml | 4 +- + .../bindings/pinctrl/renesas,rzg2l-pinctrl.yaml | 4 +- + .../bindings/pinctrl/semtech,sx1501q.yaml | 4 +- + .../bindings/pinctrl/ti,palmas-pinctrl.yaml | 170 +++ + .../bindings/pinctrl/xlnx,zynqmp-pinctrl.yaml | 46 +- + drivers/pinctrl/airoha/Kconfig | 15 +- + drivers/pinctrl/airoha/Makefile | 1 + + drivers/pinctrl/airoha/airoha-common.h | 3 + + drivers/pinctrl/airoha/pinctrl-airoha.c | 33 +- + drivers/pinctrl/airoha/pinctrl-an7563.c | 1 + + drivers/pinctrl/airoha/pinctrl-an7581.c | 1 + + drivers/pinctrl/airoha/pinctrl-an7583.c | 1 + + drivers/pinctrl/airoha/pinctrl-en7523.c | 1 + + drivers/pinctrl/airoha/pinctrl-en7528.c | 1204 ++++++++++++++++++++ + drivers/pinctrl/core.c | 4 +- + drivers/pinctrl/microchip/pinctrl-mpfs-mssio.c | 2 +- + drivers/pinctrl/pinconf.c | 2 +- + drivers/pinctrl/pinctrl-amd.c | 8 +- + drivers/pinctrl/pinctrl-equilibrium.c | 2 +- + drivers/pinctrl/pinctrl-st.c | 2 +- + drivers/pinctrl/pinmux.c | 2 +- + drivers/pinctrl/renesas/pinctrl-rzg2l.c | 412 ++++--- + drivers/pinctrl/renesas/pinctrl-rzt2h.c | 182 +-- + drivers/pinctrl/stm32/Kconfig | 2 +- + drivers/pinctrl/sunplus/Kconfig | 8 +- + drivers/pinctrl/sunxi/Kconfig | 5 + + drivers/pinctrl/sunxi/Makefile | 1 + + drivers/pinctrl/sunxi/pinctrl-sun20i-d1.c | 2 +- + drivers/pinctrl/sunxi/pinctrl-sun55i-a523-r.c | 3 +- + drivers/pinctrl/sunxi/pinctrl-sun55i-a523.c | 2 +- + drivers/pinctrl/sunxi/pinctrl-sun60i-a733.c | 50 + + drivers/pinctrl/sunxi/pinctrl-sunxi.c | 41 +- + drivers/pinctrl/sunxi/pinctrl-sunxi.h | 71 +- + include/linux/pinctrl/consumer.h | 1 + + 51 files changed, 2258 insertions(+), 449 deletions(-) + create mode 100644 Documentation/devicetree/bindings/pinctrl/econet,en7528-pinctrl.yaml + delete mode 100644 Documentation/devicetree/bindings/pinctrl/pinctrl-palmas.txt + create mode 100644 Documentation/devicetree/bindings/pinctrl/ti,palmas-pinctrl.yaml + create mode 100644 drivers/pinctrl/airoha/pinctrl-en7528.c + create mode 100644 drivers/pinctrl/sunxi/pinctrl-sun60i-a733.c +Merging pinctrl-intel/for-next (c016587866e57 pinctrl: intel: Try to retrieve driver data for pure platform drivers) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/intel.git pinctrl-intel/for-next +Merge made by the 'ort' strategy. + drivers/pinctrl/intel/pinctrl-intel.c | 51 ++++++++++++++++++++++++----------- + drivers/pinctrl/intel/pinctrl-intel.h | 2 +- + 2 files changed, 36 insertions(+), 17 deletions(-) +Merging pinctrl-renesas/renesas-pinctrl (0cd4a7b3a4883 pinctrl: renesas: rzt2h: Add a helper for reading PFC) +$ git merge -m Merge branch 'renesas-pinctrl' of https://git.kernel.org/pub/scm/linux/kernel/git/geert/renesas-drivers.git pinctrl-renesas/renesas-pinctrl +Already up to date. +Merging pinctrl-samsung/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/pinctrl/samsung.git pinctrl-samsung/for-next +Already up to date. +Merging pinctrl-qcom/pinctrl-qcom/for-next (4d7c9430a26ae pinctrl: qcom: tlmm-test: Add const to reg_names allocation type) +$ git merge -m Merge branch 'pinctrl-qcom/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git pinctrl-qcom/pinctrl-qcom/for-next +Auto-merging drivers/pinctrl/qcom/pinctrl-spmi-gpio.c +Merge made by the 'ort' strategy. + .../pinctrl/qcom,hawi-lpass-lpi-pinctrl.yaml | 109 ++ + .../bindings/pinctrl/qcom,kuno-tlmm.yaml | 110 ++ + .../bindings/pinctrl/qcom,msm8952-pinctrl.yaml | 146 +++ + .../bindings/pinctrl/qcom,pmic-gpio.yaml | 3 + + .../bindings/pinctrl/qcom,sc8280xp-tlmm.yaml | 3 + + drivers/pinctrl/qcom/Kconfig | 10 + + drivers/pinctrl/qcom/Kconfig.msm | 19 + + drivers/pinctrl/qcom/Makefile | 3 + + drivers/pinctrl/qcom/pinctrl-apq8064.c | 188 +-- + drivers/pinctrl/qcom/pinctrl-apq8084.c | 502 ++++---- + drivers/pinctrl/qcom/pinctrl-hawi-lpass-lpi.c | 244 ++++ + drivers/pinctrl/qcom/pinctrl-ipq4019.c | 208 ++-- + drivers/pinctrl/qcom/pinctrl-ipq8064.c | 208 ++-- + drivers/pinctrl/qcom/pinctrl-kuno.c | 801 +++++++++++++ + drivers/pinctrl/qcom/pinctrl-lpass-lpi.c | 6 +- + drivers/pinctrl/qcom/pinctrl-lpass-lpi.h | 17 + + drivers/pinctrl/qcom/pinctrl-msm.h | 25 - + drivers/pinctrl/qcom/pinctrl-msm8952.c | 1257 ++++++++++++++++++++ + drivers/pinctrl/qcom/pinctrl-spmi-gpio.c | 1 + + drivers/pinctrl/qcom/tlmm-test.c | 2 +- + 20 files changed, 3282 insertions(+), 580 deletions(-) + create mode 100644 Documentation/devicetree/bindings/pinctrl/qcom,hawi-lpass-lpi-pinctrl.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/qcom,kuno-tlmm.yaml + create mode 100644 Documentation/devicetree/bindings/pinctrl/qcom,msm8952-pinctrl.yaml + create mode 100644 drivers/pinctrl/qcom/pinctrl-hawi-lpass-lpi.c + create mode 100644 drivers/pinctrl/qcom/pinctrl-kuno.c + create mode 100644 drivers/pinctrl/qcom/pinctrl-msm8952.c +Merging pwm/pwm/for-next (e74b9a7ee5007 pwm: tiehrpwm: Clear period on disable to allow reconfiguration) +$ git merge -m Merge branch 'pwm/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ukleinek/linux.git pwm/pwm/for-next +Merge made by the 'ort' strategy. + .../devicetree/bindings/pwm/pwm-tipwmss.txt | 58 --------------- + .../devicetree/bindings/pwm/ti,am33xx-pwmss.yaml | 82 ++++++++++++++++++++++ + Documentation/driver-api/pwm.rst | 2 +- + drivers/pwm/core.c | 1 + + drivers/pwm/pwm-brcmstb.c | 2 +- + drivers/pwm/pwm-ipq.c | 2 +- + drivers/pwm/pwm-iqs620a.c | 23 +----- + drivers/pwm/pwm-loongson.c | 2 + + drivers/pwm/pwm-lp3943.c | 3 + + drivers/pwm/pwm-mtk-disp.c | 4 +- + drivers/pwm/pwm-pca9685.c | 2 +- + drivers/pwm/pwm-renesas-tpu.c | 23 ++++-- + drivers/pwm/pwm-tiehrpwm.c | 3 + + 13 files changed, 118 insertions(+), 89 deletions(-) + delete mode 100644 Documentation/devicetree/bindings/pwm/pwm-tipwmss.txt + create mode 100644 Documentation/devicetree/bindings/pwm/ti,am33xx-pwmss.yaml +Merging ktest/for-next (932cdaf3e273a ktest: Add logfile to failure directory) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/rostedt/linux-ktest.git ktest/for-next +Already up to date. +Merging kselftest/next (30af56a227e27 selftests/ftrace: skip gcov symbols when picking a function to probe) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kselftest/next +Merge made by the 'ort' strategy. + .../ftrace/test.d/kprobe/kprobe_eventname.tc | 2 +- + tools/testing/selftests/resctrl/cat_test.c | 23 ++- + tools/testing/selftests/resctrl/mba_test.c | 1 + + tools/testing/selftests/resctrl/mbm_test.c | 1 + + tools/testing/selftests/resctrl/resctrl.h | 2 + + tools/testing/selftests/resctrl/resctrl_val.c | 169 +++++++++++---------- + 6 files changed, 115 insertions(+), 83 deletions(-) +Merging kunit/test (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'test' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kunit/test +Already up to date. +Merging kunit-next/kunit (e38f53f048246 kunit: Return void from kunit_run_all_tests()) +$ git merge -m Merge branch 'kunit' of https://git.kernel.org/pub/scm/linux/kernel/git/shuah/linux-kselftest.git kunit-next/kunit +Merge made by the 'ort' strategy. + include/kunit/test.h | 5 ++--- + lib/kunit/executor.c | 5 ++--- + 2 files changed, 4 insertions(+), 6 deletions(-) +Merging livepatching/for-next (5d791d3396ca4 selftests/livepatch: filter debug messages in check_result()) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/livepatching/livepatching.git livepatching/for-next +Merge made by the 'ort' strategy. + include/linux/livepatch.h | 4 ++++ + kernel/livepatch/core.c | 10 ++++++++++ + tools/testing/selftests/livepatch/functions.sh | 12 +++++++++--- + 3 files changed, 23 insertions(+), 3 deletions(-) +Merging rtc/rtc-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'rtc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/abelloni/linux.git rtc/rtc-next +Already up to date. +Merging nvdimm/libnvdimm-for-next (e99cb3ecd8334 nvdimm-btt: clean up kernel-doc warnings) +$ git merge -m Merge branch 'libnvdimm-for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/nvdimm/nvdimm.git nvdimm/libnvdimm-for-next +Already up to date. +Merging at24/at24/for-next (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'at24/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git at24/at24/for-next +Already up to date. +Merging ntb/ntb-next (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'ntb-next' of https://github.com/jonmason/ntb.git ntb/ntb-next +Already up to date. +Merging seccomp/for-next/seccomp (832b9b176be06 seccomp: restore knotif->state when SECCOMP_ADDFD_FLAG_SEND is interrupted) +$ git merge -m Merge branch 'for-next/seccomp' of https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git seccomp/for-next/seccomp +Merge made by the 'ort' strategy. + kernel/seccomp.c | 7 +- + tools/testing/selftests/seccomp/seccomp_bpf.c | 179 ++++++++++++++++++++++++++ + 2 files changed, 184 insertions(+), 2 deletions(-) +Merging slimbus/for-next (4350d70546699 Merge branch 'slimbus-for-v7.4' into slimbus-for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/srini/slimbus.git slimbus/for-next +Merge made by the 'ort' strategy. + drivers/slimbus/messaging.c | 2 +- + drivers/slimbus/qcom-ngd-ctrl.c | 82 ++++++++++++++++++++++++++++++++++++++++- + drivers/slimbus/slimbus.h | 14 ++++++- + include/linux/slimbus.h | 2 +- + 4 files changed, 95 insertions(+), 5 deletions(-) +Merging nvmem/for-next (41f42ff4ae134 Merge branches 'nvmem-fixes' and 'nvmem-for-v7.4' into nvmem-for-next) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/srini/nvmem.git nvmem/for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + .../devicetree/bindings/mfd/mediatek,mt6397.yaml | 21 ++++ + .../devicetree/bindings/nvmem/qcom,qfprom.yaml | 1 + + .../bindings/nvmem/socionext,uniphier-efuse.yaml | 6 +- + MAINTAINERS | 5 + + drivers/nvmem/Kconfig | 14 ++- + drivers/nvmem/Makefile | 2 + + drivers/nvmem/core.c | 128 ++++++++++++--------- + drivers/nvmem/internals.h | 5 +- + drivers/nvmem/layouts.c | 13 ++- + drivers/nvmem/layouts/onie-tlv.c | 1 + + drivers/nvmem/layouts/sl28vpd.c | 1 + + drivers/nvmem/mt6323-efuse.c | 83 +++++++++++++ + drivers/nvmem/rockchip-otp.c | 15 ++- + 13 files changed, 236 insertions(+), 59 deletions(-) + create mode 100644 drivers/nvmem/mt6323-efuse.c +Merging hyperv/hyperv-next (be0cfab740e58 clocksource: hyper-v: Remove support for stimer interrupts in message mode) +$ git merge -m Merge branch 'hyperv-next' of https://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux.git hyperv/hyperv-next +Already up to date. +Merging auxdisplay/for-next (f63dc0eea92d0 docs: ABI: auxdisplay: use literal blocks for commands) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/andy/linux-auxdisplay.git auxdisplay/for-next +Merge made by the 'ort' strategy. + Documentation/ABI/testing/sysfs-auxdisplay-linedisp | 9 ++++++--- + drivers/auxdisplay/arm-charlcd.c | 7 ++----- + drivers/auxdisplay/panel.c | 5 +---- + 3 files changed, 9 insertions(+), 12 deletions(-) +Merging kgdb/kgdb/for-next (fdbdd0ccb30af kdb: remove redundant check for scancode 0xe0) +$ git merge -m Merge branch 'kgdb/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/danielt/linux.git kgdb/kgdb/for-next +Already up to date. +Merging hmm/hmm (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'hmm' of https://git.kernel.org/pub/scm/linux/kernel/git/rdma/rdma.git hmm/hmm +Already up to date. +Merging cfi/cfi/next (dc59e4fea9d83 Linux 7.2-rc1) +$ git merge -m Merge branch 'cfi/next' of https://git.kernel.org/pub/scm/linux/kernel/git/mtd/linux.git cfi/cfi/next +Already up to date. +Merging mhi/mhi-next (710bf7329abf8 bus: mhi: host: pci_generic: Add DUN channel configuration) +$ git merge -m Merge branch 'mhi-next' of https://git.kernel.org/pub/scm/linux/kernel/git/mani/mhi.git mhi/mhi-next +Merge made by the 'ort' strategy. + drivers/bus/mhi/ep/main.c | 4 ++-- + drivers/bus/mhi/host/debugfs.c | 4 ++-- + drivers/bus/mhi/host/pci_generic.c | 9 ++++++++- + include/linux/mhi.h | 2 +- + 4 files changed, 13 insertions(+), 6 deletions(-) +$ git am -3 ../patches/0001-fix-up-for-net-qrtr-Drop-the-MHI-auto_queue-feature-.patch +Applying: fix up for "net: qrtr: Drop the MHI auto_queue feature for IPCR DL channels" +Using index info to reconstruct a base tree... +M drivers/net/wireless/ath/ath12k/wifi7/mhi.c +Falling back to patching base and 3-way merge... +No changes -- Patch already applied. +Merging cxl/next (71392a644e88c Merge branch 'for-7.4/cxl-misc' into cxl-for-next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/cxl/cxl.git cxl/next +Auto-merging MAINTAINERS +Auto-merging drivers/cxl/core/ras.c +Auto-merging include/linux/device.h +Merge made by the 'ort' strategy. + .../driver-api/cxl/platform/acpi/dsdt.rst | 3 +- + MAINTAINERS | 1 + + drivers/cxl/acpi.c | 10 +- + drivers/cxl/core/cdat.c | 8 +- + drivers/cxl/core/core.h | 8 +- + drivers/cxl/core/edac.c | 30 ++++-- + drivers/cxl/core/features.c | 58 ++++++---- + drivers/cxl/core/hdm.c | 120 ++++++++++++++++----- + drivers/cxl/core/mbox.c | 3 + + drivers/cxl/core/mce.c | 8 +- + drivers/cxl/core/pci.c | 9 ++ + drivers/cxl/core/port.c | 39 +++++-- + drivers/cxl/core/ras.c | 6 +- + drivers/cxl/core/region.c | 100 +++++++++++------ + drivers/cxl/core/region_pmem.c | 6 +- + drivers/cxl/core/regs.c | 14 +++ + drivers/cxl/cxl.h | 12 +++ + drivers/cxl/mem.c | 14 ++- + drivers/cxl/pci.c | 5 +- + drivers/cxl/port.c | 3 + + include/linux/device.h | 1 + + tools/testing/cxl/Kbuild | 16 +-- + tools/testing/cxl/test/cxl.c | 117 ++++++++++++++++---- + 23 files changed, 443 insertions(+), 148 deletions(-) +Merging zstd/zstd-next (65d1f5507ed2c zstd: Import upstream v1.5.7) +$ git merge -m Merge branch 'zstd-next' of https://github.com/terrelln/linux.git zstd/zstd-next +Already up to date. +Merging efi/next (7eef16311a234 Merge remote-tracking branch 'linux-efi/bootloader-info' into next) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/efi/efi.git efi/next +Auto-merging drivers/firmware/efi/libstub/Makefile +Merge made by the 'ort' strategy. + drivers/firmware/efi/capsule-loader.c | 6 +- + drivers/firmware/efi/capsule.c | 4 +- + drivers/firmware/efi/efi.c | 33 +++++- + drivers/firmware/efi/libstub/Makefile | 3 +- + drivers/firmware/efi/libstub/bli.c | 87 ++++++++++++++ + drivers/firmware/efi/libstub/efi-stub-entry.c | 2 +- + drivers/firmware/efi/libstub/efi-stub-helper.c | 108 +++++++---------- + drivers/firmware/efi/libstub/efi-stub.c | 3 +- + drivers/firmware/efi/libstub/efistub.h | 12 +- + drivers/firmware/efi/libstub/file.c | 8 +- + drivers/firmware/efi/libstub/gop.c | 23 ++-- + drivers/firmware/efi/libstub/kaslr.c | 19 ++- + drivers/firmware/efi/libstub/mem.c | 2 +- + drivers/firmware/efi/libstub/pci.c | 2 +- + drivers/firmware/efi/libstub/printk.c | 95 ++------------- + drivers/firmware/efi/libstub/random.c | 8 +- + drivers/firmware/efi/libstub/riscv.c | 2 +- + drivers/firmware/efi/libstub/smbios.c | 3 +- + drivers/firmware/efi/libstub/tpm.c | 8 +- + drivers/firmware/efi/libstub/unaccepted_memory.c | 2 +- + drivers/firmware/efi/libstub/vsprintf.c | 140 ++++++++--------------- + drivers/firmware/efi/libstub/x86-stub.c | 18 +-- + drivers/firmware/efi/libstub/zboot.c | 3 +- + drivers/firmware/efi/runtime-wrappers.c | 28 +++++ + include/linux/efi.h | 23 ++++ + include/linux/ucs2_string.h | 11 +- + lib/ucs2_string.c | 19 +-- + 27 files changed, 360 insertions(+), 312 deletions(-) + create mode 100644 drivers/firmware/efi/libstub/bli.c +Merging unicode/for-next (a511442085c14 unicode: Properly reject invalid encoding version strings) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/krisman/unicode.git unicode/for-next +Already up to date. +Merging random/master (703749b069d53 random: fix vgetrandom_opaque_params kernel-doc) +$ git merge -m Merge branch 'master' of https://git.kernel.org/pub/scm/linux/kernel/git/crng/random.git random/master +Merge made by the 'ort' strategy. + drivers/virt/vmgenid.c | 11 +++++++---- + include/linux/siphash.h | 8 ++++++-- + include/uapi/linux/random.h | 2 +- + include/vdso/getrandom.h | 2 +- + 4 files changed, 15 insertions(+), 8 deletions(-) +Merging landlock/next (02619311dbfc4 Merge branch 'lsm/dev' into master) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/mic/linux.git landlock/next +Merge made by the 'ort' strategy. +Merging sysctl/sysctl-next (4991c8b72b564 time/jiffies: Saturate in mult_hz() instead of wrapping) +$ git merge -m Merge branch 'sysctl-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sysctl/sysctl.git sysctl/sysctl-next +Auto-merging fs/dcache.c +Auto-merging fs/proc/proc_sysctl.c +Auto-merging kernel/time/jiffies.c +Merge made by the 'ort' strategy. + drivers/parport/procfs.c | 6 +- + fs/dcache.c | 2 +- + fs/file_table.c | 2 +- + fs/proc/proc_sysctl.c | 2 +- + kernel/sysctl.c | 429 +++++---- + kernel/time/jiffies.c | 2 + + lib/test_sysctl.c | 33 +- + tools/testing/selftests/sysctl/sysctl.sh | 1546 ++++++++++++------------------ + 8 files changed, 903 insertions(+), 1119 deletions(-) +Merging execve/for-next/execve (df2908090cda3 Linux 7.3-rc2) +$ git merge -m Merge branch 'for-next/execve' of https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git execve/for-next/execve +Already up to date. +Merging bitmap/bitmap-for-next (452a6d5b5e055 bitmap: test bitmap_parselist() with a group size close to UINT_MAX) +$ git merge -m Merge branch 'bitmap-for-next' of https://github.com/norov/linux.git bitmap/bitmap-for-next +Auto-merging MAINTAINERS +Merge made by the 'ort' strategy. + MAINTAINERS | 2 + + include/linux/bitfield-fix-width.h | 241 +++++++++++++++++++++++++++++++ + include/linux/bitfield.h | 49 +------ + lib/bitmap-str.c | 24 ++- + lib/test_bitmap.c | 9 ++ + tools/include/linux/bitfield-fix-width.h | 236 ++++++++++++++++++++++++++++++ + tools/include/linux/bitfield.h | 49 +------ + 7 files changed, 509 insertions(+), 101 deletions(-) + create mode 100644 include/linux/bitfield-fix-width.h + create mode 100644 tools/include/linux/bitfield-fix-width.h +Merging hte/for-next (30167fadbf87f hte: tegra194-test: shut down timer before GPIO release) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/pateldipen1984/linux.git hte/for-next +Merge made by the 'ort' strategy. + drivers/hte/hte-tegra194-test.c | 2 +- + drivers/hte/hte-tegra194.c | 14 ++++++-------- + 2 files changed, 7 insertions(+), 9 deletions(-) +Merging kspp/for-next/kspp (760f96b7f54be um: fix CONFIG_GCOV for built-in code) +$ git merge -m Merge branch 'for-next/kspp' of https://git.kernel.org/pub/scm/linux/kernel/git/kees/linux.git kspp/for-next/kspp +Auto-merging arch/um/include/asm/common.lds.S +Merge made by the 'ort' strategy. + arch/um/include/asm/common.lds.S | 1 + + drivers/misc/lkdtm/core.c | 16 +-- + fs/signalfd.c | 28 ++++- + include/linux/fortify-string.h | 2 - + scripts/coccinelle/api/kmalloc_objs.cocci | 161 +++++++++++++++++++++----- + scripts/gcc-plugins/randomize_layout_plugin.c | 63 +++++++++- + 6 files changed, 219 insertions(+), 52 deletions(-) +Merging nolibc/for-next (da27722051fdf tools/nolibc: add support for pread() and pwrite()) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/nolibc/linux-nolibc.git nolibc/for-next +Auto-merging tools/testing/selftests/nolibc/Makefile.nolibc +Merge made by the 'ort' strategy. + tools/include/nolibc/Makefile | 19 ++- + tools/include/nolibc/arch-arm.h | 7 +- + tools/include/nolibc/arch-hexagon.h | 159 +++++++++++++++++++++++++ + tools/include/nolibc/arch-mips.h | 8 +- + tools/include/nolibc/arch-parisc.h | 28 ++++- + tools/include/nolibc/arch-powerpc.h | 7 +- + tools/include/nolibc/arch-sh.h | 18 +++ + tools/include/nolibc/arch-x86.h | 2 + + tools/include/nolibc/arch.h | 7 ++ + tools/include/nolibc/compiler.h | 9 ++ + tools/include/nolibc/dirent.h | 25 +++- + tools/include/nolibc/err.h | 2 +- + tools/include/nolibc/nolibc.h | 1 + + tools/include/nolibc/stdlib.h | 2 +- + tools/include/nolibc/sys.h | 60 ++++++++-- + tools/include/nolibc/sys/select.h | 6 +- + tools/include/nolibc/sys/sendfile.h | 41 +++++++ + tools/include/nolibc/types.h | 2 +- + tools/include/nolibc/unistd.h | 4 + + tools/testing/selftests/nolibc/Makefile.nolibc | 15 ++- + tools/testing/selftests/nolibc/nolibc-test.c | 159 +++++++++++++++++++++++-- + tools/testing/selftests/nolibc/run-tests.sh | 30 ++++- + 22 files changed, 554 insertions(+), 57 deletions(-) + create mode 100644 tools/include/nolibc/arch-hexagon.h + create mode 100644 tools/include/nolibc/sys/sendfile.h +Merging iommufd/for-next (54dadb030c7e2 iommufd: Fix iommufd_hw_capabilities reference in kernel-doc) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/jgg/iommufd.git iommufd/for-next +Auto-merging drivers/iommu/intel/nested.c +Merge made by the 'ort' strategy. + .../iommu/arm/arm-smmu-v3/arm-smmu-v3-iommufd.c | 199 ++++++++++++++++----- + drivers/iommu/intel/nested.c | 50 +++--- + drivers/iommu/iommufd/hw_pagetable.c | 25 +-- + drivers/iommu/iommufd/selftest.c | 163 ++++++++--------- + include/linux/iommu.h | 6 +- + include/linux/iommufd.h | 2 + + include/uapi/linux/iommufd.h | 6 +- + tools/testing/selftests/iommu/iommufd.c | 14 +- + tools/testing/selftests/iommu/iommufd_fail_nth.c | 14 +- + 9 files changed, 315 insertions(+), 164 deletions(-) +Merging turbostat/next (ccdfcb7e7ab84 turbostat 2026.09.28) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/lenb/linux.git turbostat/next +Merge made by the 'ort' strategy. + tools/power/x86/turbostat/turbostat.c | 36 ++++++++++++++++++++++------------- + 1 file changed, 23 insertions(+), 13 deletions(-) +Merging pwrseq/pwrseq/for-next (09baa2f4caab0 Merge tag 'pwrseq-is-controllable-for-v7.4' of git://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux into pwrseq/for-next) +$ git merge -m Merge branch 'pwrseq/for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/brgl/linux.git pwrseq/pwrseq/for-next +Auto-merging drivers/power/sequencing/pwrseq-pcie-m2.c +Merge made by the 'ort' strategy. + drivers/power/sequencing/Kconfig | 19 + + drivers/power/sequencing/Makefile | 2 + + drivers/power/sequencing/core.c | 49 + + drivers/power/sequencing/pwrseq-kunit.c | 1497 ++++++++++++++++++++++ + drivers/power/sequencing/pwrseq-pcie-m2.c | 45 +- + drivers/power/sequencing/pwrseq-qcom-wcn.c | 30 + + drivers/power/sequencing/pwrseq-renesas-pwrrdy.c | 142 ++ + include/linux/pwrseq/consumer.h | 7 + + include/linux/pwrseq/provider.h | 8 + + 9 files changed, 1797 insertions(+), 2 deletions(-) + create mode 100644 drivers/power/sequencing/pwrseq-kunit.c + create mode 100644 drivers/power/sequencing/pwrseq-renesas-pwrrdy.c +Merging capabilities-next/caps-next (507adb6448378 security: commoncap: clarify CAP_FS_SET comment in cap_task_fix_setuid()) +$ git merge -m Merge branch 'caps-next' of https://git.kernel.org/pub/scm/linux/kernel/git/sergeh/linux.git capabilities-next/caps-next +Auto-merging security/commoncap.c +Merge made by the 'ort' strategy. + security/commoncap.c | 10 ++++++---- + 1 file changed, 6 insertions(+), 4 deletions(-) +$ git am -3 ../patches/0001-sign-file-Fix-up-merge-issue.patch +Applying: sign-file: Fix up merge issue +Using index info to reconstruct a base tree... +M scripts/sign-file.c +Falling back to patching base and 3-way merge... +Auto-merging scripts/sign-file.c +No changes -- Patch already applied. +Merging ipe/next (bcaa4d1d69368 ipe: fix invalid sgid value in audit event documentation) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/wufan/ipe.git ipe/next +Merge made by the 'ort' strategy. + Documentation/admin-guide/LSM/ipe.rst | 4 ++-- + 1 file changed, 2 insertions(+), 2 deletions(-) +Merging kcsan/next (a8488ecbd7ba4 kcsan: avoid unintended access checking in NMIs) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/melver/linux.git kcsan/next +Already up to date. +Merging crc/crc-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'crc-next' of https://git.kernel.org/pub/scm/linux/kernel/git/ebiggers/linux.git crc/crc-next +Already up to date. +Merging keys-next/keys-next (965e9a2cf23b0 pkcs7: Change a pr_warn() to pr_warn_once()) +$ git merge -m Merge branch 'keys-next' of https://git.kernel.org/pub/scm/linux/kernel/git/dhowells/linux-fs.git keys-next/keys-next +Already up to date. +Merging fwctl/for-next (cee9395acd804 Linux 7.3-rc1) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/fwctl/fwctl.git fwctl/for-next +Already up to date. +Merging devsec-tsm/next (3177779ae17db virt: coco: change tsm_class to a const struct) +$ git merge -m Merge branch 'next' of https://git.kernel.org/pub/scm/linux/kernel/git/devsec/tsm.git devsec-tsm/next +Already up to date. +Merging hisilicon/for-next (a5db65458a911 Merge branch 'next/dt64' into for-next) +$ git merge -m Merge branch 'for-next' of https://github.com/hisilicon/linux-hisi.git hisilicon/for-next +Merge made by the 'ort' strategy. +Merging device-id/device-id-rework (995832b2cebe6 Replace by more specific (c files)) +$ git merge -m Merge branch 'device-id-rework' of https://git.kernel.org/pub/scm/linux/kernel/git/ukleinek/linux.git device-id/device-id-rework +Already up to date. +Merging kthread/for-next (fa39ec4f89f26 doc: Add housekeeping documentation) +$ git merge -m Merge branch 'for-next' of https://git.kernel.org/pub/scm/linux/kernel/git/frederic/linux-dynticks.git kthread/for-next +Already up to date. +Merging pagemap-headers/headers (e02cb91d4644d ksm: Remove pagemap.h include) +$ git merge -m Merge branch 'headers' of git://git.infradead.org/users/willy/pagecache.git pagemap-headers/headers +Auto-merging arch/alpha/include/asm/pgtable.h +Auto-merging arch/arc/include/asm/pgtable-levels.h +Auto-merging arch/microblaze/include/asm/pgtable.h +Auto-merging arch/sparc/include/asm/pgtable_64.h +Auto-merging arch/sparc/include/asm/tlb_64.h +Auto-merging arch/x86/virt/vmx/tdx/tdx.c +Auto-merging block/fops.c +Auto-merging drivers/dma/ste_dma40.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gpuvm.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_cs.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_gem.c +Auto-merging drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.c +Auto-merging drivers/gpu/drm/amd/amdkfd/kfd_migrate.c +CONFLICT (content): Merge conflict in drivers/gpu/drm/amd/amdkfd/kfd_migrate.c +Auto-merging drivers/gpu/drm/i915/display/intel_display_power.c +Auto-merging drivers/gpu/drm/msm/msm_fb.c +Auto-merging drivers/gpu/drm/msm/msm_gem_shrinker.c +Auto-merging drivers/gpu/drm/nouveau/nouveau_dmem.c +Auto-merging drivers/gpu/drm/omapdrm/dss/hdmi4.c +Auto-merging drivers/gpu/drm/omapdrm/dss/hdmi5.c +Auto-merging drivers/gpu/drm/panel/panel-ilitek-ili9806e-core.c +Auto-merging drivers/gpu/drm/panthor/panthor_gem.c +Auto-merging drivers/hwmon/pmbus/pmbus_core.c +Auto-merging drivers/i2c/busses/i2c-amd-asf-plat.c +Auto-merging drivers/i2c/busses/i2c-gxp.c +Auto-merging drivers/i2c/busses/i2c-k1.c +Auto-merging drivers/iio/accel/fxls8962af-core.c +Auto-merging drivers/iio/adc/ti-ads1298.c +Auto-merging drivers/iio/light/rohm-bu27034.c +Auto-merging drivers/iio/pressure/bmp280-core.c +Auto-merging drivers/leds/leds-turris-omnia.c +Auto-merging drivers/media/platform/st/stm32/stm32-csi.c +Auto-merging drivers/media/platform/ti/omap3isp/ispvideo.c +Auto-merging drivers/media/usb/go7007/go7007-v4l2.c +Auto-merging drivers/media/usb/gspca/gspca.c +Auto-merging drivers/mfd/88pm886.c +Auto-merging drivers/mfd/cs42l43.c +Auto-merging drivers/mfd/twl-core.c +Auto-merging drivers/mmc/core/core.c +Auto-merging drivers/mmc/core/host.c +Auto-merging drivers/mmc/host/sh_mmcif.c +Auto-merging drivers/net/ethernet/atheros/atl1e/atl1e.h +Auto-merging drivers/pinctrl/pinctrl-sx150x.c +Auto-merging drivers/platform/arm64/huawei-gaokun-ec.c +Auto-merging drivers/platform/arm64/lenovo-yoga-c630.c +Auto-merging drivers/power/supply/bq25630_charger.c +Auto-merging drivers/regulator/fixed.c +Auto-merging drivers/regulator/tps65185.c +Auto-merging drivers/ufs/host/ufs-mediatek.c +Auto-merging drivers/ufs/host/ufs-qcom.c +Auto-merging drivers/usb/typec/mux/it5205.c +Auto-merging fs/aio.c +Auto-merging fs/buffer.c +Auto-merging fs/file_table.c +Auto-merging fs/inode.c +Auto-merging fs/nfs/blocklayout/blocklayout.c +Auto-merging fs/nfs/nfstrace.h +Auto-merging fs/nfs/pnfs.c +Auto-merging fs/nfsd/nfs4proc.c +Auto-merging include/drm/drm_print.h +Auto-merging include/linux/ceph/libceph.h +Auto-merging include/linux/i3c/master.h +Auto-merging include/linux/nfs_fs.h +Auto-merging include/linux/nfs_page.h +Auto-merging include/linux/swap.h +Auto-merging kernel/power/snapshot.c +Auto-merging net/ceph/osd_client.c +CONFLICT (content): Merge conflict in net/ceph/osd_client.c +Auto-merging net/ceph/pagevec.c +Auto-merging net/sunrpc/auth_gss/gss_krb5_crypto.c +Auto-merging net/sunrpc/xdr.c +Auto-merging net/sunrpc/xprtsock.c +Auto-merging security/commoncap.c +Auto-merging security/selinux/hooks.c +Auto-merging security/selinux/selinuxfs.c +Auto-merging security/smack/smack_lsm.c +Resolved 'drivers/gpu/drm/amd/amdkfd/kfd_migrate.c' using previous resolution. +Resolved 'net/ceph/osd_client.c' using previous resolution. +Automatic merge failed; fix conflicts and then commit the result. +$ git commit --no-edit -v -a +[master 53d1b34b8f9fe] Merge branch 'headers' of git://git.infradead.org/users/willy/pagecache.git +$ git diff -M --stat --summary HEAD^.. + arch/alpha/include/asm/pgtable.h | 2 +- + arch/arc/include/asm/Kbuild | 1 + + arch/arc/include/asm/pgtable-levels.h | 2 ++ + arch/arc/include/asm/tlb.h | 12 ---------- + arch/arm/include/asm/pgalloc.h | 2 -- + arch/arm/include/asm/tlb.h | 2 -- + arch/arm/mach-pxa/pxa3xx.c | 1 + + arch/arm64/include/asm/tlb.h | 3 --- + arch/hexagon/include/asm/tlb.h | 1 - + arch/microblaze/include/asm/pgtable.h | 2 +- + arch/nios2/include/asm/tlb.h | 1 - + arch/openrisc/include/asm/Kbuild | 1 + + arch/openrisc/include/asm/tlb.h | 26 ---------------------- + arch/powerpc/include/asm/tlb.h | 2 -- + arch/sh/include/asm/pgtable.h | 2 +- + arch/sh/include/asm/tlb.h | 1 - + arch/sparc/include/asm/pgtable_64.h | 2 +- + arch/sparc/include/asm/tlb_64.h | 1 - + arch/x86/include/asm/pgalloc.h | 1 - + arch/x86/virt/vmx/tdx/tdx.c | 1 + + arch/xtensa/kernel/hibernate.c | 1 + + block/fops.c | 1 + + drivers/char/agp/backend.c | 1 - + drivers/char/agp/generic.c | 1 - + drivers/char/agp/intel-agp.c | 1 - + drivers/char/agp/intel-gtt.c | 1 - + drivers/char/agp/uninorth-agp.c | 3 ++- + drivers/dma/ste_dma40.c | 1 + + drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gpuvm.c | 1 - + drivers/gpu/drm/amd/amdgpu/amdgpu_cs.c | 1 - + drivers/gpu/drm/amd/amdgpu/amdgpu_gem.c | 1 - + drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.c | 3 +-- + drivers/gpu/drm/amd/amdkfd/kfd_migrate.c | 1 + + drivers/gpu/drm/amd/amdkfd/kfd_priv.h | 1 - + drivers/gpu/drm/display/drm_dp_cec.c | 1 + + drivers/gpu/drm/etnaviv/etnaviv_gpu.c | 1 + + drivers/gpu/drm/i915/display/intel_display_power.c | 1 + + drivers/gpu/drm/i915/gem/i915_gem_userptr.c | 1 + + drivers/gpu/drm/msm/msm_fb.c | 2 ++ + drivers/gpu/drm/msm/msm_gem_shrinker.c | 2 ++ + drivers/gpu/drm/nouveau/nouveau_dmem.c | 1 + + drivers/gpu/drm/omapdrm/dss/hdmi4.c | 1 + + drivers/gpu/drm/omapdrm/dss/hdmi5.c | 1 + + drivers/gpu/drm/panel/panel-ilitek-ili9806e-core.c | 1 + + drivers/gpu/drm/panfrost/panfrost_gem.c | 2 ++ + drivers/gpu/drm/panthor/panthor_gem.c | 3 +++ + drivers/hwmon/pmbus/pmbus_core.c | 1 + + drivers/i2c/busses/i2c-amd-asf-plat.c | 1 + + drivers/i2c/busses/i2c-gxp.c | 1 + + drivers/i2c/busses/i2c-k1.c | 1 + + drivers/i2c/busses/i2c-pasemi-platform.c | 1 + + drivers/iio/accel/fxls8962af-core.c | 1 + + drivers/iio/adc/nct7201.c | 1 + + drivers/iio/adc/ti-ads1298.c | 1 + + drivers/iio/light/rohm-bu27034.c | 1 + + drivers/iio/pressure/bmp280-core.c | 1 + + drivers/iio/pressure/bmp280.h | 1 + + drivers/input/misc/aw86927.c | 1 + + drivers/input/touchscreen/goodix_berlin_core.c | 1 + + drivers/input/touchscreen/hynitron_cstxxx.c | 1 + + drivers/input/touchscreen/imagis.c | 1 + + drivers/leds/leds-turris-omnia.c | 1 + + drivers/media/common/videobuf2/frame_vector.c | 1 - + drivers/media/pci/cx18/cx18-driver.h | 1 - + drivers/media/pci/ivtv/ivtv-driver.h | 2 +- + drivers/media/platform/st/stm32/stm32-csi.c | 1 + + drivers/media/platform/ti/omap3isp/ispvideo.c | 1 - + drivers/media/usb/go7007/go7007-v4l2.c | 1 - + drivers/media/usb/gspca/gspca.c | 1 - + drivers/mfd/88pm886.c | 1 + + drivers/mfd/abx500-core.c | 1 + + drivers/mfd/adp5585.c | 1 + + drivers/mfd/cs40l50-core.c | 1 + + drivers/mfd/cs42l43.c | 1 + + drivers/mfd/rt5120.c | 1 + + drivers/mfd/tps65219.c | 1 + + drivers/mfd/twl-core.c | 2 +- + drivers/mmc/core/core.c | 1 - + drivers/mmc/core/host.c | 1 - + drivers/mmc/host/renesas_sdhi_internal_dmac.c | 1 - + drivers/mmc/host/renesas_sdhi_sys_dmac.c | 1 - + drivers/mmc/host/sh_mmcif.c | 1 - + drivers/mmc/host/tmio_mmc.h | 1 - + drivers/mmc/host/tmio_mmc_core.c | 1 - + drivers/mmc/host/usdhi6rol0.c | 1 - + drivers/net/ethernet/atheros/atl1c/atl1c.h | 1 - + drivers/net/ethernet/atheros/atl1e/atl1e.h | 1 - + drivers/net/ethernet/intel/e1000/e1000.h | 1 - + drivers/net/ethernet/intel/e1000e/netdev.c | 1 - + drivers/net/ethernet/intel/igb/igb_main.c | 1 - + drivers/net/ethernet/intel/igbvf/netdev.c | 1 - + drivers/phy/st/phy-stm32-combophy.c | 1 + + drivers/pinctrl/pinctrl-sx150x.c | 1 + + drivers/platform/arm64/huawei-gaokun-ec.c | 1 + + drivers/platform/arm64/lenovo-yoga-c630.c | 2 +- + .../platform/x86/intel/int3472/clk_and_regulator.c | 1 + + drivers/power/supply/bq25630_charger.c | 1 + + drivers/power/supply/rk817_charger.c | 2 +- + drivers/regulator/fixed.c | 1 + + drivers/regulator/fp9931.c | 1 + + drivers/regulator/mt6360-regulator.c | 1 + + drivers/regulator/qcom-labibb-regulator.c | 1 + + drivers/regulator/rtq2208-regulator.c | 1 + + drivers/regulator/tps65185.c | 1 + + drivers/regulator/tps65219-regulator.c | 1 + + drivers/regulator/tps6594-regulator.c | 1 + + drivers/soc/xilinx/zynqmp_power.c | 1 + + drivers/ufs/host/ufs-mediatek.c | 1 + + drivers/ufs/host/ufs-qcom.c | 1 + + drivers/usb/core/hcd.c | 1 + + drivers/usb/typec/mux/it5205.c | 1 + + drivers/usb/typec/wusb3801.c | 1 + + drivers/video/fbdev/core/fb_procfs.c | 1 + + drivers/xen/balloon.c | 1 + + fs/aio.c | 1 + + fs/buffer.c | 1 + + fs/file_table.c | 1 + + fs/inode.c | 1 + + fs/nfs/blocklayout/blocklayout.c | 1 + + fs/nfs/nfs42proc.c | 1 + + fs/nfs/nfstrace.h | 1 + + fs/nfs/pnfs.c | 1 + + fs/nfsd/nfs4proc.c | 1 + + include/drm/drm_print.h | 1 + + include/drm/ttm/ttm_tt.h | 1 - + include/linux/balloon.h | 1 - + include/linux/ceph/libceph.h | 1 - + include/linux/i3c/master.h | 1 + + include/linux/ksm.h | 1 - + include/linux/mempolicy.h | 1 - + include/linux/nfs_fs.h | 1 - + include/linux/nfs_page.h | 1 - + include/linux/suspend.h | 1 - + include/linux/swap.h | 1 - + include/sound/cs35l41.h | 1 + + kernel/power/snapshot.c | 1 + + net/ceph/osd_client.c | 1 - + net/ceph/pagevec.c | 1 + + net/core/datagram.c | 1 - + net/rds/rdma.c | 1 - + net/sunrpc/auth_gss/auth_gss.c | 1 - + net/sunrpc/auth_gss/gss_krb5_crypto.c | 1 - + net/sunrpc/auth_gss/gss_krb5_wrap.c | 1 - + net/sunrpc/auth_gss/svcauth_gss.c | 1 - + net/sunrpc/cache.c | 1 - + net/sunrpc/rpc_pipe.c | 1 - + net/sunrpc/socklib.c | 1 - + net/sunrpc/xdr.c | 1 - + net/sunrpc/xprtsock.c | 1 - + security/commoncap.c | 3 --- + security/inode.c | 1 - + security/selinux/hooks.c | 1 - + security/selinux/selinuxfs.c | 1 - + security/smack/smack_lsm.c | 1 - + sound/soc/codecs/cs35l41-lib.c | 1 + + 155 files changed, 98 insertions(+), 118 deletions(-) + delete mode 100644 arch/arc/include/asm/tlb.h + delete mode 100644 arch/openrisc/include/asm/tlb.h diff --git a/localversion-next b/localversion-next new file mode 100644 index 00000000000000..00c02a32fcdc86 --- /dev/null +++ b/localversion-next @@ -0,0 +1 @@ +-next-20260930 From 340590505d17fce619d2dff2388fba7ac655d568 Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Tue, 18 Aug 2026 20:14:06 +0000 Subject: [PATCH 1004/1012] ogc: linux-unstable: first commit -- 2.47.3 (cherry picked from commit 536526d1b587ace66924207d8dcbfd0a50279aec) (cherry picked from commit 0ebc551a74d0ee721a6f59870a5a88ff88e42268) (cherry picked from commit 2e5980a43d81039ee5a4b02fc8ff6637fead084d) -- 2.47.3 -- 2.47.3 (cherry picked from commit 82fa59412110c4bc4906a7f0e76c12fb533a7ec2) -- 2.47.3 (cherry picked from commit d62da27d35f725a5e3aa04a18c3f2387893ad262) (cherry picked from commit 9f7995156eb955d05d0fe39e76c1f33d40db26e8) (cherry picked from commit fc175a9a34c517a967c218f74f2d3768ff18d543) From dd4b6df2d44cbbc9faa4be94077ba78ddd5d7921 Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Tue, 1 Sep 2026 19:15:25 +0000 Subject: [PATCH 1005/1012] [NOT_FOR-UPSTREAM] ogc: linux-unstable: add github workflow (cherry picked from commit 64a16187d10090ade4d9134f4047f343e9856346) (cherry picked from commit a2c9baf1d2b5c60a0126f9e226187545905993cd) (cherry picked from commit 0424d14eecfff8eae20e0b78f9a62950f024f459) (cherry picked from commit 1cbd534c2db8c6fb3b3153708c9cbcff4975cf35) --- .github/packaging/PKGBUILD | 262 ++++++++++++++++++ .github/packaging/boot-smoke-test.sh | 234 ++++++++++++++++ .github/packaging/config.fragment | 124 +++++++++ .github/packaging/fedora/kernel.spec | 279 +++++++++++++++++++ .github/packaging/merge-fragments.sh | 62 +++++ .github/workflows/build-arch-packages.yml | 163 +++++++++++ .github/workflows/build-debian-packages.yml | 292 ++++++++++++++++++++ .github/workflows/build-fedora-packages.yml | 207 ++++++++++++++ .github/workflows/build-kernel.yml | 129 +++++++++ .github/workflows/publish-release.yml | 157 +++++++++++ .github/workflows/sync-linux-next.yml | 115 ++++++++ .github/workflows/test-pr.yml | 227 +++++++++++++++ 12 files changed, 2251 insertions(+) create mode 100644 .github/packaging/PKGBUILD create mode 100755 .github/packaging/boot-smoke-test.sh create mode 100644 .github/packaging/config.fragment create mode 100644 .github/packaging/fedora/kernel.spec create mode 100644 .github/packaging/merge-fragments.sh create mode 100644 .github/workflows/build-arch-packages.yml create mode 100644 .github/workflows/build-debian-packages.yml create mode 100644 .github/workflows/build-fedora-packages.yml create mode 100644 .github/workflows/build-kernel.yml create mode 100644 .github/workflows/publish-release.yml create mode 100644 .github/workflows/sync-linux-next.yml create mode 100644 .github/workflows/test-pr.yml diff --git a/.github/packaging/PKGBUILD b/.github/packaging/PKGBUILD new file mode 100644 index 00000000000000..85dcd59ff2abe7 --- /dev/null +++ b/.github/packaging/PKGBUILD @@ -0,0 +1,262 @@ +# SPDX-License-Identifier: GPL-2.0-only +# Maintainer: OpenGamingCollective +# +# PKGBUILD for the linux-unstable-ogc kernel. +# +# It is designed to be built by the CI job defined in +# .github/workflows/build-kernel.yml, which stages the following next to this +# file before invoking makepkg: +# - linux.tar.gz -> tarball of the checked-out kernel tree (no VCS data) +# - config -> Arch linux-headers .config with the OGC +# kernel-packages config fragments already applied +# - config.fragment -> repo-local overrides merged on top of it +# +# Based on the official Arch Linux kernel PKGBUILD and on the (known to work +# with linux-next) https://github.com/NeroReflex/linux-bisector PKGBUILD. + +pkgbase=linux-unstable-ogc +# Placeholder: makepkg refuses an empty pkgver before pkgver() runs; the real +# version is computed dynamically by pkgver() once sources are extracted. +pkgver=0.0.0 +pkgrel=1 +pkgdesc='linux-next kernel for the Open Gaming Collective' +url='https://github.com/OpenGamingCollective/linux-unstable' +arch=(x86_64) +license=(GPL2) +makedepends=( + bc + cpio + gettext + libelf + pahole + perl + python + tar + xz + gcc + ccache + git + + llvm + clang + lld + + # CONFIG_RUST=y in the base config + rust + rust-bindgen +) +options=('!strip') +source=( + 'linux.tar.gz' + 'config' + 'config.fragment' +) +b2sums=( + 'SKIP' + 'SKIP' + 'SKIP' +) + +export KBUILD_BUILD_HOST=archlinux +export KBUILD_BUILD_USER=$pkgbase +export KBUILD_BUILD_TIMESTAMP="" +export CC="ccache clang" +export MAKEFLAGS="-j$(nproc)" + +_make() { + test -s version + LLVM=1 LLVM_IAS=1 WERROR=0 KBUILD_BUILD_TIMESTAMP="" make CC="$CC" KERNELRELEASE="$(.. + $(cat localversion-next) + # + the git short sha recorded by CI in .build_commit + # e.g. "7.2.0" + "-next-20260818" + ".g4ca2fc86" -> "7.2.0-next-20260818.g4ca2fc86". + # Pacman does not allow hyphens in pkgver, so they become underscores: + # 7.2.0-next-20260818.g4ca2fc86 -> 7.2.0_next_20260818.g4ca2fc86 + cd "$srcdir/linux" + local ver + ver="$(make -s kernelversion)$(cat localversion-next 2>/dev/null || true)" + if [[ -s .build_commit ]]; then + ver+=".g$(<.build_commit)" + fi + printf '%s\n' "${ver//-/_}" +} + +prepare() { + cd "$srcdir/linux" + + # glibc >= 2.42 made strstr/strchr const-correct; some trees hardcode + # -Werror in tools/lib/bpf which then fails to compile. Keep builds green. + if [[ -f tools/lib/bpf/Makefile ]]; then + sed -i 's/ -Werror -Wall/ -Wall/' tools/lib/bpf/Makefile || true + fi + + echo "Setting version..." + # localversion* files are picked up sorted by name, so the final kernel + # release becomes e.g.: 7.2.0-next-20260818-unstable-ogc-g4ca2fc86-1 + # The -g suffix comes from the .build_commit file the CI writes into + # the tarball (same trick linux-bisector uses with .bisector_commit). + local _commit="" + if [[ -s .build_commit ]]; then + _commit="-g$(<.build_commit)" + fi + echo "${pkgbase#linux}${_commit}" > localversion.10-pkgname + echo "-$pkgrel" > localversion.20-pkgrel + LLVM=1 LLVM_IAS=1 make defconfig + LLVM=1 LLVM_IAS=1 make -s kernelrelease > version + LLVM=1 LLVM_IAS=1 make mrproper + + echo "Setting config..." + cp ../config .config + _make olddefconfig + + echo "Applying OGC config fragment..." + scripts/kconfig/merge_config.sh -m .config ../config.fragment + _make olddefconfig + + diff -u ../config .config || : + + echo "Prepared $pkgbase version $( +# Env: KREL expected kernel release string; when set, the boot banner +# "Linux version " must appear on every console log. +# BOOT_SMOKE_TIMEOUT override the per-boot timeout in seconds +# (default: 300 under KVM, 1200 under TCG). +# Requires (installed by the calling CI job): qemu-system-x86_64, cpio, +# gzip, a C compiler (gcc or clang), an OVMF build for the UEFI +# leg (packages: ovmf / edk2-ovmf), coreutils (timeout, find). + +set -euo pipefail + +die() { echo "::error::$*" >&2; exit 1; } + +IMAGE="${1:?usage: boot-smoke-test.sh }" +[ -s "$IMAGE" ] || die "kernel image '$IMAGE' not found or empty" + +for tool in qemu-system-x86_64 cpio gzip timeout sha256sum find truncate; do + command -v "$tool" >/dev/null || die "required tool '$tool' not installed" +done +CC_BIN="$(command -v gcc || command -v clang || true)" +[ -n "$CC_BIN" ] || die "no C compiler (gcc or clang) found" + +WORK="$(mktemp -d /tmp/boot-smoke.XXXXXX)" +trap 'rm -rf "$WORK"' EXIT + +# --------------------------------------------------------------------------- +# PID 1 for the test initramfs: freestanding, no libc, raw x86_64 syscalls, +# so it compiles with gcc or clang on any distro without extra static-libc +# packages (the Fedora build env has no glibc-static, for example). +# --------------------------------------------------------------------------- +cat > "$WORK/init.c" <<'EOF' +/* + * Tiny freestanding PID 1 for the QEMU boot smoke test: print BOOT_OK on + * the serial console, then power the VM off so QEMU exits on its own. + */ +static long sys3(long nr, long a, long b, long c) +{ + long ret; + + __asm__ volatile ("syscall" + : "=a"(ret) + : "a"(nr), "D"(a), "S"(b), "d"(c) + : "rcx", "r11", "memory"); + return ret; +} + +void _start(void) +{ + static const char msg[] = + "BOOT_OK: linux-unstable-ogc booted to userspace init\n"; + long fd; + + /* CONFIG_DEVTMPFS_MOUNT does not apply to initramfs, so mount + * devtmpfs ourselves to get /dev/ttyS0. */ + sys3(83, (long)"/dev", 0755, 0); /* mkdir */ + sys3(165, (long)"devtmpfs", (long)"/dev", (long)"devtmpfs"); /* mount */ + + /* Announce success on the serial port and on stdio (fd 1 exists + * only when the kernel could open /dev/console from the initramfs). */ + fd = sys3(2, (long)"/dev/ttyS0", 1, 0); /* open */ + if (fd >= 0) + sys3(1, fd, (long)msg, sizeof(msg) - 1); /* write */ + sys3(1, 1, (long)msg, sizeof(msg) - 1); /* write */ + + /* reboot(LINUX_REBOOT_CMD_POWER_OFF) -> ACPI S5 -> QEMU exits. */ + sys3(169, 0xfee1deadL, 672274793L, 0x4321fedcL); + for (;;) + sys3(34, 0, 0, 0); /* pause */ +} +EOF + +mkdir -p "$WORK/initramfs/dev" +"$CC_BIN" -Os -g0 -static -no-pie -nostdlib -ffreestanding \ + -fno-stack-protector -fno-asynchronous-unwind-tables \ + -fno-unwind-tables -Wl,-z,noexecstack \ + -o "$WORK/initramfs/init" "$WORK/init.c" +# /dev/console lets the kernel wire init's stdio to the console; creating it +# needs mknod privileges and may fail for unprivileged callers. Harmless: the +# init above then opens /dev/ttyS0 itself after mounting devtmpfs. +mknod -m 600 "$WORK/initramfs/dev/console" c 5 1 2>/dev/null || true + +(cd "$WORK/initramfs" && find . -print0 | cpio --null -o -H newc --quiet) \ + | gzip -1 > "$WORK/initrd.img" + +# Use KVM when the runner exposes it, software emulation (TCG) otherwise. +# TCG boots a distro-config kernel in a couple of minutes; the generous +# timeout also covers hangs, which are exactly what this test must catch. +if [ -w /dev/kvm ]; then + ACCEL="kvm" + ACCEL_ARGS=(-accel kvm -cpu host) + TMO=300 +else + ACCEL="tcg" + ACCEL_ARGS=(-accel tcg -cpu max) + TMO=1200 +fi + +# --------------------------------------------------------------------------- +# Locate an OVMF build for the UEFI leg. Layouts, by distro: +# Debian: /usr/share/OVMF/OVMF_CODE(_4M).fd (package: ovmf) +# Fedora: /usr/share/edk2/ovmf/OVMF_CODE.fd (package: edk2-ovmf) +# Arch: /usr/share/edk2/x64/OVMF_CODE.4m.fd (package: edk2-ovmf) +# qemu: /usr/share/qemu/edk2-x86_64-code.fd (qemu-system-data) +# CODE and VARS must be the same flash size, hence the paired candidates. +# --------------------------------------------------------------------------- +OVMF_PAIR="" +for pair in \ + "/usr/share/OVMF/OVMF_CODE_4M.fd:/usr/share/OVMF/OVMF_VARS_4M.fd" \ + "/usr/share/OVMF/OVMF_CODE.fd:/usr/share/OVMF/OVMF_VARS.fd" \ + "/usr/share/edk2/x64/OVMF_CODE.4m.fd:/usr/share/edk2/x64/OVMF_VARS.4m.fd" \ + "/usr/share/edk2/x64/OVMF_CODE.fd:/usr/share/edk2/x64/OVMF_VARS.fd" \ + "/usr/share/edk2/ovmf/OVMF_CODE.fd:/usr/share/edk2/ovmf/OVMF_VARS.fd" \ + "/usr/share/ovmf/x64/OVMF_CODE.fd:/usr/share/ovmf/x64/OVMF_VARS.fd" \ + "/usr/share/qemu/edk2-x86_64-code.fd:/usr/share/qemu/edk2-x86_64-vars.fd" \ + ; do + if [ -r "${pair%%:*}" ] && [ -r "${pair##*:}" ]; then + OVMF_PAIR="$pair" + break + fi +done +[ -n "$OVMF_PAIR" ] || die "no OVMF (UEFI) firmware found; install the 'ovmf' (Debian) or 'edk2-ovmf' (Arch/Fedora) package" +OVMF_CODE="${OVMF_PAIR%%:*}" +cp "${OVMF_PAIR##*:}" "$WORK/OVMF_VARS.fd" # pflash needs a writable copy +echo "UEFI firmware: $OVMF_CODE" + +# Blank disks so the storage controllers have something to enumerate. +for d in nvme virtio scsi; do + truncate -s 64M "$WORK/disk-$d.img" +done + +# Device spread common to every boot: covers the storage/USB/net drivers a +# gaming kernel is most likely to boot from. Attached whether or not the +# kernel contains them; drivers that are modules are simply not probed, +# which costs nothing. +DRIVERS_ARGS=( + -drive if=none,id=dsk-nvme,format=raw,file="$WORK/disk-nvme.img" + -device nvme,drive=dsk-nvme,serial=ogcsmoke + -drive if=none,id=dsk-virtio,format=raw,file="$WORK/disk-virtio.img" + -device virtio-blk-pci,drive=dsk-virtio + -drive if=none,id=dsk-scsi,format=raw,file="$WORK/disk-scsi.img" + -device virtio-scsi-pci,id=scsi0 + -device scsi-hd,drive=dsk-scsi + -device e1000e,netdev=net0 -netdev user,id=net0,restrict=on + -device qemu-xhci -device usb-tablet +) + +# run_boot — boot, wait for QEMU to exit or the +# timeout to fire, then fail hard if the marker (or the expected release +# banner) is missing from the serial console log. +run_boot() { + local name="$1"; shift + local serial="$WORK/serial-$name.log" + : > "$serial" + + echo "[$name] Booting $(basename "$IMAGE") (sha256 $(sha256sum "$IMAGE" | cut -c1-16)...) with QEMU ($ACCEL, timeout ${TMO}s)" + timeout "$TMO" qemu-system-x86_64 "${ACCEL_ARGS[@]}" "$@" \ + "${DRIVERS_ARGS[@]}" \ + -m 2048 -smp 2 -nodefaults \ + -display none -monitor none -no-reboot \ + -serial "file:$serial" \ + -kernel "$IMAGE" -initrd "$WORK/initrd.img" \ + -append "console=ttyS0,115200n8 rdinit=/init panic=-1 nokaslr" \ + || true + # A panic with panic=-1 reboots instantly and -no-reboot makes QEMU + # exit, so both "qemu exited by itself" and "timeout killed it" end up + # in the marker check below. + + if ! grep -q "BOOT_OK" "$serial"; then + echo "::error::QEMU boot smoke test FAILED in the '$name' configuration: no BOOT_OK marker on the serial console (accel=$ACCEL, timeout=${TMO}s). The kernel is unbootable." + echo "----- last 250 lines of the $name guest serial console -----" + tail -n 250 "$serial" || true + echo "-------------------------------------------------------------" + exit 1 + fi + if [ -n "${KREL:-}" ] && ! grep -qF "Linux version ${KREL} " "$serial"; then + echo "::error::QEMU boot smoke test FAILED in the '$name' configuration: the booted kernel banner does not advertise release '${KREL}'." + grep -m1 "^Linux version" "$serial" || true + exit 1 + fi + echo "[$name] Boot smoke test PASSED: kernel booted to userspace init and powered off. Serial console tail:" + tail -n 10 "$serial" || true +} + +# 1. Classic BIOS boot: SeaBIOS firmware, i440fx (PIIX IDE built in). +run_boot bios-pc -machine pc + +# 2. UEFI boot: OVMF firmware, q35 (ICH9 AHCI + PCIe built in), kernel +# launched through the EFI stub — the path handhelds actually use. +run_boot uefi-q35 \ + -machine q35 \ + -drive if=pflash,format=raw,readonly=on,file="$OVMF_CODE" \ + -drive if=pflash,format=raw,file="$WORK/OVMF_VARS.fd" + +if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then + { + echo '### QEMU boot smoke test' + echo + echo "- Image: $(basename "$IMAGE") (sha256 $(sha256sum "$IMAGE" | cut -c1-12)...)" + echo "- Accelerator: ${ACCEL}, timeout ${TMO}s per boot" + echo "- Matrix: bios-pc (SeaBIOS) and uefi-q35 (OVMF EFI stub), NVMe + virtio-blk + virtio-scsi + e1000e + xHCI attached" + echo '- Result: BOOT_OK in both configurations - kernel decompressed, booted and reached userspace init' + } >> "$GITHUB_STEP_SUMMARY" +fi diff --git a/.github/packaging/config.fragment b/.github/packaging/config.fragment new file mode 100644 index 00000000000000..8d455d78cc2417 --- /dev/null +++ b/.github/packaging/config.fragment @@ -0,0 +1,124 @@ +# OGC config overrides for linux-unstable-ogc +# +# This fragment is merged LAST, on top of the OGC config (Arch linux-headers +# base + OpenGamingCollective/kernel-packages config/*.{set,unset} fragments) +# via scripts/kconfig/merge_config.sh + olddefconfig in the PKGBUILD. +# +# Target hardware: modern gaming desktops and handhelds (Steam Deck, ROG Ally, +# Legion Go, MSI Claw, GPD, Ayaneo, OneXPlayer...). Keep IMU/gyro/accel/IIO +# drivers, Bluetooth, UVC webcams and all OGC handheld drivers (those live in +# drivers/hid, drivers/platform/x86 — NOT in staging, so STAGING can go). + +# --- Kernel identity -------------------------------------------------------- +# Distro base configs carry their own release suffix (Arch "-arch1", Debian +# "-1-amd64", ...). It must be empty so that uname -r is identical across +# the Arch/Fedora/Debian packages: -unstable-ogc-g-1 (the suffix +# itself comes from the localversion.10/20 files written by the packaging). +CONFIG_LOCALVERSION="" +# CONFIG_LOCALVERSION_AUTO is not set + +# --- Build fixes ------------------------------------------------------------ +# drivers/net/ethernet/cadence/macb_main.c fails to compile in linux-next +# (implicit declarations of macb_alloc_tieoff/macb_free_tieoff). Cadence +# MACB/GEM is an ARM SoC NIC, never present on x86_64 gaming hardware. +# MACB_PCI and MACB_USE_HWSTAMP depend on MACB and drop out automatically. +# CONFIG_MACB is not set + +# --- Build speed ------------------------------------------------------------ +# Keep DEBUG_INFO/DEBUG_INFO_BTF=y (OGC requires them for BPF/sched_ext) but +# skip running pahole on every module (thousands of invocations saved). +# CONFIG_DEBUG_INFO_BTF_MODULES is not set + +# --- Xen guest support (not running inside a Xen VM) ------------------------ +# CONFIG_XEN is not set + +# --- Ancient graphics / buses ------------------------------------------------ +# AGP is a pre-PCIe bus; radeon covers pre-GCN AMD GPUs (12+ years old). +# CONFIG_AGP is not set +# CONFIG_DRM_RADEON is not set +# PCMCIA/CardBus slots (2000s laptops) +# CONFIG_PCCARD is not set +# FireWire (IEEE 1394) ports +# CONFIG_FIREWIRE is not set +# Parallel port (printers/scanners from the 90s) +# CONFIG_PARPORT is not set +# IndustryPack carrier boards +# CONFIG_IPACK_BUS is not set +# 1-Wire bus +# CONFIG_W1 is not set +# Analog gameport joysticks (pre-USB) +# CONFIG_GAMEPORT is not set +# Floppy disk controller +# CONFIG_BLK_DEV_FD is not set +# Server BMC management +# CONFIG_IPMI_HANDLER is not set + +# --- Enterprise / datacenter storage ---------------------------------------- +# Fibre Channel HBAs and enterprise RAID controllers +# CONFIG_SCSI_LPFC is not set +# CONFIG_SCSI_QLA_FC is not set +# CONFIG_QEDF is not set +# CONFIG_FCOE is not set +# CONFIG_LIBFC is not set +# CONFIG_MEGARAID_SAS is not set +# CONFIG_MEGARAID_LEGACY is not set +# CONFIG_MEGARAID_MAILBOX is not set +# CONFIG_FUSION is not set +# CONFIG_SCSI_HPSA is not set +# CONFIG_SCSI_SMARTPQI is not set + +# --- Ancient Ethernet NICs --------------------------------------------------- +# Vendor menus for 10/100 hardware from the 90s/00s (3Com, Tulip, NatSemi, +# NE2000-era, SiS 900, VIA Rhine/Velocity, Adaptec starfire). Modern NICs +# (Intel/Realtek/Marvell/Broadcom/Aquantia) are untouched. +# CONFIG_NET_VENDOR_3COM is not set +# CONFIG_NET_VENDOR_ADAPTEC is not set +# CONFIG_NET_TULIP is not set +# CONFIG_NET_VENDOR_NATSEMI is not set +# CONFIG_NET_VENDOR_8390 is not set +# CONFIG_NET_VENDOR_SIS is not set +# CONFIG_NET_VENDOR_VIA is not set + +# --- Specialized/legacy networking ------------------------------------------ +# CONFIG_ATM is not set +# CONFIG_RDS is not set +# CONFIG_TIPC is not set +# CONFIG_PHONET is not set +# CONFIG_IEEE802154 is not set +# CONFIG_CAN is not set +# CONFIG_NFC is not set +# CONFIG_INFINIBAND is not set + +# --- Analog/digital TV, radio, legacy webcams -------------------------------- +# UVC (USB_VIDEO_CLASS) webcams stay enabled — handhelds use them. +# CONFIG_MEDIA_ANALOG_TV_SUPPORT is not set +# CONFIG_MEDIA_DIGITAL_TV_SUPPORT is not set +# CONFIG_MEDIA_RADIO_SUPPORT is not set +# CONFIG_USB_GSPCA is not set + +# --- Ancient/cluster filesystems ---------------------------------------------- +# Desktop FS (ext4/btrfs/xfs/f2fs/exfat/ntfs3/vfat/nfs/cifs) untouched. +# CONFIG_MINIX_FS is not set +# CONFIG_UFS_FS is not set +# CONFIG_BFS_FS is not set +# CONFIG_GFS2_FS is not set +# CONFIG_OCFS2_FS is not set +# CONFIG_AFS_FS is not set + +# --- Chemical / gas / air-quality sensors ------------------------------------ +# IMU/gyro/accelerometer/pressure/temp/humidity IIO drivers are KEPT +# (handheld controllers need them); only gas/VOC/CO2/PM sensors removed. +# CONFIG_CCS811 is not set +# CONFIG_SPS30 is not set +# CONFIG_PMS7003 is not set +# CONFIG_SENSIRION_SGP30 is not set +# CONFIG_SENSIRION_SGP40 is not set +# CONFIG_SCD30_CORE is not set +# CONFIG_SCD4X is not set +# CONFIG_VZ89X is not set +# CONFIG_BME680 is not set + +# --- Staging drivers ---------------------------------------------------------- +# All OGC handheld drivers live in mainline trees (drivers/hid, +# drivers/platform/x86), not staging. +# CONFIG_STAGING is not set \ No newline at end of file diff --git a/.github/packaging/fedora/kernel.spec b/.github/packaging/fedora/kernel.spec new file mode 100644 index 00000000000000..2165fefce635a9 --- /dev/null +++ b/.github/packaging/fedora/kernel.spec @@ -0,0 +1,279 @@ +# SPDX-License-Identifier: GPL-2.0-only +# +# RPM spec for the linux-unstable-ogc kernel (linux-next based). +# +# Built by .github/workflows/build-kernel.yml ("fedora" job) which, before +# invoking rpmbuild: +# - substitutes @@KBASEVER@@ / @@KVERDOTTED@@ / @@SHA8@@ placeholders below +# - stages SOURCES/linux.tar.gz (kernel tree contents, root dir "linux/", +# no VCS data, localversion-next included) +# - stages SOURCES/config (Fedora kernel-core .config with the OGC +# kernel-packages fragments and the local +# config.fragment already applied) +# +# Derived from the OpenGamingCollective/kernel-packages fedora/kernel.spec +# (itself based on CachyOS/Nobara), trimmed down to core/modules/devel only. +# The kernel is compiled with clang (LLVM=1) exactly like the Arch packages. + +%global _default_patch_fuzz 2 + +# See https://fedoraproject.org/wiki/Changes/SetBuildFlagsBuildCheck +%if 0%{?fedora} >= 37 +%undefine _auto_set_build_flags +%endif + +%define _build_id_links none +%define _disable_source_fetch 1 +# no debuginfo generation, no brp strip/mangle of kernel binaries +%define debug_package %{nil} +%define __spec_install_post /usr/lib/rpm/brp-compress || : + +# ---- substituted by CI ------------------------------------------------------ +%define kbasever @@KBASEVER@@ +%define sha8 @@SHA8@@ +# ---------------------------------------------------------------------------- + +Version: @@KVERDOTTED@@ +Release: 1.g%{sha8}%{?dist} + +%define rpmver %{version}-%{release} +# Kernel release string (uname -r). Identical to what the localversion* files +# below produce during the build, and identical to the Arch packages built +# from the same commit: +# -unstable-ogc-g-1 +%define kverstr %{kbasever}-unstable-ogc-g%{sha8}-1 +# RPM dependency versions may contain at most ONE hyphen (V-R separator), so +# the full kernel release string cannot be used as a Provides: version. Use a +# 1:1 dotted translation for the *-uname-r provides instead. +%define kverdot %(echo "%{kverstr}" | sed -e "s/-/./g") + +Name: kernel-unstable-ogc +Summary: The linux-next kernel for the Open Gaming Collective +License: GPLv2 +URL: https://github.com/OpenGamingCollective/linux-unstable +Group: System Environment/Kernel +ExclusiveArch: x86_64 +Source0: linux.tar.gz +Source1: config + +BuildRequires: bash, coreutils, make, tar, findutils, gawk, diffutils, m4 +BuildRequires: bc, bison, flex, perl-interpreter, perl-Carp, binutils +BuildRequires: xz, zstd, kmod, python3 +BuildRequires: elfutils-libelf-devel, elfutils-devel +BuildRequires: openssl, openssl-devel +BuildRequires: dwarves, hmaccalc +# clang toolchain (like the Arch packages) +BuildRequires: clang, lld, llvm, ccache +# needed when the base config enables CONFIG_RUST. The bindgen binary +# package was renamed from rust-bindgen to bindgen in newer Fedora releases; +# accept either so this spec works across distro versions. +BuildRequires: rust, rust-src +BuildRequires: (bindgen or rust-bindgen) + +# All kernel make invocations: clang via ccache, deterministic version strings +%define kmake make CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc KBUILD_BUILD_TIMESTAMP="" + +%description +This package is a meta package that pulls in the linux-unstable-ogc kernel +(a linux-next snapshot for the Open Gaming Collective) and its matching +modules. + +%package core +Summary: The linux-unstable-ogc kernel (vmlinuz and core files) +Group: System Environment/Kernel +Provides: installonlypkg(kernel) +Provides: %{name}-core-uname-r = %{kverdot} +Requires: bash, coreutils, kmod +Requires: /usr/bin/kernel-install +Requires: %{name}-modules = %{rpmver} +Recommends: linux-firmware +%description core +This package contains the linux-unstable-ogc kernel image (vmlinuz), +System.map, the build configuration and the module symbol version file. + +%package modules +Summary: Kernel modules to match the linux-unstable-ogc core kernel +Group: System Environment/Kernel +Provides: installonlypkg(kernel-module) +Provides: %{name}-modules-uname-r = %{kverdot} +Supplements: %{name}-core = %{rpmver} +# kmod needed for depmod in %%post +Requires: kmod +%description modules +This package provides the kernel modules for the linux-unstable-ogc kernel. + +%package devel +Summary: Development files for building external modules +Group: Development/System +AutoReqProv: no +Requires: findutils, make, perl-interpreter, flex, bison +Requires: elfutils-libelf-devel, openssl-devel, gcc +Requires: clang, llvm, lld +Provides: %{name}-devel-uname-r = %{kverdot} +Enhances: akmods +Enhances: dkms +%description devel +This package provides the headers, scripts and tooling (objtool, +resolve_btfids) needed to build out-of-tree kernel modules against the +linux-unstable-ogc kernel (%{kverstr}). + +%prep +%setup -q -n linux + +cp %{SOURCE1} .config + +# The Fedora distro config references Fedora-only certificate files that do +# not exist in this tree; use the ephemeral in-tree key instead. +scripts/config --set-str SYSTEM_TRUSTED_KEYS "" +scripts/config --set-str SYSTEM_REVOCATION_KEYS "" +# CONFIG_MODULE_SIG_KEY points at the Red Hat signing cert in the distro +# config, which does not exist in this tree. Reset it to the kbuild default +# ("certs/signing_key.pem"): an ephemeral self-signed key generated during +# the build and trusted by the kernel itself. The Fedora config sets +# CONFIG_MODULE_SIG_ALL=y, and with an empty key string scripts/Makefile.modinst +# resolves sig-key to "./", making sign-file read a directory as the private +# key (SSL DECODER error) on every module. +scripts/config --set-str MODULE_SIG_KEY "certs/signing_key.pem" + +# Deterministic / distro-agnostic build identity +scripts/config -u DEFAULT_HOSTNAME +scripts/config --set-str BUILD_SALT "%{kverstr}" + +# Kernel release suffix, same scheme as the Arch packages: +# -unstable-ogc-g-1 +echo "-unstable-ogc-g%{sha8}" > localversion.10-pkgname +echo "-1" > localversion.20-pkgrel + +# glibc >= 2.42 const-correctness vs -Werror in tools/lib/bpf (used by +# resolve_btfids when DEBUG_INFO_BTF=y); keep the build green. +if [ -f tools/lib/bpf/Makefile ]; then + sed -i 's/ -Werror -Wall/ -Wall/' tools/lib/bpf/Makefile || true +fi + +%{kmake} olddefconfig + +# Fail fast if the release string ever drifts from the spec +REL="$(make -s kernelrelease)" +if [ "$REL" != "%{kverstr}" ]; then + echo "kernelrelease '$REL' does not match spec kverstr '%{kverstr}'" >&2 + exit 1 +fi +cp .config config-linux-unstable-ogc + +%build +%{kmake} %{?_smp_mflags} all + +%install +MODDIR="%{buildroot}/lib/modules/%{kverstr}" +DEVEL="%{buildroot}%{_prefix}/src/kernels/%{kverstr}" + +mkdir -p "%{buildroot}/boot" "$MODDIR" + +echo "Installing boot image..." +ImageName="$(make -s image_name | tail -n 1)" +install -m 0644 "$ImageName" "$MODDIR/vmlinuz" +chmod 0755 "$MODDIR/vmlinuz" + +echo "Installing modules..." +# Modules are compressed by modules_install itself (CONFIG_MODULE_COMPRESS_*). +# depmod runs from %%post at install time. +%{kmake} %{?_smp_mflags} KERNELRELEASE=%{kverstr} \ + INSTALL_MOD_PATH=%{buildroot} INSTALL_MOD_STRIP=1 \ + DEPMOD=/doesnt/exist modules_install + +echo "Installing core files..." +cp System.map "$MODDIR/System.map" +cp .config "$MODDIR/config" +gzip -c9 < Module.symvers > "$MODDIR/symvers.gz" +(cd "$MODDIR" && sha512hmac vmlinuz > .vmlinuz.hmac) + +# ---- kernel-devel ----------------------------------------------------------- +echo "Preparing kernel-devel..." +rm -f "$MODDIR"/build "$MODDIR"/source +ln -s "%{_prefix}/src/kernels/%{kverstr}" "$MODDIR/build" +(cd "$MODDIR" && ln -s build source) +mkdir -p "$MODDIR"/updates "$MODDIR"/weak-updates "$DEVEL" + +find . -type f \( -name 'Makefile*' -o -name 'Kconfig*' \) -print0 \ + | xargs -0 cp --parents -t "$DEVEL" +cp -a include "$DEVEL"/ +cp -a arch/x86/include "$DEVEL"/arch/x86/ +if [ -f arch/x86/kernel/module.lds ]; then + cp -a --parents arch/x86/kernel/module.lds "$DEVEL"/ +fi +cp -a scripts "$DEVEL"/ +rm -rf "$DEVEL"/scripts/tracing +rm -f "$DEVEL"/scripts/spdxcheck.py +cp Module.symvers System.map .config "$DEVEL"/ + +mkdir -p "$DEVEL"/tools/{objtool,bpf/resolve_btfids,lib,build} +cp -a tools/objtool/objtool "$DEVEL"/tools/objtool/ || : +cp -a tools/bpf/resolve_btfids/resolve_btfids "$DEVEL"/tools/bpf/resolve_btfids/ || : +cp -a tools/include "$DEVEL"/tools/ +cp -a tools/lib/subcmd "$DEVEL"/tools/lib/ +cp -a tools/lib/bpf "$DEVEL"/tools/lib/ +cp -a tools/build/Build.include tools/build/fixdep.c "$DEVEL"/tools/build/ +cp -a tools/scripts/utilities.mak "$DEVEL"/tools/scripts/ 2>/dev/null || : +cp -a --parents arch/x86/entry/syscalls/syscall_32.tbl "$DEVEL"/ +cp -a --parents arch/x86/entry/syscalls/syscall_64.tbl "$DEVEL"/ +cp -a arch/x86/tools "$DEVEL"/arch/x86/ + +# Drop intermediate build artifacts from devel tree +find "$DEVEL" \( -name '*.o' -o -name '*.cmd' -o -name '.*.cmd' \) -delete + +# Timestamps must line up so external module builds do not rerun kconfig +touch -r "$DEVEL"/Makefile \ + "$DEVEL"/include/generated/uapi/linux/version.h \ + "$DEVEL"/include/config/auto.conf + +%post core +# nothing to do at this point + +%posttrans core +# Runs after ALL packages of this transaction have been installed and their +# %%post scriptlets (incl. depmod from -modules) have run. +if [ -x /usr/bin/kernel-install ]; then + /usr/bin/kernel-install add %{kverstr} /lib/modules/%{kverstr}/vmlinuz || exit $? +fi +if [ -x /usr/sbin/grubby ]; then + grubby --set-default="/boot/vmlinuz-%{kverstr}" || : +fi + +%preun core +if [ "$1" = "0" ] && [ -x /usr/bin/kernel-install ]; then + /usr/bin/kernel-install remove %{kverstr} /lib/modules/%{kverstr}/vmlinuz || exit $? +fi + +%post modules +/sbin/depmod -a %{kverstr} + +%files +# meta package: everything lives in the subpackages + +%files core +%ghost /boot/vmlinuz-%{kverstr} +%ghost /boot/initramfs-%{kverstr}.img +/lib/modules/%{kverstr}/vmlinuz +/lib/modules/%{kverstr}/.vmlinuz.hmac +/lib/modules/%{kverstr}/System.map +/lib/modules/%{kverstr}/config +/lib/modules/%{kverstr}/symvers.gz + +%files modules +/lib/modules/%{kverstr}/ +%exclude /lib/modules/%{kverstr}/vmlinuz +%exclude /lib/modules/%{kverstr}/.vmlinuz.hmac +%exclude /lib/modules/%{kverstr}/System.map +%exclude /lib/modules/%{kverstr}/config +%exclude /lib/modules/%{kverstr}/symvers.gz +%exclude /lib/modules/%{kverstr}/build +%exclude /lib/modules/%{kverstr}/source + +%files devel +/usr/src/kernels/%{kverstr} +/lib/modules/%{kverstr}/build +/lib/modules/%{kverstr}/source + +%changelog +* Thu Aug 20 2026 OpenGamingCollective CI +- Initial linux-unstable-ogc spec, generated by CI from linux-next. \ No newline at end of file diff --git a/.github/packaging/merge-fragments.sh b/.github/packaging/merge-fragments.sh new file mode 100644 index 00000000000000..e8525b2d86ce8e --- /dev/null +++ b/.github/packaging/merge-fragments.sh @@ -0,0 +1,62 @@ +#!/usr/bin/env bash +# Textually merge kernel config fragments into a base .config file. +# +# Usage: merge-fragments.sh [...] +# +# Recognized fragment line formats (fragments are applied in order, later +# fragments win): +# CONFIG_X=value set X to value +# CONFIG_X unset X (kernel-configurator *.unset format) +# "# CONFIG_X is not set" unset X (kconfig fragment format) +# Blank lines and other comments are ignored. +# +# This replicates the semantics of the OpenGamingCollective +# kernel-configurator action (*.config.set / *.config.unset fragments) and +# additionally accepts regular kconfig fragment files, so both the OGC +# fragments and the repo-local config.fragment can be handled uniformly. +# The final `make olddefconfig` (run by the PKGBUILD / RPM spec) turns the +# textual result into a consistent kconfig. + +set -euo pipefail + +if [ "$#" -lt 2 ]; then + echo "usage: $0 ..." >&2 + exit 2 +fi + +CFG="$(realpath "$1")" +shift + +set_key() { + local key="$1" value="$2" + if grep -q "^${key}=" "$CFG"; then + sed -i "s|^${key}=.*|${key}=${value}|" "$CFG" + elif grep -q "^# ${key} is not set$" "$CFG"; then + sed -i "s|^# ${key} is not set\$|${key}=${value}|" "$CFG" + else + printf '%s=%s\n' "${key}" "${value}" >> "$CFG" + fi +} + +unset_key() { + local key="$1" + if grep -q "^${key}=" "$CFG"; then + sed -i "s|^${key}=.*|# ${key} is not set|" "$CFG" + fi +} + +for frag in "$@"; do + frag="$(realpath "$frag")" + while IFS= read -r line || [ -n "$line" ]; do + line="${line#"${line%%[![:space:]]*}"}" + line="${line%"${line##*[![:space:]]}"}" + [ -z "$line" ] && continue + if [[ "$line" =~ ^#\ (CONFIG_[A-Za-z0-9_]+)\ is\ not\ set$ ]]; then + unset_key "${BASH_REMATCH[1]}" + elif [[ "$line" =~ ^(CONFIG_[A-Za-z0-9_]+)=(.*)$ ]]; then + set_key "${BASH_REMATCH[1]}" "${BASH_REMATCH[2]}" + elif [[ "$line" =~ ^(CONFIG_[A-Za-z0-9_]+)$ ]]; then + unset_key "${BASH_REMATCH[1]}" + fi + done < "$frag" +done \ No newline at end of file diff --git a/.github/workflows/build-arch-packages.yml b/.github/workflows/build-arch-packages.yml new file mode 100644 index 00000000000000..748a422d732aa4 --- /dev/null +++ b/.github/workflows/build-arch-packages.yml @@ -0,0 +1,163 @@ +# Arch Linux packaging, split out of build-kernel.yml. +# Called (via `uses:`) by build-kernel.yml; runs inside the same workflow +# run, so the artifacts uploaded here are seen by publish-release.yml. + +name: Build Arch Linux packages + +on: + workflow_call: + inputs: + krel: + description: Full kernel release string (uname -r), e.g. 6.12.0-next-20250101-unstable-ogc-g12345678-1 + type: string + required: true + +permissions: + contents: read + +env: + # Config fragments maintained by the OGC kernel-packages repository. + # The distro base config is extracted from the official Arch + # `linux-headers` package — same approach as the kernel-packages + # arch.yaml workflow. + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + build: + name: Build Arch Linux packages + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/archlinux:base-devel + env: + PKGDIR: /tmp/pkgbuild + DISTDIR: /tmp/dist + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + KREL: ${{ inputs.krel }} + steps: + - name: Show disk space + run: df -h / + + - name: Bootstrap Arch Linux build environment + run: | + set -euxo pipefail + # Refresh keyring first to avoid signature failures on stale images + pacman -Sy --needed --noconfirm archlinux-keyring + pacman -Su --needed --noconfirm + # Everything makepkg needs (mirrors the makedepends of the PKGBUILD). + # rust + rust-bindgen are needed because the Arch config enables + # CONFIG_RUST=y. + pacman -S --needed --noconfirm \ + bc cpio gettext libelf pahole perl python tar xz zstd \ + gcc clang llvm lld ccache pigz file curl git \ + rust rust-bindgen + # makepkg refuses to run as root + useradd -m build + install -d -o build -g build -m 0777 /ccache + + - name: Download staged sources + uses: actions/download-artifact@v8 + with: + name: kernel-sources + path: /tmp/stage + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-arch-${{ github.sha }} + restore-keys: | + ccache-arch- + + - name: Prepare ccache for the build user + run: chown -R build:build /ccache && su build -c 'ccache -s' || true + + - name: Assemble Arch kernel config + working-directory: /tmp/stage + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: extract the official Arch Linux kernel .config from + # the distro's `linux-headers` package (same method as the OGC + # kernel-packages arch.yaml workflow). + # ------------------------------------------------------------------ + CACHE="/tmp/pkgcache" + mkdir -p "$CACHE" + pacman -Sw --noconfirm --cachedir "$CACHE" linux-headers + PKG="$(find "$CACHE" -maxdepth 1 -name 'linux-headers-*.pkg.tar.zst' -type f)" + if [ -z "$PKG" ]; then + echo "::error::linux-headers package not found in cache"; exit 1 + fi + CFG="$(tar --zstd -tf "$PKG" | grep -E '^usr/lib/modules/[^/]+/build/\.config$' || true)" + if [ -z "$CFG" ]; then + echo "::error::kernel .config not found inside $PKG"; exit 1 + fi + tar --zstd -xOf "$PKG" "$CFG" > config + test -s config + rm -rf "$CACHE" + + # ------------------------------------------------------------------ + # Layer the OGC config fragments from kernel-packages on top, then + # the repo-local config.fragment (which wins). Order is critical: + # unsets apply first (removing things we explicitly don't want), + # then sets apply (enabling things we do want), so explicit enables + # can override explicit disables. + # ------------------------------------------------------------------ + for f in arch.config.set ogc.config.set arch.config.unset ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash ./merge-fragments.sh config \ + arch.config.unset ogc.config.unset \ + arch.config.set ogc.config.set \ + config.fragment + echo "Config after fragment merge (head):" + head -n 3 config + + - name: Stage PKGBUILD directory + run: | + set -euxo pipefail + mkdir -p "${PKGDIR}" "${DISTDIR}" + cd /tmp/stage + cp PKGBUILD config.fragment linux.tar.gz config "${PKGDIR}/" + chown -R build:build "${PKGDIR}" + + - name: Build packages with makepkg + run: | + set -euxo pipefail + cd "${PKGDIR}" + runuser -u build -- env HOME=/home/build CCACHE_DIR=/ccache \ + makepkg -f --noconfirm --noprogressbar + + ls -lh ./*.pkg.tar.zst + cp -v ./*.pkg.tar.zst "${DISTDIR}/" + # Ship the exact .config used for the build as well + cp -v "src/linux/.config" "${DISTDIR}/config-arch-${KREL}" + + - name: Boot smoke test in QEMU (release gate) + run: | + set -euxo pipefail + pacman -S --needed --noconfirm qemu-system-x86 edk2-ovmf + # ---------------------------------------------------------------- + # Boot the exact kernel image that ships in the package. This + # step fails the job (and with it the release) if the kernel + # panics, hangs or never reaches userspace, so an unbootable + # kernel is never uploaded or published. + # ---------------------------------------------------------------- + WORK=/tmp/bootsmoke + rm -rf "$WORK"; mkdir -p "$WORK" + PKG="$(find /tmp/dist -name 'linux-unstable-ogc-*.pkg.tar.zst' ! -name '*-headers-*' | head -n1)" + test -n "$PKG" + tar --zstd -xf "$PKG" -C "$WORK" "usr/lib/modules/${KREL}/vmlinuz" + test -s "$WORK/usr/lib/modules/${KREL}/vmlinuz" + bash /tmp/stage/boot-smoke-test.sh "$WORK/usr/lib/modules/${KREL}/vmlinuz" + + - name: Clean build tree (free disk before cache save) + run: rm -rf "${PKGDIR}/src" "${PKGDIR}/pkg" ; df -h / + + - name: Upload Arch packages + uses: actions/upload-artifact@v7 + with: + name: arch-packages + path: /tmp/dist + retention-days: 3 diff --git a/.github/workflows/build-debian-packages.yml b/.github/workflows/build-debian-packages.yml new file mode 100644 index 00000000000000..6eef922dc09eb4 --- /dev/null +++ b/.github/workflows/build-debian-packages.yml @@ -0,0 +1,292 @@ +# Debian .deb packaging, split out of build-kernel.yml. +# Called (via `uses:`) by build-kernel.yml; runs inside the same workflow +# run, so the artifacts uploaded here are seen by publish-release.yml. + +name: Build Debian packages + +on: + workflow_call: + inputs: + krel: + description: Full kernel release string (uname -r), e.g. 6.12.0-next-20250101-unstable-ogc-g12345678-1 + type: string + required: true + kverdot: + description: Kernel version with hyphens replaced by dots, for KDEB_PKGVERSION + type: string + required: true + sha8: + description: Short (8 char) source commit the packages are built from + type: string + required: true + +permissions: + contents: read + +env: + # Config fragments maintained by the OGC kernel-packages repository. + # The distro base config comes from the official Debian + # linux-config/linux-image packages. + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + build: + name: Build Debian packages + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/library/debian:trixie + env: + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + KREL: ${{ inputs.krel }} + KVERDOT: ${{ inputs.kverdot }} + SHA8: ${{ inputs.sha8 }} + steps: + - name: Show disk space + run: df -h / + + - name: Install build tools + run: | + set -euxo pipefail + apt-get update + # Satisfies the Build-Depends generated by scripts/package/mkdebian + # (debhelper-compat, bc, bison, flex, kmod, libdw/libelf/libssl-dev, + # python3, rsync) plus the clang toolchain used by all OGC builds. + apt-get install -y --no-install-recommends \ + build-essential debhelper rsync \ + bc bison flex python3 kmod \ + libelf-dev libdw-dev libssl-dev zlib1g-dev \ + dwarves zstd xz-utils \ + clang llvm lld ccache curl ca-certificates git \ + openssl + # dpkg-buildpackage refuses to run as root + useradd -m builder + install -d -o builder -g builder -m 0777 /ccache + + - name: Download staged sources + uses: actions/download-artifact@v8 + with: + name: kernel-sources + path: /tmp/stage + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-debian-${{ github.sha }} + restore-keys: | + ccache-debian- + + - name: Assemble Debian kernel config + working-directory: /tmp/stage + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: the official Debian kernel configuration. It ships in + # the small linux-config- package; fall back to /boot/config-* + # of the linux-image- package if that is unavailable. + # ------------------------------------------------------------------ + ABI="$(apt-cache depends linux-image-amd64 \ + | awk '/^ *Depends: *linux-image-[0-9]/ {print $2; exit}' \ + | sed 's/^linux-image-//')" + test -n "${ABI}" + echo "Debian kernel ABI: ${ABI}" + # linux-config is versioned by major.minor only (e.g. 6.12) and + # stores the per-flavour config xz-compressed, e.g. + # /usr/src/linux-config-6.12/config.amd64_none_amd64.xz + KMAJMIN="$(printf '%s' "${ABI}" | cut -d. -f1,2)" + mkdir -p /tmp/pkgcfg + if apt-get download "linux-config-${KMAJMIN}"; then + dpkg-deb -x linux-config-"${KMAJMIN}"_*.deb /tmp/pkgcfg + BASE="$(find /tmp/pkgcfg/usr/src -name 'config.amd64_none_amd64.xz' | head -n1)" + test -s "${BASE}" + xz -dc "${BASE}" > config + else + apt-get download "linux-image-${ABI}" + dpkg-deb -x linux-image-"${ABI}"_*.deb /tmp/pkgcfg + BASE="$(find /tmp/pkgcfg/boot -maxdepth 1 -name 'config-*' | head -n1)" + test -s "${BASE}" + cp "${BASE}" config + fi + test -s config + rm -rf /tmp/pkgcfg + + # ------------------------------------------------------------------ + # Layer the OGC config fragments and the repo-local config.fragment + # (which wins). kernel-packages ships no debian-specific fragments. + # Order is critical: unsets apply first, then sets, so explicit + # enables can override explicit disables. + # ------------------------------------------------------------------ + for f in ogc.config.set ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash ./merge-fragments.sh config \ + ogc.config.unset \ + ogc.config.set \ + config.fragment + echo "Config after fragment merge (head):" + head -n 3 config + + - name: Ensure pahole >= 1.31 (sched_ext/BTF) + run: | + set -euxo pipefail + VER="$(pahole --version | tr -d 'v')" + MAJ="${VER%%.*}"; MIN="${VER#*.}"; MIN="${MIN%%.*}" + echo "Installed pahole: ${VER}" + if [ "$MAJ" -lt 1 ] || { [ "$MAJ" -eq 1 ] && [ "$MIN" -lt 31 ]; }; then + echo "pahole < 1.31 breaks sched_ext; building dwarves 1.31 from source" + apt-get install -y --no-install-recommends cmake pkg-config + cd /tmp + curl -fsSLO https://fedorapeople.org/~acme/dwarves/dwarves-1.31.tar.xz + echo "0a7f255ccacf8cc7f8cd119099eb327179b4b3c67cb015af646af6d0cb03054d dwarves-1.31.tar.xz" | sha256sum -c + tar -xf dwarves-1.31.tar.xz + cmake -B dwarves-1.31/build -S dwarves-1.31 \ + -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr + make -C dwarves-1.31/build -j"$(nproc)" install + fi + pahole --version + + - name: Build .deb packages with the in-tree packaging + run: | + set -euxo pipefail + mkdir -p /build + tar -xzf /tmp/stage/linux.tar.gz -C /build + cd /build/linux + + # Config adjustments + kernel release suffix, identical scheme to + # the Arch and Fedora packages (uname -r == ${KREL}). + cp /tmp/stage/config .config + # Debian's config references distro-only certificate files that do + # not exist in this tree; use the ephemeral in-tree key instead. + scripts/config --set-str SYSTEM_TRUSTED_KEYS "" + scripts/config --set-str SYSTEM_REVOCATION_KEYS "" + # CONFIG_MODULE_SIG_KEY points at the Debian signing cert, which + # does not exist in this tree. Reset it to the kbuild default + # ("certs/signing_key.pem"): an ephemeral self-signed key generated + # during the build and trusted by the kernel itself. The Debian + # config sets CONFIG_MODULE_SIG_ALL=y, and with an empty key string + # scripts/Makefile.modinst resolves sig-key to "./", making + # sign-file read a directory as the private key (SSL DECODER error) + # on every module. + scripts/config --set-str MODULE_SIG_KEY "certs/signing_key.pem" + scripts/config -u DEFAULT_HOSTNAME + scripts/config --set-str BUILD_SALT "${KREL}" + echo "-unstable-ogc-g${SHA8}" > localversion.10-pkgname + echo "-1" > localversion.20-pkgrel + # The staged tarball carries no VCS data, but mkdebian (via + # gen-diff-patch) runs "git diff HEAD" and aborts on failure. + # A throwaway repo with everything committed makes that a no-op. + git init -q + git config user.email "ci@opengamingcollective.org" + git config user.name "OpenGamingCollective CI" + git add -A + git commit -qm "linux-unstable-ogc ${KREL}" + chown -R builder:builder /build + + # scripts/setlocalversion appends a "+" whenever a git repo exists + # and LOCALVERSION is unset and the HEAD is not at an annotated + # version tag. The throwaway repo below would therefore corrupt the + # release string. Setting LOCALVERSION to the empty string (as + # documented in the script) suppresses that suffix everywhere. + runuser -u builder -- env HOME=/home/builder CCACHE_DIR=/ccache \ + LOCALVERSION= \ + make CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 \ + KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc \ + olddefconfig + + # Fail fast if the release string ever drifts from the other distros + REL="$(runuser -u builder -- env HOME=/home/builder LOCALVERSION= make -s kernelrelease)" + if [ "$REL" != "${KREL}" ]; then + echo "kernelrelease '$REL' does not match expected '${KREL}'" >&2 + exit 1 + fi + + # Generate the debian/ directory with the tree's own packaging. + # mkdebian is invoked directly as a script: a bare "make debian" is + # NOT a top-level make target in this tree (only *-pkg patterns are + # delegated to scripts/Makefile.package), and going through make + # without the exact same CC flags as the olddefconfig above made + # kbuild re-sync the config interactively. As a plain sh script it + # touches nothing kbuild-related. + # KDEB_PKGVERSION must not contain hyphens except the final revision + # separator (Debian policy), hence the dotted translation. + KDEBVER="${KVERDOT}.unstable.ogc.g${SHA8}-1" + runuser -u builder -- env HOME=/home/builder \ + srctree="$PWD" \ + ARCH=x86_64 SRCARCH=x86 UTS_MACHINE=x86_64 \ + KERNELRELEASE="${KREL}" \ + KCONFIG_CONFIG=.config \ + KDEB_SOURCENAME=linux-unstable-ogc \ + KDEB_PKGVERSION="${KDEBVER}" \ + KDEB_CHANGELOG_DIST=trixie \ + DEBFULLNAME="OpenGamingCollective CI" \ + DEBEMAIL="ci@opengamingcollective.org" \ + sh scripts/package/mkdebian + echo "debian arch: $(cat debian/arch)" + + # Kbuild only honours command-line variables, so inject the + # clang/ccache toolchain into the generated debian/rules (this is + # what "make bindeb-pkg" would otherwise lose). LOCALVERSION= (set, + # empty) keeps setlocalversion from appending "+" inside the build. + sed -i 's|^make-opts = |make-opts = CC="ccache clang" LLVM=1 LLVM_IAS=1 WERROR=0 LOCALVERSION= KBUILD_BUILD_HOST=ogc-ci KBUILD_BUILD_USER=kernel-unstable-ogc |' debian/rules + grep -n '^make-opts' debian/rules + + # Same invocation as "make bindeb-pkg" (scripts/Makefile.package), + # but with parallel jobs, --no-check-builddeps (the build deps are + # preinstalled above; the generated Build-Depends-Arch also names a + # distro-only cross-gcc package that need not exist), and without + # the multi-GB debug-symbol package (all other distro jobs ship no + # debug packages either). + runuser -u builder -- env HOME=/home/builder CCACHE_DIR=/ccache \ + LOCALVERSION= \ + DEB_BUILD_PROFILES="pkg.linux-unstable-ogc.nokerneldbg" \ + dpkg-buildpackage --build=binary --no-pre-clean --unsigned-changes \ + --no-check-builddeps \ + -R'make -f debian/rules' -j"$(nproc)" -a"$(cat debian/arch)" + + ls -lh /build/*.deb + + - name: Collect Debian artifacts + run: | + set -euxo pipefail + DISTDIR=/tmp/dist + mkdir -p "${DISTDIR}" + cp -v "/build/linux-image-${KREL}"_*.deb "${DISTDIR}/" + cp -v "/build/linux-headers-${KREL}"_*.deb "${DISTDIR}/" + cp -v /build/linux/.config "${DISTDIR}/config-debian-${KREL}" + # linux-libc-dev is intentionally NOT shipped: installing it would + # replace the distribution's own linux-libc-dev package. + ls -lh "${DISTDIR}" + + - name: Boot smoke test in QEMU (release gate) + run: | + set -euxo pipefail + apt-get update -qq + # qemu boots the kernel (OVMF provides the UEFI leg); cpio builds + # the test initramfs. + apt-get install -y --no-install-recommends qemu-system-x86 cpio ovmf + # ---------------------------------------------------------------- + # Boot the exact kernel image that ships in the linux-image .deb. + # This step fails the job (and with it the release) if the kernel + # panics, hangs or never reaches userspace, so an unbootable + # kernel is never uploaded or published. + # ---------------------------------------------------------------- + WORK=/tmp/bootsmoke + rm -rf "$WORK"; mkdir -p "$WORK" + DEB="$(find /tmp/dist -name "linux-image-${KREL}_*.deb" | head -n1)" + test -n "$DEB" + dpkg-deb -x "$DEB" "$WORK" + IMG="$(find "$WORK" -name 'vmlinuz-*' -type f | head -n1)" + test -n "$IMG" + bash /tmp/stage/boot-smoke-test.sh "$IMG" + + - name: Clean build tree (free disk before cache save) + run: rm -rf /build ; df -h / + + - name: Upload Debian packages + uses: actions/upload-artifact@v7 + with: + name: debian-packages + path: /tmp/dist + retention-days: 3 diff --git a/.github/workflows/build-fedora-packages.yml b/.github/workflows/build-fedora-packages.yml new file mode 100644 index 00000000000000..0a95d181f8ab77 --- /dev/null +++ b/.github/workflows/build-fedora-packages.yml @@ -0,0 +1,207 @@ +# Fedora RPM packaging, split out of build-kernel.yml. +# Called (via `uses:`) by build-kernel.yml; runs inside the same workflow +# run, so the artifacts uploaded here are seen by publish-release.yml. + +name: Build Fedora RPM packages + +on: + workflow_call: + inputs: + krel: + description: Full kernel release string (uname -r), e.g. 6.12.0-next-20250101-unstable-ogc-g12345678-1 + type: string + required: true + kbase: + description: Base kernel version incl. prerelease suffix but no packaging suffix (e.g. 6.12.0-next-20250101) + type: string + required: true + kverdot: + description: Kernel version with hyphens replaced by dots, for the RPM Version field + type: string + required: true + sha8: + description: Short (8 char) source commit the packages are built from + type: string + required: true + +permissions: + contents: read + +env: + # Config fragments maintained by the OGC kernel-packages repository. + # The distro base config is extracted from the official Fedora + # `kernel-core` package — same approach as the kernel-packages + # fedora.yaml workflow. + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + build: + name: Build Fedora RPM packages + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/library/fedora:43 + env: + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + KREL: ${{ inputs.krel }} + KBASE: ${{ inputs.kbase }} + KVERDOT: ${{ inputs.kverdot }} + SHA8: ${{ inputs.sha8 }} + steps: + - name: Show disk space + run: df -h / + + - name: Install build tools + run: | + set -euxo pipefail + dnf -y install dnf5-plugins rpm-build + dnf -y install cpio curl findutils tar gzip + mkdir -p /ccache + + - name: Download staged sources + uses: actions/download-artifact@v8 + with: + name: kernel-sources + path: /tmp/stage + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-fedora-${{ github.sha }} + restore-keys: | + ccache-fedora- + + - name: Assemble Fedora kernel config + working-directory: /tmp/stage + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: extract the official Fedora kernel .config from the + # distro's `kernel-core` package (same method as the OGC + # kernel-packages fedora.yaml workflow). + # ------------------------------------------------------------------ + CACHE="/tmp/pkgcache" + mkdir -p "$CACHE" + dnf download --destdir "$CACHE" kernel-core + RPM="$(find "$CACHE" -maxdepth 1 -name 'kernel-core-*.rpm' -type f)" + if [ -z "$RPM" ]; then + echo "::error::kernel-core package not found in cache"; exit 1 + fi + CFG="$(rpm -qlp "$RPM" | grep -E '^/lib/modules/[^/]+/config$' | head -n1 || true)" + if [ -z "$CFG" ]; then + echo "::error::kernel config not found inside $RPM"; exit 1 + fi + rpm2cpio "$RPM" | cpio -i --to-stdout ".${CFG}" > config + test -s config + rm -rf "$CACHE" + + # ------------------------------------------------------------------ + # Layer the OGC config fragments on top, then the repo-local + # config.fragment (which wins). Order is critical: unsets apply + # first (removing things we explicitly don't want), then sets apply + # (enabling things we do want), so explicit enables can override + # explicit disables. + # ------------------------------------------------------------------ + for f in fedora.config.set ogc.config.set fedora.config.unset ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash ./merge-fragments.sh config \ + fedora.config.unset ogc.config.unset \ + fedora.config.set ogc.config.set \ + config.fragment + echo "Config after fragment merge (head):" + head -n 3 config + + - name: Finalize spec and install build dependencies + working-directory: /tmp/stage + run: | + set -euxo pipefail + # Substitute the CI placeholders (must happen before `dnf builddep` + # so RPM can parse Version:/Release:). + sed -i \ + -e "s/@@KBASEVER@@/${KBASE}/" \ + -e "s/@@KVERDOTTED@@/${KVERDOT}/" \ + -e "s/@@SHA8@@/${SHA8}/" \ + kernel.spec + grep -n '^Version:\|^Release:\|%define kbasever\|%define sha8' kernel.spec + ! grep -q '@@' kernel.spec + + # Installs the toolchain from the spec BuildRequires, including the + # distro dwarves/pahole. Run BEFORE the pahole check below so a + # source-built pahole is not overwritten by builddep. + dnf -y builddep kernel.spec + + - name: Ensure pahole >= 1.31 (sched_ext/BTF) + working-directory: /tmp + run: | + set -euxo pipefail + VER="$(pahole --version | tr -d 'v')" + MAJ="${VER%%.*}"; MIN="${VER#*.}"; MIN="${MIN%%.*}" + echo "Installed pahole: ${VER}" + if [ "$MAJ" -lt 1 ] || { [ "$MAJ" -eq 1 ] && [ "$MIN" -lt 31 ]; }; then + echo "pahole < 1.31 breaks sched_ext; building dwarves 1.31 from source" + dnf -y install cmake make gcc elfutils-devel zlib-devel + curl -fsSLO https://fedorapeople.org/~acme/dwarves/dwarves-1.31.tar.xz + echo "0a7f255ccacf8cc7f8cd119099eb327179b4b3c67cb015af646af6d0cb03054d dwarves-1.31.tar.xz" | sha256sum -c + tar -xf dwarves-1.31.tar.xz + cmake -B dwarves-1.31/build -S dwarves-1.31 \ + -DCMAKE_BUILD_TYPE=Release -DCMAKE_INSTALL_PREFIX=/usr -D__LIB=lib + make -C dwarves-1.31/build -j"$(nproc)" install + fi + pahole --version + + - name: Build RPMs with rpmbuild + working-directory: /tmp + run: | + set -euxo pipefail + TOPDIR=/tmp/rpmbuild + mkdir -p "${TOPDIR}"/{BUILD,BUILDROOT,RPMS,SOURCES,SPECS,SRPMS} + cp /tmp/stage/linux.tar.gz "${TOPDIR}/SOURCES/" + cp /tmp/stage/config "${TOPDIR}/SOURCES/config" + cp /tmp/stage/kernel.spec "${TOPDIR}/SPECS/kernel.spec" + + rpmbuild --define "_topdir ${TOPDIR}" -ba "${TOPDIR}/SPECS/kernel.spec" + + ls -lh "${TOPDIR}"/RPMS/x86_64/ + + - name: Collect Fedora artifacts + run: | + set -euxo pipefail + DISTDIR=/tmp/dist + mkdir -p "${DISTDIR}" + cp -v /tmp/rpmbuild/RPMS/x86_64/*.rpm "${DISTDIR}/" + cp -v /tmp/stage/config "${DISTDIR}/config-fedora-${KREL}" + ls -lh "${DISTDIR}" + + - name: Boot smoke test in QEMU (release gate) + run: | + set -euxo pipefail + dnf -y install qemu-system-x86 edk2-ovmf + # ---------------------------------------------------------------- + # Boot the exact kernel image that ships in the kernel-core RPM. + # This step fails the job (and with it the release) if the kernel + # panics, hangs or never reaches userspace, so an unbootable + # kernel is never uploaded or published. + # ---------------------------------------------------------------- + WORK=/tmp/bootsmoke + rm -rf "$WORK"; mkdir -p "$WORK" + RPM="$(find /tmp/dist -name 'kernel-unstable-ogc-core-*.rpm' | head -n1)" + test -n "$RPM" + # rpm cpio entries are stored with a "./" prefix (same trick as + # the config extraction above). + rpm2cpio "$RPM" | cpio -idm --quiet -D "$WORK" "./lib/modules/${KREL}/vmlinuz" || true + IMG="$WORK/lib/modules/${KREL}/vmlinuz" + test -s "$IMG" || { echo "::error::vmlinuz not found inside ${RPM}"; exit 1; } + bash /tmp/stage/boot-smoke-test.sh "$IMG" + + - name: Clean build tree (free disk before cache save) + run: rm -rf /tmp/rpmbuild/BUILD /tmp/rpmbuild/BUILDROOT ; df -h / + + - name: Upload Fedora packages + uses: actions/upload-artifact@v7 + with: + name: fedora-packages + path: /tmp/dist + retention-days: 3 diff --git a/.github/workflows/build-kernel.yml b/.github/workflows/build-kernel.yml new file mode 100644 index 00000000000000..8ca98fa697d2d6 --- /dev/null +++ b/.github/workflows/build-kernel.yml @@ -0,0 +1,129 @@ +# Entry point of the linux-unstable-ogc build pipeline. The heavy lifting +# lives in reusable workflows next to this file: +# +# build-arch-packages.yml - Arch Linux .pkg.tar.zst builds +# build-fedora-packages.yml - Fedora RPM builds +# build-debian-packages.yml - Debian .deb builds +# publish-release.yml - collects artifacts and cuts the GitHub release +# +# They are called with `uses:` so everything stays a single workflow run: +# artifacts uploaded by the build jobs are downloaded by publish-release.yml, +# and the needs: chain below keeps the release gated on the QEMU boot smoke +# tests that run inside each build workflow. + +name: Build & release linux-unstable-ogc + +on: + push: + branches: [master] + workflow_dispatch: {} + +permissions: + contents: write + +concurrency: + group: kernel-build-${{ github.ref }} + cancel-in-progress: true + +jobs: + prepare: + name: Prepare sources and version + runs-on: ubuntu-latest + outputs: + kver: ${{ steps.kernel.outputs.kver }} + next: ${{ steps.kernel.outputs.next }} + sha8: ${{ steps.kernel.outputs.sha8 }} + krel: ${{ steps.kernel.outputs.krel }} + tag: ${{ steps.kernel.outputs.tag }} + kbase: ${{ steps.kernel.outputs.kbase }} + kverdot: ${{ steps.kernel.outputs.kverdot }} + steps: + - name: Checkout kernel sources + uses: actions/checkout@v7 + with: + fetch-depth: 1 + + - name: Compute kernel version + id: kernel + run: | + set -euxo pipefail + KVER="$(make -s kernelversion)" + NEXT="$(cat localversion-next 2>/dev/null || true)" + SHA8="$(git rev-parse --short=8 HEAD)" + KREL="${KVER}${NEXT}-unstable-ogc-g${SHA8}-1" + TAG="v${KVER}${NEXT}-g${SHA8}" + # RPM Version: field must not contain hyphens + KVERDOT="$(printf '%s%s' "${KVER}" "${NEXT}" | tr '-' '.')" + echo "kver=${KVER}" >> "$GITHUB_OUTPUT" + echo "next=${NEXT}" >> "$GITHUB_OUTPUT" + echo "sha8=${SHA8}" >> "$GITHUB_OUTPUT" + echo "krel=${KREL}" >> "$GITHUB_OUTPUT" + echo "tag=${TAG}" >> "$GITHUB_OUTPUT" + echo "kbase=${KVER}${NEXT}" >> "$GITHUB_OUTPUT" + echo "kverdot=${KVERDOT}" >> "$GITHUB_OUTPUT" + echo "Kernel release: ${KREL}" + echo "Release tag: ${TAG}" + + - name: Stage sources and packaging files + run: | + set -euxo pipefail + # Commit marker consumed by the Arch PKGBUILD (same trick as + # linux-bisector). Written before the tarball is created so it is + # included in it. + git rev-parse --short=8 HEAD > .build_commit + + STAGE="/tmp/stage" + mkdir -p "${STAGE}" + cp .github/packaging/PKGBUILD "${STAGE}/PKGBUILD" + cp .github/packaging/merge-fragments.sh "${STAGE}/merge-fragments.sh" + cp .github/packaging/config.fragment "${STAGE}/config.fragment" + cp .github/packaging/fedora/kernel.spec "${STAGE}/kernel.spec" + cp .github/packaging/boot-smoke-test.sh "${STAGE}/boot-smoke-test.sh" + + # Source tarball: contents of the repo as ./linux/, without VCS data. + tar --exclude-vcs -I 'gzip -1' -cf "${STAGE}/linux.tar.gz" \ + --transform 's|^\./|linux/|' -C "$GITHUB_WORKSPACE" . + ls -lh "${STAGE}" + + - name: Upload staged sources + uses: actions/upload-artifact@v7 + with: + name: kernel-sources + path: /tmp/stage + retention-days: 3 + + arch: + name: Arch packages + needs: prepare + uses: ./.github/workflows/build-arch-packages.yml + with: + krel: ${{ needs.prepare.outputs.krel }} + + fedora: + name: Fedora packages + needs: prepare + uses: ./.github/workflows/build-fedora-packages.yml + with: + krel: ${{ needs.prepare.outputs.krel }} + kbase: ${{ needs.prepare.outputs.kbase }} + kverdot: ${{ needs.prepare.outputs.kverdot }} + sha8: ${{ needs.prepare.outputs.sha8 }} + + debian: + name: Debian packages + needs: prepare + uses: ./.github/workflows/build-debian-packages.yml + with: + krel: ${{ needs.prepare.outputs.krel }} + kverdot: ${{ needs.prepare.outputs.kverdot }} + sha8: ${{ needs.prepare.outputs.sha8 }} + + release: + name: Publish GitHub release + needs: [prepare, arch, fedora, debian] + uses: ./.github/workflows/publish-release.yml + with: + tag: ${{ needs.prepare.outputs.tag }} + krel: ${{ needs.prepare.outputs.krel }} + kver: ${{ needs.prepare.outputs.kver }} + next: ${{ needs.prepare.outputs.next }} diff --git a/.github/workflows/publish-release.yml b/.github/workflows/publish-release.yml new file mode 100644 index 00000000000000..e9f641a6b36644 --- /dev/null +++ b/.github/workflows/publish-release.yml @@ -0,0 +1,157 @@ +# Release publisher, split out of build-kernel.yml. +# Called (via `uses:`) by build-kernel.yml after all three distro builds +# (each gated on its QEMU boot smoke test) have succeeded. + +name: Publish linux-unstable-ogc release + +on: + workflow_call: + inputs: + tag: + description: Release tag, e.g. v6.12.0-next-20250101-g12345678 + type: string + required: true + krel: + description: Full kernel release string (uname -r) + type: string + required: true + kver: + description: Base kernel version from `make kernelversion` + type: string + required: true + next: + description: localversion-next suffix (may be empty) + type: string + required: true + +permissions: + contents: write + +env: + # Config fragments maintained by the OGC kernel-packages repository + # (referenced in the release notes). + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + release: + name: Create GitHub release + runs-on: ubuntu-latest + env: + TAG: ${{ inputs.tag }} + KREL: ${{ inputs.krel }} + KVER: ${{ inputs.kver }} + NEXT: ${{ inputs.next }} + steps: + - name: Download package artifacts + uses: actions/download-artifact@v8 + with: + path: /tmp/dist + pattern: "*-packages" + + - name: Flatten artifact directory + run: | + set -euxo pipefail + cd /tmp/dist + find . -mindepth 2 -maxdepth 2 -type f -exec mv -t . {} + + find . -mindepth 1 -type d -delete + ls -lh + + - name: Generate checksums + run: | + set -euxo pipefail + cd /tmp/dist + sha256sum * > SHA256SUMS + cat SHA256SUMS + + - name: Create GitHub release and upload packages + env: + GH_TOKEN: ${{ github.token }} + run: | + set -euxo pipefail + + # Idempotency: if a stale release exists for this tag (e.g. re-run + # of the same commit), remove it so this run can recreate it. + if gh release view "$TAG" --repo "$GITHUB_REPOSITORY" 2>/dev/null; then + echo "Deleting existing release $TAG, will recreate it" + gh release delete "$TAG" --repo "$GITHUB_REPOSITORY" --yes --cleanup-tag + fi + # In case a tag without a release is left over, drop it too. + gh api -X DELETE "repos/${GITHUB_REPOSITORY}/git/refs/tags/${TAG}" >/dev/null 2>&1 || true + + NOTES="$(mktemp)" + { + echo "Automated build of linux-unstable-ogc." + echo + echo "- Source commit: https://github.com/${GITHUB_REPOSITORY}/commit/${GITHUB_SHA}" + echo "- Kernel release: ${KREL}" + echo "- Upstream version: ${KVER}${NEXT}" + echo "- Base configs: Arch \`linux-headers\` + Fedora \`kernel-core\`, plus [OGC kernel-packages fragments](${KERNEL_PACKAGES_RAW}/config)" + echo "- Compiler: clang / LLVM=1 (with ccache)" + echo "- Boot smoke test: every packaged kernel (Arch, Fedora, Debian) was booted in QEMU and reached userspace init before publishing" + echo + echo "### Artifacts" + echo + echo "#### Arch Linux" + echo + echo '- `linux-unstable-ogc` — kernel image and modules' + echo '- `linux-unstable-ogc-headers` — headers for building external modules' + echo "- \`config-arch-${KREL}\` — the exact .config used for this build" + echo + echo "#### Fedora" + echo + echo '- `kernel-unstable-ogc-core` — kernel image (vmlinuz) and core files' + echo '- `kernel-unstable-ogc-modules` — kernel modules' + echo '- `kernel-unstable-ogc-devel` — headers for building external modules' + echo "- \`config-fedora-${KREL}\` — the exact .config used for this build" + echo + echo "#### Debian (and derivatives)" + echo + echo "- \`linux-image-${KREL}\` — kernel image and modules" + echo "- \`linux-headers-${KREL}\` — headers for building external modules" + echo "- \`config-debian-${KREL}\` — the exact .config used for this build" + echo + echo '- `SHA256SUMS` — checksums of all artifacts' + echo + echo "### Install" + echo + echo "Arch Linux:" + echo + echo '```sh' + echo 'sudo pacman -U linux-unstable-ogc-headers-*.pkg.tar.zst linux-unstable-ogc-*.pkg.tar.zst' + echo '```' + echo + echo "Fedora:" + echo + echo '```sh' + echo 'sudo dnf install ./kernel-unstable-ogc-core-*.rpm ./kernel-unstable-ogc-modules-*.rpm' + echo '```' + echo + echo "Debian and derivatives:" + echo + echo '```sh' + echo 'sudo apt install ./linux-image-*.deb ./linux-headers-*.deb' + echo '```' + echo + echo "> The initramfs is generated automatically on install (mkinitcpio hooks" + echo "> on Arch, kernel-install/dracut on Fedora, initramfs-tools hooks on Debian)." + } > "$NOTES" + + cd /tmp/dist + # Upload every artifact produced by the distro jobs (package files + # plus the config-* files). SHA256SUMS itself is included by ./*. + gh release create "$TAG" \ + --repo "$GITHUB_REPOSITORY" \ + --target "$GITHUB_SHA" \ + --title "linux-unstable-ogc ${KREL}" \ + --notes-file "$NOTES" \ + ./* + + - name: Summary + run: | + { + echo "## linux-unstable-ogc build" + echo + echo "- Kernel release: \`${KREL}\`" + echo "- Release: https://github.com/${GITHUB_REPOSITORY}/releases/tag/${TAG}" + echo "- QEMU boot smoke test: passed (Arch, Fedora and Debian kernel images booted to userspace init)" + } >> "$GITHUB_STEP_SUMMARY" diff --git a/.github/workflows/sync-linux-next.yml b/.github/workflows/sync-linux-next.yml new file mode 100644 index 00000000000000..45b236c3f24cb6 --- /dev/null +++ b/.github/workflows/sync-linux-next.yml @@ -0,0 +1,115 @@ +name: Sync linux-next -> master (replay fork commits) + +on: + schedule: + - cron: "0 2 * * *" # every day at 02:00 UTC + workflow_dispatch: {} + +permissions: + contents: write + +concurrency: + group: sync-linux-next + cancel-in-progress: false + +jobs: + sync: + runs-on: ubuntu-latest + steps: + - name: Checkout fork + uses: actions/checkout@v7 + with: + ref: master + fetch-depth: 0 + + - name: Configure upstream + run: | + git remote remove upstream || true + git remote add upstream https://git.kernel.org/pub/scm/linux/kernel/git/next/linux-next.git + git fetch --no-tags upstream --prune + + - name: Replay fork commits on top of upstream/master + run: | + set -euo pipefail + + git checkout master + + # REQUIRED for cherry-pick commit creation (runner has no identity by default) + git config user.name "github-actions[bot]" + git config user.email "github-actions[bot]@users.noreply.github.com" + + UP_BASE="upstream/master" + git show -s --oneline "$UP_BASE" >/dev/null + + # ------------------------------------------------------------------ + # Find the upstream snapshot that master's fork commits sit on. + # + # Anchor: the empty marker commit "ogc: linux-unstable: first + # commit" is always the first fork commit ever made. Its parent is + # therefore the upstream snapshot the fork was built on. + # ------------------------------------------------------------------ + MARKER="$(git log --format="%H" --grep="^ogc: linux-unstable: first commit$" master)" + if [ -z "$MARKER" ]; then + echo "::error::Could not find the 'ogc: linux-unstable: first commit' marker; refusing to touch master." + exit 1 + fi + BASE="$(git rev-parse "${MARKER}^")" + echo "Upstream base: $(git show -s --oneline "$BASE")" + + # If master already sits on the current upstream tip there is + # nothing to sync: replaying would only churn the fork commits' + # hashes for identical content. + if [ "$(git rev-parse "$BASE")" = "$(git rev-parse "$UP_BASE")" ]; then + echo "master is already based on the current linux-next tip; nothing to do." + exit 0 + fi + + # Everything on master on top of the upstream base = the fork + # commits (including non-"ogc:"-prefixed ones, e.g. squash-merged + # PRs). --no-merges matches the cherry-pick loop below and makes + # merge commits replay as their constituent commits. + MY_COMMITS="$(git rev-list --reverse --no-merges "${BASE}..master")" + if [ -z "${MY_COMMITS// }" ]; then + echo "::error::No commits found on top of the upstream base; refusing to reset master (that would delete the fork)." + exit 1 + fi + + # Safety net: a mis-detected base would replay days of upstream + # history. The fork accumulates its own commits over time (and PR + # squash-merges add to that), so allow a generous 150; anything + # beyond that is far more likely a mis-detected base than real fork + # commits, so bail out and ask for a manual look instead of + # corrupting master. + N="$(echo "$MY_COMMITS" | wc -l)" + if [ "$N" -gt 150 ]; then + echo "::error::Refusing to replay ${N} commits (expected only fork commits). Check master's history (all fork commits must carry the 'ogc:' subject prefix)." + exit 1 + fi + + echo "Replaying ${N} commit(s):" + for c in $MY_COMMITS; do + echo " $c $(git show -s --format=%s "$c")" + done + echo + + # Reset master to the new upstream tip, then replay the fork commits. + git reset --hard "$UP_BASE" + + for c in $MY_COMMITS; do + echo "Cherry-picking: $c $(git show -s --format=%s "$c")" + + if ! git cherry-pick -x --allow-empty "$c"; then + git status || true + # If we're left mid-cherry-pick, abort to avoid a broken + # working state. Nothing was pushed, so master on origin is + # still intact. + if [ -f .git/CHERRY_PICK_HEAD ]; then + git cherry-pick --abort || true + fi + echo "::error::Cherry-pick failed for $c; nothing was pushed." + exit 1 + fi + done + + echo "Done. Pushing updated master." + git push --force-with-lease origin master diff --git a/.github/workflows/test-pr.yml b/.github/workflows/test-pr.yml new file mode 100644 index 00000000000000..29c4e2fb8f2dbe --- /dev/null +++ b/.github/workflows/test-pr.yml @@ -0,0 +1,227 @@ +name: Test PR (checkpatch + config gate + gcc build) + +on: + pull_request: + branches: [master] + +permissions: + contents: read + +concurrency: + group: pr-test-${{ github.event.pull_request.number }} + cancel-in-progress: true + +env: + KERNEL_PACKAGES_RAW: https://raw.githubusercontent.com/OpenGamingCollective/kernel-packages/main + +jobs: + checks: + name: Static checks (checkpatch, new-driver symbols) + runs-on: ubuntu-latest + outputs: + new_symbols: ${{ steps.syms.outputs.syms }} + steps: + - name: Checkout PR merge result + uses: actions/checkout@v7 + with: + # Default ref for pull_request is the merge commit refs/pull/N/merge. + # depth 2 fetches the merge commit AND its parents, so HEAD^1 + # (current master tip) is present and the PR diff is exactly + # "what landing this PR changes on master". + fetch-depth: 2 + + - name: Compute PR patch and new driver symbols + id: syms + run: | + set -euo pipefail + # For pull_request, checkout gets the merge commit refs/pull/N/merge: + # parent 1 = master tip, parent 2 = PR head. Fall back to + # origin/master if HEAD is not a merge commit for any reason. + BASE="$(git rev-parse --verify -q HEAD^1 || true)" + if [ -z "$BASE" ]; then + echo "HEAD^1 unresolvable; falling back to origin/master" + git fetch --depth=1 origin master + BASE="$(git rev-parse --verify -q FETCH_HEAD || true)" + fi + if [ -z "$BASE" ]; then + echo "::error::Could not determine the base commit for the PR diff." + exit 1 + fi + git diff "$BASE" HEAD > /tmp/pr.patch + + if [ ! -s /tmp/pr.patch ]; then + echo "Empty patch; nothing to check." + echo "syms=" >> "$GITHUB_OUTPUT" + exit 0 + fi + + # Selectable symbols introduced under drivers/ (added + # "config FOO" / "menuconfig FOO" lines in any Kconfig file). + # NOTE: grep exits 1 on zero matches; under `set -euo pipefail` + # that would abort the step, so it is neutralised here only + # (git diff / awk failures still propagate). + SYMS="$(git diff "$BASE" HEAD -- drivers/ \ + | { grep -E '^\+[[:space:]]*(menu)?config[[:space:]]+[A-Z0-9_]+' || true; } \ + | awk '{print $NF}' | sort -u | tr '\n' ' ')" + echo "syms=${SYMS}" >> "$GITHUB_OUTPUT" + { + echo "### PR checks" + echo + echo "- Changed files: $(git diff --name-only "$BASE" HEAD | wc -l)" + echo "- New driver CONFIG symbols: ${SYMS:-none}" + } >> "$GITHUB_STEP_SUMMARY" + + - name: checkpatch (fail on errors) + run: | + set -uo pipefail + if [ ! -s /tmp/pr.patch ]; then + echo "Empty patch; skipping checkpatch." + exit 0 + fi + # --no-signoff: internal fork PRs do not require Signed-off-by. + perl scripts/checkpatch.pl --no-signoff /tmp/pr.patch \ + > /tmp/checkpatch.out 2>&1 || true + cat /tmp/checkpatch.out + + if grep -q '^ERROR:' /tmp/checkpatch.out; then + NERRS="$(grep -c '^ERROR:' /tmp/checkpatch.out)" + echo "::error::checkpatch reported ${NERRS} error(s); fix them before merge (warnings do not block)." + exit 1 + fi + echo "checkpatch: no errors." + + build: + name: Build with GCC (Arch config + fragments) + needs: checks + runs-on: ubuntu-latest + timeout-minutes: 330 + container: + image: docker.io/archlinux:base-devel + env: + CCACHE_DIR: /ccache + CCACHE_MAXSIZE: 10G + NEW_SYMBOLS: ${{ needs.checks.outputs.new_symbols }} + steps: + - name: Show disk space + run: df -h / + + - name: Bootstrap build environment + run: | + set -euxo pipefail + pacman -Sy --needed --noconfirm archlinux-keyring + pacman -Su --needed --noconfirm + # base-devel provides gcc/make/perl; the rest mirrors the makedepends + # of the release PKGBUILD minus clang and the rust toolchain. + pacman -S --needed --noconfirm \ + bc cpio gettext libelf pahole perl python tar xz zstd \ + file curl git ccache + + - name: Checkout PR merge result + uses: actions/checkout@v7 + with: + fetch-depth: 1 + + - name: Restore compiler cache + uses: actions/cache@v6 + with: + path: /ccache + key: ccache-pr-gcc-${{ github.sha }} + restore-keys: | + ccache-pr-gcc- + + - name: Assemble test config (same as the Arch release build) + run: | + set -euxo pipefail + # ------------------------------------------------------------------ + # Base config: official Arch Linux kernel .config extracted from the + # distro's linux-headers package (same method as the release build). + # ------------------------------------------------------------------ + CACHE="/tmp/pkgcache" + mkdir -p "$CACHE" + pacman -Sw --noconfirm --cachedir "$CACHE" linux-headers + PKG="$(find "$CACHE" -maxdepth 1 -name 'linux-headers-*.pkg.tar.zst' -type f)" + if [ -z "$PKG" ]; then + echo "::error::linux-headers package not found in cache"; exit 1 + fi + CFG="$(tar --zstd -tf "$PKG" | grep -E '^usr/lib/modules/[^/]+/build/\.config$' || true)" + if [ -z "$CFG" ]; then + echo "::error::kernel .config not found inside $PKG"; exit 1 + fi + tar --zstd -xOf "$PKG" "$CFG" > .config + test -s .config + rm -rf "$CACHE" + + # ------------------------------------------------------------------ + # OGC kernel-packages fragments + the repo-local config.fragment + # (merged last, wins). Same order as the release build. + # ------------------------------------------------------------------ + for f in arch.config.set ogc.config.set arch.config.unset ogc.config.unset; do + curl -fsSL "${KERNEL_PACKAGES_RAW}/config/${f}" -o "${f}" + done + bash .github/packaging/merge-fragments.sh .config \ + arch.config.set ogc.config.set \ + arch.config.unset ogc.config.unset \ + .github/packaging/config.fragment + + # Same fixups as the release packaging... + scripts/config --set-str SYSTEM_TRUSTED_KEYS "" + scripts/config --set-str SYSTEM_REVOCATION_KEYS "" + scripts/config --set-str MODULE_SIG_KEY "certs/signing_key.pem" + scripts/config --set-str CONFIG_LOCALVERSION "" || true + # ...plus test-only tweaks: + # No rust toolchain is installed in this job; the release builds + # (clang) compile the rust bits, this job focuses on the C parts. + scripts/config -d RUST + make olddefconfig + + - name: "Gate: new driver symbols must be enabled in the config" + run: | + set -euo pipefail + if [ -z "${NEW_SYMBOLS// }" ]; then + echo "No new driver CONFIG symbols in this PR; skipping gate." + exit 0 + fi + echo "Gate symbols: ${NEW_SYMBOLS}" + + FAIL=0 + for s in ${NEW_SYMBOLS}; do + if grep -qE "^CONFIG_${s}=[ym]$" .config; then + echo "OK: CONFIG_${s} is enabled." + elif grep -q "^# CONFIG_${s} is not set$" .config; then + echo "::error::CONFIG_${s} was added by this PR but is disabled in the merged config." + echo "::error::Enable it in .github/packaging/config.fragment (CONFIG_${s}=y or =m) if it should ship." + FAIL=1 + elif grep -q "^CONFIG_${s}=" .config; then + echo "::error::CONFIG_${s} is set but not to y/m: $(grep "^CONFIG_${s}=" .config)" + FAIL=1 + else + echo "::error::CONFIG_${s} is absent from the final .config (its dependencies or its vendor menu are off in the merged config)." + echo "::error::If this driver should ship, enable it (and its dependencies) in .github/packaging/config.fragment." + FAIL=1 + fi + done + if [ "$FAIL" -ne 0 ]; then + echo "::error::config gate failed: new drivers must be enabled in the OGC config." + exit 1 + fi + echo "config gate passed." + + - name: Build kernel with GCC + run: | + set -euxo pipefail + ccache -s || true + make CC="ccache gcc" WERROR=0 -j"$(nproc)" all + echo "Kernel release: $(make -s kernelrelease)" + ccache -s + df -h / + + - name: Summary + if: always() + run: | + { + echo "### GCC compile test" + echo + echo "- Compiler: gcc (Arch Linux) via ccache" + echo "- Config: Arch linux-headers base + OGC fragments + repo config.fragment (Rust disabled for this test)" + echo "- New driver symbols checked: ${NEW_SYMBOLS:-none}" + } >> "$GITHUB_STEP_SUMMARY" \ No newline at end of file From f31c35d096451bf274236ae09b305395be16b881 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Tomasz=20Paku=C5=82a?= Date: Tue, 3 Feb 2026 18:56:14 +0000 Subject: [PATCH 1006/1012] drm/amd/display: Add CH7218 PCON ID MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit [Why] Chrontel CH7218 found in Ugreen DP -> HDMI 2.1 adapter (model 85564) works perfectly with VRR after testing. VRR and FreeSync compatibility is explicitly advertised as a feature so it's addition is a formality. Support FreeSync info packet passthrough and "generic" HDMI VRR. [How] Add CH7218's ID to dm_helpers_is_vrr_pcon_allowed() Closes: https://gitlab.freedesktop.org/drm/amd/-/issues/4773 Signed-off-by: Tomasz Pakuła (cherry picked from commit 7b2436287ed953496ad1c9eb5820f00b75db597f) (cherry picked from commit a9c75486e9c655c604e05d55e1546f4e44b1bfd2) (cherry picked from commit 63b562c3902fe1324bc0e81d3ccf78f95a811e1a) (cherry picked from commit e20d4609955d599e1cb81a5277a4dbb308c649e1) (cherry picked from commit 39c6e1f4320b1b3416c8c64a7ef9de6d733289ab) --- drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c | 1 + drivers/gpu/drm/amd/display/include/ddc_service_types.h | 1 + 2 files changed, 2 insertions(+) diff --git a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c index 3217aa82bea248..ec312eba80caba 100644 --- a/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c +++ b/drivers/gpu/drm/amd/display/amdgpu_dm/amdgpu_dm_helpers.c @@ -1681,6 +1681,7 @@ STATIC_IFN_KUNIT const uint32_t dm_freesync_pcon_whitelist[] = { DP_BRANCH_DEVICE_ID_90CC24, DP_BRANCH_DEVICE_ID_001CF8, DP_BRANCH_DEVICE_ID_001FF2, + DP_BRANCH_DEVICE_ID_2B02F0, }; EXPORT_IF_KUNIT(dm_freesync_pcon_whitelist); diff --git a/drivers/gpu/drm/amd/display/include/ddc_service_types.h b/drivers/gpu/drm/amd/display/include/ddc_service_types.h index 827e9bd7c5cff3..d2a8e712d4a712 100644 --- a/drivers/gpu/drm/amd/display/include/ddc_service_types.h +++ b/drivers/gpu/drm/amd/display/include/ddc_service_types.h @@ -37,6 +37,7 @@ #define DP_BRANCH_DEVICE_ID_001CF8 0x001CF8 #define DP_BRANCH_DEVICE_ID_0060AD 0x0060AD #define DP_BRANCH_DEVICE_ID_001FF2 0x001FF2 +#define DP_BRANCH_DEVICE_ID_2B02F0 0x2B02F0 /* Chrontel CH7218 */ #define DP_BRANCH_HW_REV_10 0x10 #define DP_BRANCH_HW_REV_20 0x20 From 1d4e5afac4427067d36bf2b0bfa9108ffa99da09 Mon Sep 17 00:00:00 2001 From: Ahmed Yaseen Date: Tue, 18 Aug 2026 06:53:12 +0500 Subject: [PATCH 1007/1012] [FOR-UPSTREAM] HID: asus: add ROG Zephyrus Duo GX651AR keyboard The detachable keyboard shipped with the ROG Zephyrus Duo GX651AR (0b05:1ce6) is a ROG N-Key keyboard, but it is not listed in asus_devices[], so its interfaces are left to hid-generic and its vendor usages are never mapped by asus_input_mapping(). Add it with QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD, matching the other ROG N-Key keyboards. Tested-by: Cymirk Signed-off-by: Ahmed Yaseen (cherry picked from commit 8d70b5f90b41b98f9fa6293e03be78c31bac8c36) (cherry picked from commit 9582bd441f1fefbe164037ec5b587b9db39d1167) (cherry picked from commit 29ee8b6c10c5d31601444dd69bf04a10c66c8288) (cherry picked from commit 9bd6541370e7ca2553bdd8d4f8465b80a0a325f7) --- drivers/hid/hid-asus.c | 3 +++ drivers/hid/hid-ids.h | 1 + 2 files changed, 4 insertions(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index 7dc6417fe2622a..a3584d819c696f 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1695,6 +1695,9 @@ static const struct hid_device_id asus_devices[] = { { HID_I2C_DEVICE(USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD2), QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD | QUIRK_HID_FN_LOCK }, + { HID_USB_DEVICE(USB_VENDOR_ID_ASUSTEK, + USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3), + QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, { HID_USB_DEVICE(USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_Z13_LIGHTBAR), QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, diff --git a/drivers/hid/hid-ids.h b/drivers/hid/hid-ids.h index a9537b7bb03d12..53502440403437 100644 --- a/drivers/hid/hid-ids.h +++ b/drivers/hid/hid-ids.h @@ -227,6 +227,7 @@ #define USB_DEVICE_ID_ASUSTEK_ROG_KEYBOARD3 0x1822 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD 0x1866 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD2 0x19b6 +#define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3 0x1ce6 #define USB_DEVICE_ID_ASUSTEK_ROG_Z13_FOLIO 0x1a30 #define USB_DEVICE_ID_ASUSTEK_ROG_Z13_LIGHTBAR 0x18c6 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_ALLY 0x1abe From 26d962c1a76bba672d49d56fae0fd6b54ca9b86a Mon Sep 17 00:00:00 2001 From: Ahmed Yaseen Date: Tue, 18 Aug 2026 09:46:22 +0500 Subject: [PATCH 1008/1012] [FOR-UPSTREAM] HID: asus: force input connection on vendor-only N-Key interfaces On the ROG Zephyrus Duo GX651AR (0b05:1ce6) the hotkeys live on report 0x5a on an interface whose descriptor holds nothing but two ASUS vendor collections. Neither satisfies IS_INPUT_APPLICATION(), so hidinput_connect() creates no input device, asus_input_mapping() never runs and every hotkey is dropped by asus_event() as unmapped. Set HID_QUIRK_HIDINPUT_FORCE on ROG N-Key interfaces that carry an ASUS vendor input report so those usages get mapped. Interfaces left with no mapped usage are still discarded by hidinput_has_been_populated(). The vendor check reads report_enum[HID_INPUT_REPORT], so interfaces with no input reports, such as the RGB control interface, are unaffected. Tested-by: Cymirk Signed-off-by: Ahmed Yaseen (cherry picked from commit 69887e3b53350a792c34272d0b101a21131a7ed7) (cherry picked from commit 4e707839a861d371b367c285665fd77cbfd1a113) (cherry picked from commit b3ceeae53756d8b73e99bf00a412136cdd130853) (cherry picked from commit c492209aed4a61e14efa274aa3cf153d55895d2e) --- drivers/hid/hid-asus.c | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index a3584d819c696f..502cd54fad793a 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1480,6 +1480,14 @@ static int asus_probe(struct hid_device *hdev, const struct hid_device_id *id) is_vendor = true; } + /* + * A vendor collection may be the only application collection on the + * interface, which hidinput_connect() otherwise skips, leaving the + * hotkey usages unmapped. Unpopulated inputs are dropped later. + */ + if (is_vendor && (drvdata->quirks & QUIRK_ROG_NKEY_KEYBOARD)) + hdev->quirks |= HID_QUIRK_HIDINPUT_FORCE; + ret = asus_worker_create(hdev, drvdata); if (ret) { hid_warn(hdev, "Failed to initialize worker: %d\n", ret); From d8a5f656c56d014e1e91a3737f310fd06e4e82bb Mon Sep 17 00:00:00 2001 From: Ahmed Yaseen Date: Sat, 22 Aug 2026 16:28:13 +0500 Subject: [PATCH 1009/1012] [FOR-UPSTREAM] HID: asus: add ROG Zephyrus Duo GX651AR keyboard over Bluetooth The GX651AR keyboard enumerates as 0b05:1ce6 over USB but pairs as 0b05:1ce7 in Bluetooth mode, where the keyboard, consumer and both ASUS vendor collections (reports 0x5a and 0x5d) sit on a single HID device. Add it with the same quirks as the USB entry. Bind to HID_GROUP_GENERIC so that hid-multitouch keeps the digitizer. Tested-by: Cymirk Signed-off-by: Ahmed Yaseen (cherry picked from commit b0dbc09e46a5d146c25081943f16fc0b1d0c0492) (cherry picked from commit f460429e28dc70ff7572c384a7a40ee66868e4c9) (cherry picked from commit af0d8926a206e3c4a6fa7544ba6a5aa0d63da03a) (cherry picked from commit 98993b7ccce3d580574b0bc6c5c04e9b7c266967) --- drivers/hid/hid-asus.c | 3 +++ drivers/hid/hid-ids.h | 1 + 2 files changed, 4 insertions(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index 502cd54fad793a..c63e03a876a32f 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1744,6 +1744,9 @@ static const struct hid_device_id asus_devices[] = { { HID_DEVICE(BUS_USB, HID_GROUP_GENERIC, USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_Z13_FOLIO), QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, + { HID_DEVICE(BUS_BLUETOOTH, HID_GROUP_GENERIC, + USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3_BT), + QUIRK_USE_KBD_BACKLIGHT | QUIRK_ROG_NKEY_KEYBOARD }, { HID_DEVICE(BUS_USB, HID_GROUP_GENERIC, USB_VENDOR_ID_ASUSTEK, USB_DEVICE_ID_ASUSTEK_T101HA_KEYBOARD) }, { } diff --git a/drivers/hid/hid-ids.h b/drivers/hid/hid-ids.h index 53502440403437..c9b2c7782d9a1d 100644 --- a/drivers/hid/hid-ids.h +++ b/drivers/hid/hid-ids.h @@ -228,6 +228,7 @@ #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD 0x1866 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD2 0x19b6 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3 0x1ce6 +#define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_KEYBOARD3_BT 0x1ce7 #define USB_DEVICE_ID_ASUSTEK_ROG_Z13_FOLIO 0x1a30 #define USB_DEVICE_ID_ASUSTEK_ROG_Z13_LIGHTBAR 0x18c6 #define USB_DEVICE_ID_ASUSTEK_ROG_NKEY_ALLY 0x1abe From 397e1aa78b270d2d1a1a43198dc3d54c19140d31 Mon Sep 17 00:00:00 2001 From: Ahmed Yaseen Date: Sat, 22 Aug 2026 16:35:21 +0500 Subject: [PATCH 1010/1012] [FOR-UPSTREAM] HID: asus: map the tent mode key on ROG Zephyrus Duo GX651AR Fn+F12 on the GX651AR keyboard emits ASUS vendor code 0x9c, which asus_input_mapping() does not know about, so asus_event() drops it as unmapped. Map it to KEY_F19. F13 to F18 are already used for ASUS toggles that have no generic keycode. Tested-by: Cymirk Signed-off-by: Ahmed Yaseen (cherry picked from commit 337a811e8f0536e8ff649c533d00bfa973e5618b) (cherry picked from commit a980051bc52de3cea2bf95e56cf6de936da24b57) (cherry picked from commit aafe6db9d256a66ed1fd70aa9a1e6e36d643dac3) (cherry picked from commit 6bc535e4cf27505b1dbd89507d938703a4b20f9e) --- drivers/hid/hid-asus.c | 1 + 1 file changed, 1 insertion(+) diff --git a/drivers/hid/hid-asus.c b/drivers/hid/hid-asus.c index c63e03a876a32f..4112c0fc7aaf97 100644 --- a/drivers/hid/hid-asus.c +++ b/drivers/hid/hid-asus.c @@ -1267,6 +1267,7 @@ static int asus_input_mapping(struct hid_device *hdev, case 0xa6: asus_map_key_clear(KEY_F16); break; /* ROG Ally QAM button */ case 0xa7: asus_map_key_clear(KEY_F17); break; /* ROG Ally ROG long-press */ case 0xa8: asus_map_key_clear(KEY_F18); break; /* ROG Ally ROG long-press-release */ + case 0x9c: asus_map_key_clear(KEY_F19); break; /* Zephyrus Duo tent mode */ default: /* ASUS lazily declares 256 usages, ignore the rest, From cf9935674fc584c44b6efa4c5eb4c3cb486d3a9c Mon Sep 17 00:00:00 2001 From: Denis Benato Date: Sat, 5 Sep 2026 13:39:38 +0000 Subject: [PATCH 1011/1012] [NOT-FOR-UPSTREAM] ogc: linux-unstable: boot smoke test: pin loglevel=7 and dump serial log head on banner failure The smoke test greps the captured serial console log for the kernel banner ("Linux version "), but the banner is printed at KERN_NOTICE while some distro configs default the console to a quieter level: Arch ships CONFIG_CONSOLE_LOGLEVEL_DEFAULT=4, which suppresses notice (and info) messages entirely. The result was a confusing failure mode: the kernel booted to userspace just fine (BOOT_OK marker, printed by init directly on /dev/ttyS0), yet the banner check failed because the console log only carried a few high-priority lines. Pin loglevel=7 on the test kernel command line so every distro kernel logs verbosely enough for the banner to be captured, and replace the useless "^Linux version" grep in the failure path (banner lines are prefixed with a "[ 0.000000] " timestamp, so that grep could never match) with a dump of the first serial console lines. Verified locally against a kernel built with the exact Arch config pipeline: the run failed identically to CI before the change and passes both the bios-pc and uefi-q35 legs after it. (cherry picked from commit af2b9020fdbc50ab19fb22a82ca861adac7dc7dd) (cherry picked from commit 98d39d0b1f5fd301c87b7b67f504a73ba3fbc269) (cherry picked from commit adfb8ee52243077a3c71798a874dbc75b6687e5b) --- .github/packaging/boot-smoke-test.sh | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/.github/packaging/boot-smoke-test.sh b/.github/packaging/boot-smoke-test.sh index 021e8e596998f3..f315029982fd72 100755 --- a/.github/packaging/boot-smoke-test.sh +++ b/.github/packaging/boot-smoke-test.sh @@ -34,6 +34,10 @@ # "Linux version " must appear on every console log. # BOOT_SMOKE_TIMEOUT override the per-boot timeout in seconds # (default: 300 under KVM, 1200 under TCG). +# The kernel command line pins loglevel=7: some distro configs default the +# console to a quieter level (Arch ships CONSOLE_LOGLEVEL_DEFAULT=4), which +# would suppress the KERN_NOTICE boot banner and make the banner check below +# fail on an otherwise perfectly bootable kernel. # Requires (installed by the calling CI job): qemu-system-x86_64, cpio, # gzip, a C compiler (gcc or clang), an OVMF build for the UEFI # leg (packages: ovmf / edk2-ovmf), coreutils (timeout, find). @@ -190,7 +194,7 @@ run_boot() { -display none -monitor none -no-reboot \ -serial "file:$serial" \ -kernel "$IMAGE" -initrd "$WORK/initrd.img" \ - -append "console=ttyS0,115200n8 rdinit=/init panic=-1 nokaslr" \ + -append "console=ttyS0,115200n8 rdinit=/init panic=-1 nokaslr loglevel=7" \ || true # A panic with panic=-1 reboots instantly and -no-reboot makes QEMU # exit, so both "qemu exited by itself" and "timeout killed it" end up @@ -205,7 +209,12 @@ run_boot() { fi if [ -n "${KREL:-}" ] && ! grep -qF "Linux version ${KREL} " "$serial"; then echo "::error::QEMU boot smoke test FAILED in the '$name' configuration: the booted kernel banner does not advertise release '${KREL}'." - grep -m1 "^Linux version" "$serial" || true + # Show what the console actually carried: log lines carry a + # "[ 0.000000] " timestamp prefix, so anchor-free context of + # the early console output is what makes this diagnosable. + echo "----- first 25 lines of the $name guest serial console -----" + head -n 25 "$serial" | sed 's/\r$//' + echo "-------------------------------------------------------------" exit 1 fi echo "[$name] Boot smoke test PASSED: kernel booted to userspace init and powered off. Serial console tail:" From ced900c93ecd58eadce3c3e02b2bc4e7c41d1e4f Mon Sep 17 00:00:00 2001 From: Tiago Silva Date: Sun, 20 Sep 2026 17:59:39 -0300 Subject: [PATCH 1012/1012] platform/x86: ayaneo-ec: Add charge control quirk for AYANEO 2S The AYANEO 2S reports "AYANEO 2S" in DMI_BOARD_NAME. The existing "AYANEO 2" entry uses DMI_MATCH, a substring match, so it also matches the 2S. That entry selects quirk_fan, so the board gets fan control but charge control is never registered: probe() skips the battery hook and no charge_behaviour attribute appears on the battery. The failure is silent: no error path is involved, only a false condition, so nothing is logged. The only symptom is the missing sysfs attribute. The EC charge register works on this board. Tested on an AYANEO 2S (BIOS 2.15_S20, 7840U) by adding the entry below and using the resulting attribute: # echo inhibit-charge > /sys/class/power_supply/BAT0/charge_behaviour (~20s later) # cat /sys/class/power_supply/BAT0/power_now ; cat .../status 0 Not charging # echo auto > /sys/class/power_supply/BAT0/charge_behaviour (~20s later) 22903000 Charging The EC does not act on the write immediately. In testing it took anywhere from a few seconds to about a minute. Narrow the "AYANEO 2" entry to DMI_EXACT_MATCH and add an entry for "AYANEO 2S" selecting quirk_charge_limit, which enables fan control as well. "AYANEO 2" and "AYANEO 2S" are the only board names in that family, so the exact match does not drop fan control from any other board. The out-of-tree ShadowBlip ayaneo-platform driver made the same distinction: it matched "AYANEO 2S" with DMI_EXACT_MATCH as a model of its own, separate from "AYANEO 2", and listed only the 2S in its bypass-charge switch. Assisted-by: LLM Signed-off-by: Tiago Silva --- drivers/platform/x86/ayaneo-ec.c | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/drivers/platform/x86/ayaneo-ec.c b/drivers/platform/x86/ayaneo-ec.c index 41a24e09124869..8b2e28d523bfee 100644 --- a/drivers/platform/x86/ayaneo-ec.c +++ b/drivers/platform/x86/ayaneo-ec.c @@ -79,10 +79,17 @@ static const struct dmi_system_id dmi_table[] = { { .matches = { DMI_MATCH(DMI_BOARD_VENDOR, "AYANEO"), - DMI_MATCH(DMI_BOARD_NAME, "AYANEO 2"), + DMI_EXACT_MATCH(DMI_BOARD_NAME, "AYANEO 2"), }, .driver_data = (void *)&quirk_fan, }, + { + .matches = { + DMI_MATCH(DMI_BOARD_VENDOR, "AYANEO"), + DMI_EXACT_MATCH(DMI_BOARD_NAME, "AYANEO 2S"), + }, + .driver_data = (void *)&quirk_charge_limit, + }, { .matches = { DMI_MATCH(DMI_BOARD_VENDOR, "AYANEO"),